library(ncaahoopR)
## Loading required package: dplyr
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
## Loading required package: ggplot2
library(ggplot2)
library(ggimage)
library(dplyr)
library(networkD3)
dive_2000 <- read.csv("/Users/monkey_hou/Library/Mobile Documents/com~apple~CloudDocs/Temple/2023 Spring/analytics/week11/Diving2000_mod.csv")
mean(dive_2000$Difficulty)
## [1] 2.763141
median(dive_2000$Difficulty)
## [1] 3
quantile(dive_2000$Difficulty)
## 0% 25% 50% 75% 100%
## 1.5 2.6 3.0 3.1 3.8
mean(dive_2000$AvgScore)
## [1] 6.832576
median(dive_2000$AvgScore)
## [1] 7.071429
quantile(dive_2000$AvgScore)
## 0% 25% 50% 75% 100%
## 0.8571429 6.1428571 7.0714286 7.7857143 9.9285714
difficulty <- dive_2000 %>%
count(Difficulty<= 2.0, 2.0<Difficulty & Difficulty<=2.5, 2.5<Difficulty & Difficulty<=3.0, 3.0<Difficulty & Difficulty<=3.5, 3.5<Difficulty)%>%
select(n)%>%
mutate("difficulty" = c("3.5⬆️","3.0-3.5","2.6-3.0","2.1-2.5","2.0⬇️"))
difficulty
## n difficulty
## 1 42 3.5⬆️
## 2 3122 3.0-3.5
## 3 4935 2.6-3.0
## 4 1029 2.1-2.5
## 5 1659 2.0⬇️
score <- dive_2000 %>%
count(AvgScore<= 5.0, 5.0<AvgScore & AvgScore<=6.5, 6.5<AvgScore & AvgScore<=7.5, 7.5<AvgScore & AvgScore<=8.5, 8.5<AvgScore)%>%
select(n)%>%
mutate("difficulty" = c("8.5⬆️","7.5-8.5","6.5-7.5","5.0-6.5","5.0⬇️"))
score
## n difficulty
## 1 672 8.5⬆️
## 2 3024 7.5-8.5
## 3 3556 6.5-7.5
## 4 2282 5.0-6.5
## 5 1253 5.0⬇️
dive_2000<-dive_2000 %>%
mutate("Difficulty Range" = case_when(Difficulty<=2.0 ~"2.0⬇️",2.0<Difficulty & Difficulty<=2.5 ~ "2.1-2.5", 2.5<Difficulty & Difficulty<=3.0~"2.6-3.0", 3.0<Difficulty & Difficulty<=3.5~"3.0-3.5", 3.5<Difficulty~"3.5⬆️"))%>%
mutate("Avg Score Range"= case_when(AvgScore<= 5.0~"5.0⬇️", 5.0<AvgScore & AvgScore<=6.5~"5.0-6.5", 6.5<AvgScore & AvgScore<=7.5~"6.5-7.5", 7.5<AvgScore & AvgScore<=8.5~"7.5-8.5", 8.5<AvgScore~"8.5⬆️"))%>%
mutate("all"=c("all diver"))
dive_2000%>%
group_by(`Difficulty Range`,`Avg Score Range`)
## # A tibble: 10,787 × 14
## # Groups: Difficulty Range, Avg Score Range [24]
## Event Round Diver Country Rank DiveNo Difficulty JScore Judge JCountry
## <chr> <chr> <chr> <chr> <int> <int> <dbl> <dbl> <chr> <chr>
## 1 M3mSB Final XIONG Ni CHN 1 1 3.1 8 RUIZ-PE… CUB
## 2 M3mSB Final XIONG Ni CHN 1 1 3.1 9 GEAR De… NZL
## 3 M3mSB Final XIONG Ni CHN 1 1 3.1 8.5 BOYS Be… CAN
## 4 M3mSB Final XIONG Ni CHN 1 1 3.1 8.5 JOHNSON… NOR
## 5 M3mSB Final XIONG Ni CHN 1 1 3.1 8.5 BOUSSAR… FRA
## 6 M3mSB Final XIONG Ni CHN 1 1 3.1 8.5 CALDERO… PUR
## 7 M3mSB Final XIONG Ni CHN 1 1 3.1 8.5 CRUZ Ju… ESP
## 8 M3mSB Final XIONG Ni CHN 1 2 3 8.5 RUIZ-PE… CUB
## 9 M3mSB Final XIONG Ni CHN 1 2 3 8 GEAR De… NZL
## 10 M3mSB Final XIONG Ni CHN 1 2 3 8 BOYS Be… CAN
## # … with 10,777 more rows, and 4 more variables: AvgScore <dbl>,
## # `Difficulty Range` <chr>, `Avg Score Range` <chr>, all <chr>
networkdata1<-dive_2000 %>% select(all,`Difficulty Range`,`Avg Score Range`)%>%group_by(all,`Difficulty Range`)%>%count
networkdata2<-dive_2000 %>% select(all,`Difficulty Range`,`Avg Score Range`)%>%group_by(`Difficulty Range`,`Avg Score Range`)%>%count
netdf1 <-networkdata1 %>% ungroup()%>%select("source"= all,"target"=`Difficulty Range`,"value"= n)
netdf2 <-networkdata2 %>% ungroup()%>%select("source"= `Difficulty Range`,"target"= `Avg Score Range`,"value"=n)
network<-rbind(netdf1,netdf2)
nodes <- data.frame(name=c(as.character(dive_2000$`Difficulty Range`), as.character(dive_2000$`Avg Score Range`),as.character(dive_2000$all)))%>% unique()
network$IDsource <- match(network$source, nodes$name)-1
network$IDtarget <- match(network$target, nodes$name)-1
p <- sankeyNetwork(Links = network, Nodes = nodes, Source = "IDsource", Target = "IDtarget",Value = "value", NodeID = "name", fontSize = 14, nodeWidth = 10)
## Links is a tbl_df. Converting to a plain data frame.
p <- htmlwidgets::prependContent(p, htmltools::tags$h3("2000 Olympics diving Difficulty/Average Score network graph"))
p <- htmlwidgets::prependContent(p, htmltools::tags$p("@PeterHou, Data: Diving2000_mod"))
htmlwidgets::onRender(p, '
function(el) {
var cols_x = this.sankey.nodes().map(d => d.x).filter((v, i, a) => a.indexOf(v) === i).sort(function(a, b){return a - b});
var labels = ["All Divers", "Difficulty", "Avg Scores"];
cols_x.forEach((d, i) => {
d3.select(el).select("svg")
.append("text")
.attr("x", d)
.attr("y", 12)
.attr("font-size", 12)
.text(labels[i]);
})
}
')
@PeterHou, Data: Diving2000_mod