library(rvest)
library(dplyr)
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library(stringr)
library(ggplot2)
name <- paste0("https://www.ncbi.nlm.nih.gov/biosample/?term=",31280775) # generic website + specific ID
test <- read_html(name)
table <- test %>%
html_node("table") %>%
html_table()
names(table) <- c("Questions", "Response")
head(table)
## # A tibble: 6 × 2
## Questions Response
## <chr> <chr>
## 1 dominant hand I am right handed
## 2 environmental medium feces
## 3 environmental package human-gut
## 4 host body habitat UBERON:feces
## 5 host body mass index 33.3
## 6 host body product UBERON:feces
URLname <- paste0("https://www.ncbi.nlm.nih.gov/biosample/?term=",18160416) # generic website + specific ID
URLtest <- read_html(URLname)
text <- URLtest %>%
html_node("table") %>%
html_table()
names(text) <- c("Questions", "Response")
head(text)
## # A tibble: 6 × 2
## Questions Response
## <chr> <chr>
## 1 body habitat UBERON:feces
## 2 body product UBERON:feces
## 3 tissue UBERON:feces
## 4 geographic location United Kingdom
## 5 diet not provided
## 6 dominant hand I am right handed
Creating a list of URLs to read in
URL <- c("https://www.ncbi.nlm.nih.gov/biosample/?term=31280770",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280771",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280772",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280773",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280774",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280775"
)
Creating a for loop to read through each URL
df_list <- list() # important for loops, creating a variable to save the loop outputs in
for (i in seq_along(URL)) { #for each sequence in the URL df
df_list[[i]] <- read_html(URL[i]) %>% # add to empty list the read html function
html_element("table") %>% # specifying to look for a table
html_table() # and read in as a table
}
Renaming the columns of the dataframes
df_list <-lapply(df_list, function(column) { # specifying that for each element in the list (so each dataframe)
names(column) <- c("Question", "Answer") # rename the columns to Question and Answer
column # needed to keep the dataframe with the renaming
})
df_list
## [[1]]
## # A tibble: 230 × 2
## Question Answer
## <chr> <chr>
## 1 dominant hand I am right handed
## 2 environmental medium feces
## 3 environmental package human-gut
## 4 host body habitat UBERON:feces
## 5 host body mass index 21.0
## 6 host body product UBERON:feces
## 7 host tissue sampled UBERON:feces
## 8 host height 162.0
## 9 life stage Adult
## 10 race Hispanic
## # ℹ 220 more rows
##
## [[2]]
## # A tibble: 295 × 2
## Question Answer
## <chr> <chr>
## 1 dominant hand I am right handed
## 2 environmental medium feces
## 3 environmental package human-gut
## 4 host body habitat UBERON:feces
## 5 host body mass index 26.0
## 6 host body product UBERON:feces
## 7 host tissue sampled UBERON:feces
## 8 host height 184.0
## 9 life stage Adult
## 10 race Hispanic
## # ℹ 285 more rows
##
## [[3]]
## # A tibble: 293 × 2
## Question Answer
## <chr> <chr>
## 1 dominant hand I am left handed
## 2 environmental medium feces
## 3 environmental package human-gut
## 4 host body habitat UBERON:feces
## 5 host body mass index 25.7
## 6 host body product UBERON:feces
## 7 host tissue sampled UBERON:feces
## 8 host height 159.0
## 9 life stage Adult
## 10 race Hispanic
## # ℹ 283 more rows
##
## [[4]]
## # A tibble: 295 × 2
## Question Answer
## <chr> <chr>
## 1 dominant hand I am left handed
## 2 environmental medium feces
## 3 environmental package human-gut
## 4 host body habitat UBERON:feces
## 5 host body mass index 25.4
## 6 host body product UBERON:feces
## 7 host tissue sampled UBERON:feces
## 8 host height 166.0
## 9 life stage Adult
## 10 race Hispanic
## # ℹ 285 more rows
##
## [[5]]
## # A tibble: 230 × 2
## Question Answer
## <chr> <chr>
## 1 dominant hand I am right handed
## 2 environmental medium feces
## 3 environmental package human-gut
## 4 host body habitat UBERON:feces
## 5 host body mass index 20.4
## 6 host body product UBERON:feces
## 7 host tissue sampled UBERON:feces
## 8 host height 158.0
## 9 life stage Adult
## 10 race Hispanic
## # ℹ 220 more rows
##
## [[6]]
## # A tibble: 293 × 2
## Question Answer
## <chr> <chr>
## 1 dominant hand I am right handed
## 2 environmental medium feces
## 3 environmental package human-gut
## 4 host body habitat UBERON:feces
## 5 host body mass index 33.3
## 6 host body product UBERON:feces
## 7 host tissue sampled UBERON:feces
## 8 host height 154.0
## 9 life stage Adult
## 10 race Hispanic
## # ℹ 283 more rows
First we have to filter the dataframes for the answers we wish to plot
question <- "sex" # create the question we want the answers to from the dfs
bind <- bind_rows(df_list, .id = "source") %>% # bind the question and answer together and creating a new columm with .id to label which answers came from which dataframes
filter(Question == question) # filtering the Question column by the question we wish to find all the answers to established with 'question' list
Now we have a ‘bind’ dataframe with the sex of the six participants in our list of dataframes. Now to create the piechart, we first start by making a bar chart. To make labeling the pie chart later easier, this will be build with a dataframe with the sum count of each answer:
bind_sum <- bind %>%
count(Answer)
pie = ggplot(bind_sum, aes(x = "", y = n, fill = Answer)) +
geom_bar(stat = "identity", width = 1)
pie
Next we convert the bar chart to a pie chart using polar coordinates and adding labels:
pie = pie + coord_polar("y", start = 0) +
geom_text(aes(label = Answer), position = position_stack(vjust = 0.5)) +
labs(x = NULL, y = NULL, fill = "Sex", title = "Sex of Patients Within Study") +
theme_classic() + theme(axis.line = element_blank(),
axis.text = element_blank(),
axis.ticks = element_blank(),
plot.title = element_text(hjust = 0.5, color = "#564"))
pie