Library Setup

library(rvest)
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(stringr)
library(ggplot2)

Read Data from Single Web Page - Kaufman Example

name <- paste0("https://www.ncbi.nlm.nih.gov/biosample/?term=",31280775) # generic website + specific ID

test <- read_html(name)

table <- test %>%
  html_node("table") %>%
  html_table()

names(table) <- c("Questions", "Response")

head(table)
## # A tibble: 6 × 2
##   Questions             Response         
##   <chr>                 <chr>            
## 1 dominant hand         I am right handed
## 2 environmental medium  feces            
## 3 environmental package human-gut        
## 4 host body habitat     UBERON:feces     
## 5 host body mass index  33.3             
## 6 host body product     UBERON:feces

Read Data from Single Web Page - Jason (Biosample 18160416)

URLname <- paste0("https://www.ncbi.nlm.nih.gov/biosample/?term=",18160416) # generic website + specific ID

URLtest <- read_html(URLname)

text <- URLtest %>%
  html_node("table") %>%
  html_table()

names(text) <- c("Questions", "Response")

head(text)
## # A tibble: 6 × 2
##   Questions           Response         
##   <chr>               <chr>            
## 1 body habitat        UBERON:feces     
## 2 body product        UBERON:feces     
## 3 tissue              UBERON:feces     
## 4 geographic location United Kingdom   
## 5 diet                not provided     
## 6 dominant hand       I am right handed

Read Data frum Multiple Pages

Creating a list of URLs to read in

URL <- c("https://www.ncbi.nlm.nih.gov/biosample/?term=31280770",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280771",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280772",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280773",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280774",
"https://www.ncbi.nlm.nih.gov/biosample/?term=31280775"
)

Creating a for loop to read through each URL

df_list <- list() # important for loops, creating a variable to save the loop outputs in

for (i in seq_along(URL)) { #for each sequence in the URL df
  df_list[[i]] <- read_html(URL[i]) %>% # add to empty list the read html function
    html_element("table") %>% # specifying to look for a table
    html_table() # and read in as a table
}

Renaming the columns of the dataframes

df_list <-lapply(df_list, function(column) { # specifying that for each element in the list (so each dataframe)
  names(column) <- c("Question", "Answer") # rename the columns to Question and Answer
  column # needed to keep the dataframe with the renaming
})
df_list
## [[1]]
## # A tibble: 230 × 2
##    Question              Answer           
##    <chr>                 <chr>            
##  1 dominant hand         I am right handed
##  2 environmental medium  feces            
##  3 environmental package human-gut        
##  4 host body habitat     UBERON:feces     
##  5 host body mass index  21.0             
##  6 host body product     UBERON:feces     
##  7 host tissue sampled   UBERON:feces     
##  8 host height           162.0            
##  9 life stage            Adult            
## 10 race                  Hispanic         
## # ℹ 220 more rows
## 
## [[2]]
## # A tibble: 295 × 2
##    Question              Answer           
##    <chr>                 <chr>            
##  1 dominant hand         I am right handed
##  2 environmental medium  feces            
##  3 environmental package human-gut        
##  4 host body habitat     UBERON:feces     
##  5 host body mass index  26.0             
##  6 host body product     UBERON:feces     
##  7 host tissue sampled   UBERON:feces     
##  8 host height           184.0            
##  9 life stage            Adult            
## 10 race                  Hispanic         
## # ℹ 285 more rows
## 
## [[3]]
## # A tibble: 293 × 2
##    Question              Answer          
##    <chr>                 <chr>           
##  1 dominant hand         I am left handed
##  2 environmental medium  feces           
##  3 environmental package human-gut       
##  4 host body habitat     UBERON:feces    
##  5 host body mass index  25.7            
##  6 host body product     UBERON:feces    
##  7 host tissue sampled   UBERON:feces    
##  8 host height           159.0           
##  9 life stage            Adult           
## 10 race                  Hispanic        
## # ℹ 283 more rows
## 
## [[4]]
## # A tibble: 295 × 2
##    Question              Answer          
##    <chr>                 <chr>           
##  1 dominant hand         I am left handed
##  2 environmental medium  feces           
##  3 environmental package human-gut       
##  4 host body habitat     UBERON:feces    
##  5 host body mass index  25.4            
##  6 host body product     UBERON:feces    
##  7 host tissue sampled   UBERON:feces    
##  8 host height           166.0           
##  9 life stage            Adult           
## 10 race                  Hispanic        
## # ℹ 285 more rows
## 
## [[5]]
## # A tibble: 230 × 2
##    Question              Answer           
##    <chr>                 <chr>            
##  1 dominant hand         I am right handed
##  2 environmental medium  feces            
##  3 environmental package human-gut        
##  4 host body habitat     UBERON:feces     
##  5 host body mass index  20.4             
##  6 host body product     UBERON:feces     
##  7 host tissue sampled   UBERON:feces     
##  8 host height           158.0            
##  9 life stage            Adult            
## 10 race                  Hispanic         
## # ℹ 220 more rows
## 
## [[6]]
## # A tibble: 293 × 2
##    Question              Answer           
##    <chr>                 <chr>            
##  1 dominant hand         I am right handed
##  2 environmental medium  feces            
##  3 environmental package human-gut        
##  4 host body habitat     UBERON:feces     
##  5 host body mass index  33.3             
##  6 host body product     UBERON:feces     
##  7 host tissue sampled   UBERON:feces     
##  8 host height           154.0            
##  9 life stage            Adult            
## 10 race                  Hispanic         
## # ℹ 283 more rows

Creating a piechart

First we have to filter the dataframes for the answers we wish to plot

question <- "sex" # create the question we want the answers to from the dfs
bind <- bind_rows(df_list, .id = "source") %>% # bind the question and answer together and creating a new columm with .id to label which answers came from which dataframes
  filter(Question == question) # filtering the Question column by the question we wish to find all the answers to established with 'question' list

Now we have a ‘bind’ dataframe with the sex of the six participants in our list of dataframes. Now to create the piechart, we first start by making a bar chart. To make labeling the pie chart later easier, this will be build with a dataframe with the sum count of each answer:

bind_sum <- bind %>%
  count(Answer)

pie = ggplot(bind_sum, aes(x = "", y = n, fill = Answer)) +
  geom_bar(stat = "identity", width = 1)
pie

Next we convert the bar chart to a pie chart using polar coordinates and adding labels:

pie = pie + coord_polar("y", start = 0) +
  geom_text(aes(label = Answer), position = position_stack(vjust = 0.5)) +
  labs(x = NULL, y = NULL, fill = "Sex", title = "Sex of Patients Within Study") +
  theme_classic() + theme(axis.line = element_blank(),
          axis.text = element_blank(),
          axis.ticks = element_blank(),
          plot.title = element_text(hjust = 0.5, color = "#564"))
pie