## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.2.1 ✔ readr 2.2.0
## ✔ forcats 1.0.1 ✔ stringr 1.6.0
## ✔ ggplot2 4.0.3 ✔ tibble 3.3.1
## ✔ lubridate 1.9.5 ✔ tidyr 1.3.2
## ✔ purrr 1.2.2
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
# Simply read in text lines to preserve native document state (txt files can have different types of sep)
biosample <- readLines("biosample_result (1).txt")
# View a portion of txt file to get an idea of how it's structured
head(biosample, 50)
## [1] "1: american gut project; 10317.X00185902\t"
## [2] "Identifiers: BioSample: SAMEA112990854; SRA: ERS14985877\t"
## [3] "Organism: human gut metagenome\t"
## [4] "Attributes:\t"
## [5] " /ENA-CHECKLIST\tERC000011"
## [6] " /ENA-FIRST-PUBLIC\t4/28/2023"
## [7] " /ENA-LAST-UPDATE\t4/28/2023"
## [8] " /External Id\tSAMEA112990854"
## [9] " /INSDC center alias\tUCSDMI"
## [10] " /INSDC center name\tUniversity of California San Diego Microbiome Initiative"
## [11] " /INSDC first public\t2023-04-28T16:20:26Z"
## [12] " /INSDC last update\t2023-04-28T16:20:26Z"
## [13] " /INSDC status\tpublic"
## [14] " /Submitter Id\tqiita_sid_10317:10317.X00185902"
## [15] " /acid_reflux\ti do not have this condition"
## [16] " /acne_medication\tno"
## [17] " /acne_medication_otc\tno"
## [18] " /add_adhd\ti do not have this condition"
## [19] " /age_cat\t30s"
## [20] " /alcohol_consumption\tyes"
## [21] " /alcohol_frequency\toccasionally (1-2 times/week)"
## [22] " /alcohol_types_beercider\tTRUE"
## [23] " /alcohol_types_red_wine\tTRUE"
## [24] " /alcohol_types_sour_beers\tTRUE"
## [25] " /alcohol_types_spiritshard_alcohol\tTRUE"
## [26] " /alcohol_types_unspecified\tFALSE"
## [27] " /alcohol_types_white_wine\tFALSE"
## [28] " /allergic_to_i_have_no_food_allergies_that_i_know_of\tFALSE"
## [29] " /allergic_to_peanuts\tFALSE"
## [30] " /allergic_to_shellfish\tFALSE"
## [31] " /allergic_to_tree_nuts\tFALSE"
## [32] " /allergic_to_unspecified\tFALSE"
## [33] " /alzheimers\ti do not have this condition"
## [34] " /anonymized_name\tX00185902"
## [35] " /antibiotic_history\tyear"
## [36] " /appendix_removed\tno"
## [37] " /artificial_sweeteners\trarely (a few times/month)"
## [38] " /asd\ti do not have this condition"
## [39] " /autoimmune\ti do not have this condition"
## [40] " /beet_frequency\trarely (less than once/week)"
## [41] " /birth_year\t1990"
## [42] " /bmi_cat\toverweight"
## [43] " /bowel_movement_frequency\tone"
## [44] " /bowel_movement_quality\ti tend to have normal formed stool - type 3 and 4"
## [45] " /breastmilk_formula_ensure\tno"
## [46] " /cancer\ti do not have this condition"
## [47] " /cardiovascular_disease\ti do not have this condition"
## [48] " /cat\tno"
## [49] " /cdiff\ti do not have this condition"
## [50] " /chickenpox\tyes"
# Remove / found on every line in the second column
biosample <- sub("/", "", biosample)
# Make a data frame from the above tidying step
biosample <- read.delim(text = biosample, header = TRUE)
# Check to see if this tidying was successful
head(biosample, 50)
## X1..american.gut.project..10317.X00185902
## 1 Identifiers: BioSample: SAMEA112990854; SRA: ERS14985877
## 2 Organism: human gut metagenome
## 3 Attributes:
## 4 ENA-CHECKLIST
## 5 ENA-FIRST-PUBLIC
## 6 ENA-LAST-UPDATE
## 7 External Id
## 8 INSDC center alias
## 9 INSDC center name
## 10 INSDC first public
## 11 INSDC last update
## 12 INSDC status
## 13 Submitter Id
## 14 acid_reflux
## 15 acne_medication
## 16 acne_medication_otc
## 17 add_adhd
## 18 age_cat
## 19 alcohol_consumption
## 20 alcohol_frequency
## 21 alcohol_types_beercider
## 22 alcohol_types_red_wine
## 23 alcohol_types_sour_beers
## 24 alcohol_types_spiritshard_alcohol
## 25 alcohol_types_unspecified
## 26 alcohol_types_white_wine
## 27 allergic_to_i_have_no_food_allergies_that_i_know_of
## 28 allergic_to_peanuts
## 29 allergic_to_shellfish
## 30 allergic_to_tree_nuts
## 31 allergic_to_unspecified
## 32 alzheimers
## 33 anonymized_name
## 34 antibiotic_history
## 35 appendix_removed
## 36 artificial_sweeteners
## 37 asd
## 38 autoimmune
## 39 beet_frequency
## 40 birth_year
## 41 bmi_cat
## 42 bowel_movement_frequency
## 43 bowel_movement_quality
## 44 breastmilk_formula_ensure
## 45 cancer
## 46 cardiovascular_disease
## 47 cat
## 48 cdiff
## 49 chickenpox
## 50 clinical_condition
## X
## 1
## 2
## 3
## 4 ERC000011
## 5 4/28/2023
## 6 4/28/2023
## 7 SAMEA112990854
## 8 UCSDMI
## 9 University of California San Diego Microbiome Initiative
## 10 2023-04-28T16:20:26Z
## 11 2023-04-28T16:20:26Z
## 12 public
## 13 qiita_sid_10317:10317.X00185902
## 14 i do not have this condition
## 15 no
## 16 no
## 17 i do not have this condition
## 18 30s
## 19 yes
## 20 occasionally (1-2 times/week)
## 21 TRUE
## 22 TRUE
## 23 TRUE
## 24 TRUE
## 25 FALSE
## 26 FALSE
## 27 FALSE
## 28 FALSE
## 29 FALSE
## 30 FALSE
## 31 FALSE
## 32 i do not have this condition
## 33 X00185902
## 34 year
## 35 no
## 36 rarely (a few times/month)
## 37 i do not have this condition
## 38 i do not have this condition
## 39 rarely (less than once/week)
## 40 1990
## 41 overweight
## 42 one
## 43 i tend to have normal formed stool - type 3 and 4
## 44 no
## 45 i do not have this condition
## 46 i do not have this condition
## 47 no
## 48 i do not have this condition
## 49 yes
## 50 i do not have this condition
# Rename column headers to Attributes and Responses
biosample <- biosample %>% rename(Attributes = X1..american.gut.project..10317.X00185902, Responses = X)
# Check to see if renaming was successful
names(biosample)
## [1] "Attributes" "Responses"
# Remove extraneous metadata that can be found elsewhere (i.e. page where data was downloaded from)
biosample_new <- biosample[-c(1, 2, 3, 328, 329, 330), ]
# Check to see if data frame was modified correctly
head(biosample_new)
## Attributes
## 4 ENA-CHECKLIST
## 5 ENA-FIRST-PUBLIC
## 6 ENA-LAST-UPDATE
## 7 External Id
## 8 INSDC center alias
## 9 INSDC center name
## Responses
## 4 ERC000011
## 5 4/28/2023
## 6 4/28/2023
## 7 SAMEA112990854
## 8 UCSDMI
## 9 University of California San Diego Microbiome Initiative
tail(biosample_new)
## Attributes Responses
## 322 vitamin_b_supplement_frequency never
## 323 vitamin_d_supplement_frequency never
## 324 vivid_dreams rarely (a few times/month)
## 325 weight_change increased more than 10 pounds
## 326 whole_eggs rarely (less than once/week)
## 327 whole_grain_frequency occasionally (1-2 times/week)
# Write csv to name it after BioSample identifier to make up for some metadata removed during tidying
SAMEA112990854 <- write.csv(biosample_new, file = "SAMEA112990854.csv", row.names = FALSE)