Imports

setwd("C:/Users/marc-/Documents/OneDrive/Documents/Travail/Stage IJN/romance_study_1.2")
library(dplyr)
## Warning: le package 'dplyr' a été compilé avec la version R 4.3.1
## 
## Attachement du package : 'dplyr'
## Les objets suivants sont masqués depuis 'package:stats':
## 
##     filter, lag
## Les objets suivants sont masqués depuis 'package:base':
## 
##     intersect, setdiff, setequal, union
library("readxl")
## Warning: le package 'readxl' a été compilé avec la version R 4.3.1
library(readr)
## Warning: le package 'readr' a été compilé avec la version R 4.3.1
library(stringr)
## Warning: le package 'stringr' a été compilé avec la version R 4.3.3
library(ggplot2)
## Warning: le package 'ggplot2' a été compilé avec la version R 4.3.2
new_n <- read.csv("new_n.csv")
keywords <- read_csv("keyword_transformed.csv")
## Warning: One or more parsing issues, call `problems()` on your data frame for details,
## e.g.:
##   dat <- vroom(...)
##   problems(dat)
## Rows: 9872 Columns: 3923
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr    (2): imaginary_world, psychotronic-film
## dbl (3921): imdb, escape-from-prison, prison, voice-over-narration, suicide-...
## 
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
# Retrieving the imdb ID column after import issues:
keywords <- keywords %>%
  select(-imdb) %>%
  rename(c("imdb" = "imaginary_world"))

# Scores datasets
parenting_movies <- read_csv("PARENTING_MOVIES_PROGRESS.csv")
## Rows: 3561 Columns: 229
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (192): imdb, wiki, genre_wiki, country_wiki, language_wiki, article_wiki...
## dbl  (37): ...1, latvia, israel, panama, netherlandsantilles, federalrepubli...
## 
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
paternal_movies <- read_csv("PATERNAL_MOVIES_PROGRESS.csv")
## Rows: 3561 Columns: 229
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (192): imdb, wiki, genre_wiki, country_wiki, language_wiki, article_wiki...
## dbl  (37): ...1, latvia, israel, panama, netherlandsantilles, federalrepubli...
## 
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
longterm_movies <- read_csv("LONGTERM_MOVIES_PROGRESS.csv")
## Rows: 3561 Columns: 229
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (192): imdb, wiki, genre_wiki, country_wiki, language_wiki, article_wiki...
## dbl  (37): ...1, latvia, israel, panama, netherlandsantilles, federalrepubli...
## 
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.

Data wrangling

Scores extraction

# PARENTING
# Rename the column with explanations + scores
parenting_movies <- parenting_movies %>% 
  rename(
    "parenting_exp" = "parenting_score"
  )

# Extract the scores in a dedicated column

parenting_movies <- parenting_movies %>%
  mutate(parenting_score = str_extract(parenting_exp, "SCORE = \\d+") %>%
           str_extract("\\d+") %>%
           as.numeric())

print(colnames(parenting_movies))
##   [1] "...1"                         "imdb"                        
##   [3] "wiki"                         "genre_wiki"                  
##   [5] "country_wiki"                 "language_wiki"               
##   [7] "article_wiki"                 "page_wiki"                   
##   [9] "plot_wiki"                    "title"                       
##  [11] "year"                         "date_published"              
##  [13] "genre_imdb"                   "duration"                    
##  [15] "country_imdb"                 "language_imdb"               
##  [17] "director"                     "writer"                      
##  [19] "production_company"           "actors"                      
##  [21] "description"                  "avg_vote"                    
##  [23] "votes"                        "usa_gross_income"            
##  [25] "worlwide_gross_income"        "metascore"                   
##  [27] "reviews_from_users"           "reviews_from_critics"        
##  [29] "age"                          "gender"                      
##  [31] "relation"                     "ope"                         
##  [33] "con"                          "ext"                         
##  [35] "agr"                          "neu"                         
##  [37] "plot_wiki_cleaned"            "drama"                       
##  [39] "fantasy"                      "musical"                     
##  [41] "romance"                      "action"                      
##  [43] "comedy"                       "thriller"                    
##  [45] "biography"                    "family"                      
##  [47] "sci_fi"                       "crime"                       
##  [49] "history"                      "sport"                       
##  [51] "adventure"                    "mystery"                     
##  [53] "war"                          "horror"                      
##  [55] "music"                        "animation"                   
##  [57] "western"                      "reality_tv"                  
##  [59] "film_noir"                    "documentary"                 
##  [61] "india"                        "thailand"                    
##  [63] "bangladesh"                   "sweden"                      
##  [65] "france"                       "australia"                   
##  [67] "usa"                          "uk"                          
##  [69] "austria"                      "china"                       
##  [71] "switzerland"                  "canada"                      
##  [73] "unitedarabemirates"           "japan"                       
##  [75] "qatar"                        "fiji"                        
##  [77] "pakistan"                     "iran"                        
##  [79] "netherlands"                  "germany"                     
##  [81] "southafrica"                  "nepal"                       
##  [83] "eastgermany"                  "sovietunion"                 
##  [85] "hungary"                      "kenya"                       
##  [87] "turkey"                       "uganda"                      
##  [89] "srilanka"                     "namibia"                     
##  [91] "denmark"                      "singapore"                   
##  [93] "luxembourg"                   "yugoslavia"                  
##  [95] "indonesia"                    "czechrepublic"               
##  [97] "slovakia"                     "czechoslovakia"              
##  [99] "belgium"                      "serbia"                      
## [101] "poland"                       "bhutan"                      
## [103] "taiwan"                       "greece"                      
## [105] "newzealand"                   "vanuatu"                     
## [107] "italy"                        "ireland"                     
## [109] "finland"                      "laos"                        
## [111] "colombia"                     "russia"                      
## [113] "hongkong"                     "spain"                       
## [115] "southkorea"                   "estonia"                     
## [117] "britishvirginislands"         "haiti"                       
## [119] "brazil"                       "malaysia"                    
## [121] "iceland"                      "morocco"                     
## [123] "cambodia"                     "argentina"                   
## [125] "portugal"                     "cuba"                        
## [127] "mexico"                       "westgermany"                 
## [129] "chile"                        "uruguay"                     
## [131] "norway"                       "afghanistan"                 
## [133] "philippines"                  "cyprus"                      
## [135] "uzbekistan"                   "venezuela"                   
## [137] "malta"                        "bulgaria"                    
## [139] "peru"                         "monaco"                      
## [141] "guatemala"                    "dominicanrepublic"           
## [143] "romania"                      "vietnam"                     
## [145] "latvia"                       "kazakhstan"                  
## [147] "ukraine"                      "israel"                      
## [149] "belarus"                      "armenia"                     
## [151] "lithuania"                    "republicofnorthmacedonia"    
## [153] "georgia"                      "albania"                     
## [155] "kyrgyzstan"                   "panama"                      
## [157] "netherlandsantilles"          "croatia"                     
## [159] "slovenia"                     "federalrepublicofyugoslavia" 
## [161] "algeria"                      "egypt"                       
## [163] "liechtenstein"                "bosniaandherzegovina"        
## [165] "iraq"                         "jordan"                      
## [167] "palestine"                    "lebanon"                     
## [169] "mongolia"                     "mozambique"                  
## [171] "paraguay"                     "svalbardandjanmayen"         
## [173] "ghana"                        "isleofman"                   
## [175] "burkinafaso"                  "tunisia"                     
## [177] "senegal"                      "cameroon"                    
## [179] "tajikistan"                   "angola"                      
## [181] "martinique"                   "chad"                        
## [183] "mauritania"                   "liberia"                     
## [185] "newcaledonia"                 "thedemocraticrepublicofcongo"
## [187] "greenland"                    "macao"                       
## [189] "côted.ivoire"                 "myanmar"                     
## [191] "rwanda"                       "zaire"                       
## [193] "mali"                         "guinea"                      
## [195] "northkorea"                   "nigeria"                     
## [197] "azerbaijan"                   "malawi"                      
## [199] "zimbabwe"                     "bahrain"                     
## [201] "caymanislands"                "puertorico"                  
## [203] "bahamas"                      "somalia"                     
## [205] "ethiopia"                     "bolivia"                     
## [207] "sudan"                        "bermuda"                     
## [209] "jamaica"                      "aruba"                       
## [211] "libya"                        "botswana"                    
## [213] "gibraltar"                    "trinidadandtobago"           
## [215] "serbiaandmontenegro"          "saudiarabia"                 
## [217] "gabon"                        "kuwait"                      
## [219] "kosovo"                       "zambia"                      
## [221] "nicaragua"                    "ecuador"                     
## [223] "montenegro"                   "guadeloupe"                  
## [225] "tanzania"                     "gen_love"                    
## [227] "gen_love_score"               "year_sqrt"                   
## [229] "parenting_exp"                "parenting_score"
# PATERNAL INVESTMENT
# Rename the column with explanations + scores
paternal_movies <- paternal_movies %>%
  rename(
    "paternal_exp" = "paternal_score"
  )
# Extract the scores in a dedicated column
paternal_movies <- paternal_movies %>%
  mutate(paternal_score = str_extract(paternal_exp, "SCORE = \\d+") %>%
           str_extract("\\d+") %>%
           as.numeric())

print(colnames(paternal_movies))
##   [1] "...1"                         "imdb"                        
##   [3] "wiki"                         "genre_wiki"                  
##   [5] "country_wiki"                 "language_wiki"               
##   [7] "article_wiki"                 "page_wiki"                   
##   [9] "plot_wiki"                    "title"                       
##  [11] "year"                         "date_published"              
##  [13] "genre_imdb"                   "duration"                    
##  [15] "country_imdb"                 "language_imdb"               
##  [17] "director"                     "writer"                      
##  [19] "production_company"           "actors"                      
##  [21] "description"                  "avg_vote"                    
##  [23] "votes"                        "usa_gross_income"            
##  [25] "worlwide_gross_income"        "metascore"                   
##  [27] "reviews_from_users"           "reviews_from_critics"        
##  [29] "age"                          "gender"                      
##  [31] "relation"                     "ope"                         
##  [33] "con"                          "ext"                         
##  [35] "agr"                          "neu"                         
##  [37] "plot_wiki_cleaned"            "drama"                       
##  [39] "fantasy"                      "musical"                     
##  [41] "romance"                      "action"                      
##  [43] "comedy"                       "thriller"                    
##  [45] "biography"                    "family"                      
##  [47] "sci_fi"                       "crime"                       
##  [49] "history"                      "sport"                       
##  [51] "adventure"                    "mystery"                     
##  [53] "war"                          "horror"                      
##  [55] "music"                        "animation"                   
##  [57] "western"                      "reality_tv"                  
##  [59] "film_noir"                    "documentary"                 
##  [61] "india"                        "thailand"                    
##  [63] "bangladesh"                   "sweden"                      
##  [65] "france"                       "australia"                   
##  [67] "usa"                          "uk"                          
##  [69] "austria"                      "china"                       
##  [71] "switzerland"                  "canada"                      
##  [73] "unitedarabemirates"           "japan"                       
##  [75] "qatar"                        "fiji"                        
##  [77] "pakistan"                     "iran"                        
##  [79] "netherlands"                  "germany"                     
##  [81] "southafrica"                  "nepal"                       
##  [83] "eastgermany"                  "sovietunion"                 
##  [85] "hungary"                      "kenya"                       
##  [87] "turkey"                       "uganda"                      
##  [89] "srilanka"                     "namibia"                     
##  [91] "denmark"                      "singapore"                   
##  [93] "luxembourg"                   "yugoslavia"                  
##  [95] "indonesia"                    "czechrepublic"               
##  [97] "slovakia"                     "czechoslovakia"              
##  [99] "belgium"                      "serbia"                      
## [101] "poland"                       "bhutan"                      
## [103] "taiwan"                       "greece"                      
## [105] "newzealand"                   "vanuatu"                     
## [107] "italy"                        "ireland"                     
## [109] "finland"                      "laos"                        
## [111] "colombia"                     "russia"                      
## [113] "hongkong"                     "spain"                       
## [115] "southkorea"                   "estonia"                     
## [117] "britishvirginislands"         "haiti"                       
## [119] "brazil"                       "malaysia"                    
## [121] "iceland"                      "morocco"                     
## [123] "cambodia"                     "argentina"                   
## [125] "portugal"                     "cuba"                        
## [127] "mexico"                       "westgermany"                 
## [129] "chile"                        "uruguay"                     
## [131] "norway"                       "afghanistan"                 
## [133] "philippines"                  "cyprus"                      
## [135] "uzbekistan"                   "venezuela"                   
## [137] "malta"                        "bulgaria"                    
## [139] "peru"                         "monaco"                      
## [141] "guatemala"                    "dominicanrepublic"           
## [143] "romania"                      "vietnam"                     
## [145] "latvia"                       "kazakhstan"                  
## [147] "ukraine"                      "israel"                      
## [149] "belarus"                      "armenia"                     
## [151] "lithuania"                    "republicofnorthmacedonia"    
## [153] "georgia"                      "albania"                     
## [155] "kyrgyzstan"                   "panama"                      
## [157] "netherlandsantilles"          "croatia"                     
## [159] "slovenia"                     "federalrepublicofyugoslavia" 
## [161] "algeria"                      "egypt"                       
## [163] "liechtenstein"                "bosniaandherzegovina"        
## [165] "iraq"                         "jordan"                      
## [167] "palestine"                    "lebanon"                     
## [169] "mongolia"                     "mozambique"                  
## [171] "paraguay"                     "svalbardandjanmayen"         
## [173] "ghana"                        "isleofman"                   
## [175] "burkinafaso"                  "tunisia"                     
## [177] "senegal"                      "cameroon"                    
## [179] "tajikistan"                   "angola"                      
## [181] "martinique"                   "chad"                        
## [183] "mauritania"                   "liberia"                     
## [185] "newcaledonia"                 "thedemocraticrepublicofcongo"
## [187] "greenland"                    "macao"                       
## [189] "côted.ivoire"                 "myanmar"                     
## [191] "rwanda"                       "zaire"                       
## [193] "mali"                         "guinea"                      
## [195] "northkorea"                   "nigeria"                     
## [197] "azerbaijan"                   "malawi"                      
## [199] "zimbabwe"                     "bahrain"                     
## [201] "caymanislands"                "puertorico"                  
## [203] "bahamas"                      "somalia"                     
## [205] "ethiopia"                     "bolivia"                     
## [207] "sudan"                        "bermuda"                     
## [209] "jamaica"                      "aruba"                       
## [211] "libya"                        "botswana"                    
## [213] "gibraltar"                    "trinidadandtobago"           
## [215] "serbiaandmontenegro"          "saudiarabia"                 
## [217] "gabon"                        "kuwait"                      
## [219] "kosovo"                       "zambia"                      
## [221] "nicaragua"                    "ecuador"                     
## [223] "montenegro"                   "guadeloupe"                  
## [225] "tanzania"                     "gen_love"                    
## [227] "gen_love_score"               "year_sqrt"                   
## [229] "paternal_exp"                 "paternal_score"
# LONG-TERM RELATIONSHIPS
# Rename the column with explanations + scores
longterm_movies <- longterm_movies %>%
  rename(
    "longterm_exp" = "longterm_score"
  )
# Extract the scores in a dedicated column
longterm_movies <- longterm_movies %>%
  mutate(longterm_score = str_extract(longterm_exp, "SCORE = \\d+") %>%
           str_extract("\\d+") %>%
           as.numeric())

print(colnames(longterm_movies))
##   [1] "...1"                         "imdb"                        
##   [3] "wiki"                         "genre_wiki"                  
##   [5] "country_wiki"                 "language_wiki"               
##   [7] "article_wiki"                 "page_wiki"                   
##   [9] "plot_wiki"                    "title"                       
##  [11] "year"                         "date_published"              
##  [13] "genre_imdb"                   "duration"                    
##  [15] "country_imdb"                 "language_imdb"               
##  [17] "director"                     "writer"                      
##  [19] "production_company"           "actors"                      
##  [21] "description"                  "avg_vote"                    
##  [23] "votes"                        "usa_gross_income"            
##  [25] "worlwide_gross_income"        "metascore"                   
##  [27] "reviews_from_users"           "reviews_from_critics"        
##  [29] "age"                          "gender"                      
##  [31] "relation"                     "ope"                         
##  [33] "con"                          "ext"                         
##  [35] "agr"                          "neu"                         
##  [37] "plot_wiki_cleaned"            "drama"                       
##  [39] "fantasy"                      "musical"                     
##  [41] "romance"                      "action"                      
##  [43] "comedy"                       "thriller"                    
##  [45] "biography"                    "family"                      
##  [47] "sci_fi"                       "crime"                       
##  [49] "history"                      "sport"                       
##  [51] "adventure"                    "mystery"                     
##  [53] "war"                          "horror"                      
##  [55] "music"                        "animation"                   
##  [57] "western"                      "reality_tv"                  
##  [59] "film_noir"                    "documentary"                 
##  [61] "india"                        "thailand"                    
##  [63] "bangladesh"                   "sweden"                      
##  [65] "france"                       "australia"                   
##  [67] "usa"                          "uk"                          
##  [69] "austria"                      "china"                       
##  [71] "switzerland"                  "canada"                      
##  [73] "unitedarabemirates"           "japan"                       
##  [75] "qatar"                        "fiji"                        
##  [77] "pakistan"                     "iran"                        
##  [79] "netherlands"                  "germany"                     
##  [81] "southafrica"                  "nepal"                       
##  [83] "eastgermany"                  "sovietunion"                 
##  [85] "hungary"                      "kenya"                       
##  [87] "turkey"                       "uganda"                      
##  [89] "srilanka"                     "namibia"                     
##  [91] "denmark"                      "singapore"                   
##  [93] "luxembourg"                   "yugoslavia"                  
##  [95] "indonesia"                    "czechrepublic"               
##  [97] "slovakia"                     "czechoslovakia"              
##  [99] "belgium"                      "serbia"                      
## [101] "poland"                       "bhutan"                      
## [103] "taiwan"                       "greece"                      
## [105] "newzealand"                   "vanuatu"                     
## [107] "italy"                        "ireland"                     
## [109] "finland"                      "laos"                        
## [111] "colombia"                     "russia"                      
## [113] "hongkong"                     "spain"                       
## [115] "southkorea"                   "estonia"                     
## [117] "britishvirginislands"         "haiti"                       
## [119] "brazil"                       "malaysia"                    
## [121] "iceland"                      "morocco"                     
## [123] "cambodia"                     "argentina"                   
## [125] "portugal"                     "cuba"                        
## [127] "mexico"                       "westgermany"                 
## [129] "chile"                        "uruguay"                     
## [131] "norway"                       "afghanistan"                 
## [133] "philippines"                  "cyprus"                      
## [135] "uzbekistan"                   "venezuela"                   
## [137] "malta"                        "bulgaria"                    
## [139] "peru"                         "monaco"                      
## [141] "guatemala"                    "dominicanrepublic"           
## [143] "romania"                      "vietnam"                     
## [145] "latvia"                       "kazakhstan"                  
## [147] "ukraine"                      "israel"                      
## [149] "belarus"                      "armenia"                     
## [151] "lithuania"                    "republicofnorthmacedonia"    
## [153] "georgia"                      "albania"                     
## [155] "kyrgyzstan"                   "panama"                      
## [157] "netherlandsantilles"          "croatia"                     
## [159] "slovenia"                     "federalrepublicofyugoslavia" 
## [161] "algeria"                      "egypt"                       
## [163] "liechtenstein"                "bosniaandherzegovina"        
## [165] "iraq"                         "jordan"                      
## [167] "palestine"                    "lebanon"                     
## [169] "mongolia"                     "mozambique"                  
## [171] "paraguay"                     "svalbardandjanmayen"         
## [173] "ghana"                        "isleofman"                   
## [175] "burkinafaso"                  "tunisia"                     
## [177] "senegal"                      "cameroon"                    
## [179] "tajikistan"                   "angola"                      
## [181] "martinique"                   "chad"                        
## [183] "mauritania"                   "liberia"                     
## [185] "newcaledonia"                 "thedemocraticrepublicofcongo"
## [187] "greenland"                    "macao"                       
## [189] "côted.ivoire"                 "myanmar"                     
## [191] "rwanda"                       "zaire"                       
## [193] "mali"                         "guinea"                      
## [195] "northkorea"                   "nigeria"                     
## [197] "azerbaijan"                   "malawi"                      
## [199] "zimbabwe"                     "bahrain"                     
## [201] "caymanislands"                "puertorico"                  
## [203] "bahamas"                      "somalia"                     
## [205] "ethiopia"                     "bolivia"                     
## [207] "sudan"                        "bermuda"                     
## [209] "jamaica"                      "aruba"                       
## [211] "libya"                        "botswana"                    
## [213] "gibraltar"                    "trinidadandtobago"           
## [215] "serbiaandmontenegro"          "saudiarabia"                 
## [217] "gabon"                        "kuwait"                      
## [219] "kosovo"                       "zambia"                      
## [221] "nicaragua"                    "ecuador"                     
## [223] "montenegro"                   "guadeloupe"                  
## [225] "tanzania"                     "gen_love"                    
## [227] "gen_love_score"               "year_sqrt"                   
## [229] "longterm_exp"                 "longterm_score"

Concatenating all scores df into a single one

parenting_movies_toconcat <- parenting_movies[,c("imdb", "title", "year", "parenting_exp","parenting_score")]
paternal_movies_toconcat <- paternal_movies[,c("paternal_exp","paternal_score")]
longterm_movies_toconcat <- longterm_movies[,c("longterm_exp","longterm_score")]

all_scores <- cbind(parenting_movies_toconcat, paternal_movies_toconcat, longterm_movies_toconcat)

Validity checks

Checking overlap between keywords and new_n movies with these keywords

# Overlap of new_n movies with keyword tagged movies: 1385

overlap_df <- new_n %>% 
  merge(keywords, by="imdb", all.x=FALSE)
print(nrow(overlap_df))
## [1] 1385
# Paternal keywords
father_kw_df <- keywords[, c("imdb", "father", "father-daughter-relationship", "father-son-relationship")]

new_n_father <- new_n %>% 
  merge(father_kw_df, by="imdb", all.x=TRUE) %>% 
  mutate(father_kw = ifelse(father | `father-daughter-relationship` | `father-son-relationship`, TRUE, FALSE)) %>% 
  filter(father_kw == TRUE) 

print(nrow(new_n_father))
## [1] 157
# Parenting keywords
parenting_kw_df <- keywords[,c("imdb", "father", "father-daughter-relationship", "father-son-relationship", "mother", "mother-son-relationship", "mother-daughter-relationship")]

new_n_parenting <- new_n %>% 
  merge(parenting_kw_df, by="imdb", all.x=TRUE) %>% 
  mutate(parenting_kw = ifelse(father | `father-daughter-relationship` | `father-son-relationship`| mother | `mother-son-relationship` | `mother-daughter-relationship`, TRUE, FALSE)) %>% 
  filter(parenting_kw == TRUE) 

print(nrow(new_n_parenting))
## [1] 489
# Long-term relationship keywords
longterm_kw_df <- keywords[,c("imdb", "husband-wife-relationship")]

new_n_longterm <- new_n %>% 
  merge(longterm_kw_df, by="imdb", all.x=TRUE) %>% 
  mutate(longterm_kw = ifelse(`husband-wife-relationship`, TRUE, FALSE)) %>% 
  filter(longterm_kw == TRUE) 

print(nrow(new_n_longterm))
## [1] 105
# Other jealousy keywords

jealousy_kw_df <- keywords[,c("imdb", "jealousy", "cheating-wife", "cheating-husband", "cheating-girlfriend", "cheating")]

new_n_jealousy <- new_n %>% 
  merge(jealousy_kw_df, by="imdb", all.x=TRUE) %>% 
  mutate(jealousy_kw = ifelse(jealousy | `cheating-wife` | `cheating-husband` | `cheating-girlfriend` | cheating, TRUE, FALSE)) %>% 
  filter(jealousy_kw == TRUE) 

print(nrow(new_n_jealousy))
## [1] 80

T-tests

## Parenting
all_scores <- all_scores %>%
  mutate(parenting_kw = imdb %in% new_n_parenting$imdb)

print(sum(all_scores$parenting_kw))
## [1] 489
has_parenting_kw <- all_scores$parenting_score[all_scores$parenting_kw == TRUE]
hasnot_parenting_kw <- all_scores$parenting_score[all_scores$parenting_kw == FALSE]

t.test(has_parenting_kw, hasnot_parenting_kw, mu=0)
## 
##  Welch Two Sample t-test
## 
## data:  has_parenting_kw and hasnot_parenting_kw
## t = 4.3068, df = 643.69, p-value = 1.915e-05
## alternative hypothesis: true difference in means is not equal to 0
## 95 percent confidence interval:
##  0.3003448 0.8037552
## sample estimates:
## mean of x mean of y 
##  3.354906  2.802856
## Paternal investment
all_scores <- all_scores %>%
  mutate(paternal_kw = imdb %in% new_n_father$imdb)

print(sum(all_scores$paternal_kw))
## [1] 157
has_paternal_kw <- all_scores$paternal_score[all_scores$paternal_kw == TRUE]
hasnot_paternal_kw <- all_scores$paternal_score[all_scores$paternal_kw == FALSE]

t.test(has_paternal_kw, hasnot_paternal_kw, mu=0)
## 
##  Welch Two Sample t-test
## 
## data:  has_paternal_kw and hasnot_paternal_kw
## t = -2.3228, df = 158.54, p-value = 0.02146
## alternative hypothesis: true difference in means is not equal to 0
## 95 percent confidence interval:
##  -0.93256468 -0.07544865
## sample estimates:
## mean of x mean of y 
##  1.807143  2.311150
## Long-term relationships
all_scores <- all_scores %>%
  mutate(longterm_kw = imdb %in% new_n_longterm$imdb)

print(sum(all_scores$longterm_kw))
## [1] 105
has_longterm_kw <- all_scores$longterm_score[all_scores$longterm_kw == TRUE]
hasnot_longterm_kw <- all_scores$longterm_score[all_scores$longterm_kw == FALSE]

t.test(has_longterm_kw, hasnot_longterm_kw, mu=0)
## 
##  Welch Two Sample t-test
## 
## data:  has_longterm_kw and hasnot_longterm_kw
## t = -4.3469, df = 110.6, p-value = 3.081e-05
## alternative hypothesis: true difference in means is not equal to 0
## 95 percent confidence interval:
##  -1.3627918 -0.5093394
## sample estimates:
## mean of x mean of y 
##  3.288462  4.224527

Statistical analysis

Functions

# A function for retrieving the AIC
summary_AIC <- function(model) {
  summary_output <- summary(model)
  aic <- AIC(model)
  summary_output$aic <- aic
  print(summary_output)
}

Parenting

Missing values check

sum_parenting_na <- sum(is.na(all_scores$parenting_score))
print(sum_parenting_na)
## [1] 561

Model 1

lm1_par <- lm(parenting_score ~ as.numeric(year), data=parenting_movies) # lm ignores NAs by default
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm1_par)$aic
## 
## Call:
## lm(formula = parenting_score ~ as.numeric(year), data = parenting_movies)
## 
## Residuals:
##     Min      1Q  Median      3Q     Max 
## -3.1816 -1.7771 -0.8276  0.9195  7.5769 
## 
## Coefficients:
##                    Estimate Std. Error t value Pr(>|t|)    
## (Intercept)      -30.847334   7.133165  -4.324 1.58e-05 ***
## as.numeric(year)   0.016854   0.003564   4.730 2.35e-06 ***
## ---
## Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
## 
## Residual standard error: 2.441 on 2997 degrees of freedom
##   (562 observations effacées parce que manquantes)
## Multiple R-squared:  0.007409,   Adjusted R-squared:  0.007077 
## F-statistic: 22.37 on 1 and 2997 DF,  p-value: 2.353e-06
## [1] 13867.19

Model 2

parenting_movies_grouped <- parenting_movies %>%
  filter(parenting_score != "NA") %>%
  group_by(year) %>%
  mutate(movies_per_year = n()) %>%
  ungroup()

lm2_par <- lm(parenting_score ~ as.numeric(year) + movies_per_year, data=parenting_movies_grouped)
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm2_par)$aic
## 
## Call:
## lm(formula = parenting_score ~ as.numeric(year) + movies_per_year, 
##     data = parenting_movies_grouped)
## 
## Residuals:
##     Min      1Q  Median      3Q     Max 
## -3.3776 -1.8071 -0.8199  0.9513  7.6092 
## 
## Coefficients:
##                    Estimate Std. Error t value Pr(>|t|)    
## (Intercept)      -50.816533  14.726448  -3.451 0.000567 ***
## as.numeric(year)   0.027014   0.007461   3.621 0.000299 ***
## movies_per_year   -0.004767   0.003076  -1.550 0.121280    
## ---
## Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
## 
## Residual standard error: 2.44 on 2996 degrees of freedom
##   (1 observation effacée parce que manquante)
## Multiple R-squared:  0.008204,   Adjusted R-squared:  0.007542 
## F-statistic: 12.39 on 2 and 2996 DF,  p-value: 4.373e-06
## [1] 13866.78

Paternal investment

Missing values check

sum_paternal_na <- sum(is.na(all_scores$paternal_score))
print(sum_paternal_na)
## [1] 1107

Model 1

lm1_pat <- lm(paternal_score ~ as.numeric(year), data=all_scores, na.action=na.omit) # lm ignores NAs by default
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm1_pat)$aic
## 
## Call:
## lm(formula = paternal_score ~ as.numeric(year), data = all_scores, 
##     na.action = na.omit)
## 
## Residuals:
##     Min      1Q  Median      3Q     Max 
## -2.5925 -2.1771 -0.4841  0.8048  8.1299 
## 
## Coefficients:
##                    Estimate Std. Error t value Pr(>|t|)    
## (Intercept)      -33.873372   8.558273  -3.958 7.77e-05 ***
## as.numeric(year)   0.018061   0.004275   4.224 2.48e-05 ***
## ---
## Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
## 
## Residual standard error: 2.619 on 2451 degrees of freedom
##   (1108 observations effacées parce que manquantes)
## Multiple R-squared:  0.007229,   Adjusted R-squared:  0.006823 
## F-statistic: 17.85 on 1 and 2451 DF,  p-value: 2.483e-05
## [1] 11688.83

Model 2

paternal_movies_grouped <- all_scores %>%
  filter(paternal_score != "NA") %>%
  group_by(year) %>%
  mutate(movies_per_year = n()) %>%
  ungroup()

lm2_pat <- lm(paternal_score ~ as.numeric(year) + movies_per_year, data=paternal_movies_grouped)
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm2_pat)$aic
## 
## Call:
## lm(formula = paternal_score ~ as.numeric(year) + movies_per_year, 
##     data = paternal_movies_grouped)
## 
## Residuals:
##     Min      1Q  Median      3Q     Max 
## -2.6738 -2.1906 -0.4399  0.7960  8.1183 
## 
## Coefficients:
##                    Estimate Std. Error t value Pr(>|t|)  
## (Intercept)      -42.453862  17.709200  -2.397   0.0166 *
## as.numeric(year)   0.022429   0.008975   2.499   0.0125 *
## movies_per_year   -0.002562   0.004629  -0.553   0.5800  
## ---
## Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
## 
## Residual standard error: 2.619 on 2450 degrees of freedom
##   (1 observation effacée parce que manquante)
## Multiple R-squared:  0.007353,   Adjusted R-squared:  0.006542 
## F-statistic: 9.074 on 2 and 2450 DF,  p-value: 0.0001185
## [1] 11690.52

Long-term relationships

Missing values check

sum_longterm_na <- sum(is.na(all_scores$longterm_score))
print(sum_longterm_na)
## [1] 179

Model 1

lm1_long <- lm(longterm_score ~ as.numeric(year), data=all_scores, na.action=na.omit) # lm ignores NAs by default
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm1_long)$aic
## 
## Call:
## lm(formula = longterm_score ~ as.numeric(year), data = all_scores, 
##     na.action = na.omit)
## 
## Residuals:
##    Min     1Q Median     3Q    Max 
## -4.314 -2.087 -1.073  1.872  5.941 
## 
## Coefficients:
##                   Estimate Std. Error t value Pr(>|t|)  
## (Intercept)      -9.566029   6.337073  -1.510    0.131  
## as.numeric(year)  0.006875   0.003166   2.171    0.030 *
## ---
## Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
## 
## Residual standard error: 2.305 on 3379 degrees of freedom
##   (180 observations effacées parce que manquantes)
## Multiple R-squared:  0.001393,   Adjusted R-squared:  0.001098 
## F-statistic: 4.715 on 1 and 3379 DF,  p-value: 0.02997
## [1] 15244.84

Model 2

longterm_movies_grouped <- all_scores %>%
  filter(longterm_score != "NA") %>%
  group_by(year) %>%
  mutate(movies_per_year = n()) %>%
  ungroup()

lm2_long <- lm(longterm_score ~ as.numeric(year) + movies_per_year, data=longterm_movies_grouped)
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm2_long)$aic
## 
## Call:
## lm(formula = longterm_score ~ as.numeric(year) + movies_per_year, 
##     data = longterm_movies_grouped)
## 
## Residuals:
##    Min     1Q Median     3Q    Max 
## -4.297 -2.085 -1.075  1.873  5.943 
## 
## Coefficients:
##                    Estimate Std. Error t value Pr(>|t|)
## (Intercept)      -8.0620063 13.3562317  -0.604    0.546
## as.numeric(year)  0.0061094  0.0067678   0.903    0.367
## movies_per_year   0.0003179  0.0024847   0.128    0.898
## 
## Residual standard error: 2.305 on 3378 degrees of freedom
##   (1 observation effacée parce que manquante)
## Multiple R-squared:  0.001398,   Adjusted R-squared:  0.0008071 
## F-statistic: 2.365 on 2 and 3378 DF,  p-value: 0.0941
## [1] 15246.83

Plots

Parenting plot

parenting_movies_to_plot <- parenting_movies_grouped %>%
  group_by(year) %>%
  mutate(mean_parenting_score = (mean(parenting_score))) %>%
  ungroup()
  
parenting_plot <- parenting_movies_to_plot %>%
  ggplot(aes(x=as.numeric(year), y=mean_parenting_score)) + geom_point(color="#4394b1", size=2, alpha=0.4) +
  geom_smooth(method = "lm", color = "#045c7b", se = TRUE) +
  labs(title="Evolution of Parenting scores") +
    xlab("Year") +
    ylab("Mean parenting score") +
  theme_bw() +
  theme(
        plot.title = element_text(face="bold", size=14, hjust = 0.5),  # Adjust title appearance
    axis.title.x = element_text(face="bold", size=12),  # Adjust x-axis label appearance
    axis.title.y = element_text(face="bold", size=12),  # Adjust y-axis label appearance
    axis.text.x = element_text(size=10),  # Adjust x-axis tick labels appearance
    axis.text.y = element_text(size=10),  # Adjust y-axis tick labels appearance
    legend.position = "none"
  )
  
parenting_plot
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique

## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique

## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## `geom_smooth()` using formula = 'y ~ x'
## Warning: Removed 1 rows containing non-finite values (`stat_smooth()`).
## Warning: Removed 1 rows containing missing values (`geom_point()`).

### Paternal investment plot

paternal_movies_to_plot <- paternal_movies_grouped %>%
  group_by(year) %>%
  mutate(mean_paternal_score = (mean(paternal_score))) %>%
  ungroup()
  
paternal_plot <- paternal_movies_to_plot %>%
  ggplot(aes(x=as.numeric(year), y=mean_paternal_score)) + geom_point(color="darkgreen", size=2, alpha=0.4) +
  geom_smooth(method = "lm", color = "black", se = TRUE) +
  labs(title="Evolution of paternal investment scores") +
    xlab("Year") +
    ylab("Mean paternal investment score") +
  theme_bw() +
  theme(
        plot.title = element_text(face="bold", size=14, hjust = 0.5),  # Adjust title appearance
    axis.title.x = element_text(face="bold", size=12),  # Adjust x-axis label appearance
    axis.title.y = element_text(face="bold", size=12),  # Adjust y-axis label appearance
    axis.text.x = element_text(size=10),  # Adjust x-axis tick labels appearance
    axis.text.y = element_text(size=10),  # Adjust y-axis tick labels appearance
    legend.position = "none"
  )
  
paternal_plot
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique

## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique

## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## `geom_smooth()` using formula = 'y ~ x'
## Warning: Removed 1 rows containing non-finite values (`stat_smooth()`).
## Warning: Removed 1 rows containing missing values (`geom_point()`).

### Long-term relationship plot

longterm_movies_to_plot <- longterm_movies_grouped %>%
  group_by(year) %>%
  mutate(mean_longterm_score = (mean(longterm_score))) %>%
  ungroup()
  
longterm_plot <- longterm_movies_to_plot %>%
  ggplot(aes(x=as.numeric(year), y=mean_longterm_score)) + geom_point(color="purple", size=2, alpha=0.4) +
  geom_smooth(method = "lm", color = "black", se = TRUE) +
  labs(title="Evolution of long-term relationship scores") +
    xlab("Year") +
    ylab("Mean long-term relationship score") +
  theme_bw() +
  theme(
        plot.title = element_text(face="bold", size=14, hjust = 0.5),  # Adjust title appearance
    axis.title.x = element_text(face="bold", size=12),  # Adjust x-axis label appearance
    axis.title.y = element_text(face="bold", size=12),  # Adjust y-axis label appearance
    axis.text.x = element_text(size=10),  # Adjust x-axis tick labels appearance
    axis.text.y = element_text(size=10),  # Adjust y-axis tick labels appearance
    legend.position = "none"
  )
  
longterm_plot
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique

## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique

## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## `geom_smooth()` using formula = 'y ~ x'
## Warning: Removed 1 rows containing non-finite values (`stat_smooth()`).
## Warning: Removed 1 rows containing missing values (`geom_point()`).

Utils

write.csv(all_scores, "all_scores.csv", row.names=FALSE)