setwd("C:/Users/marc-/Documents/OneDrive/Documents/Travail/Stage IJN/romance_study_1.2")
library(dplyr)
## Warning: le package 'dplyr' a été compilé avec la version R 4.3.1
##
## Attachement du package : 'dplyr'
## Les objets suivants sont masqués depuis 'package:stats':
##
## filter, lag
## Les objets suivants sont masqués depuis 'package:base':
##
## intersect, setdiff, setequal, union
library("readxl")
## Warning: le package 'readxl' a été compilé avec la version R 4.3.1
library(readr)
## Warning: le package 'readr' a été compilé avec la version R 4.3.1
library(stringr)
## Warning: le package 'stringr' a été compilé avec la version R 4.3.3
library(ggplot2)
## Warning: le package 'ggplot2' a été compilé avec la version R 4.3.2
new_n <- read.csv("new_n.csv")
keywords <- read_csv("keyword_transformed.csv")
## Warning: One or more parsing issues, call `problems()` on your data frame for details,
## e.g.:
## dat <- vroom(...)
## problems(dat)
## Rows: 9872 Columns: 3923
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (2): imaginary_world, psychotronic-film
## dbl (3921): imdb, escape-from-prison, prison, voice-over-narration, suicide-...
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
# Retrieving the imdb ID column after import issues:
keywords <- keywords %>%
select(-imdb) %>%
rename(c("imdb" = "imaginary_world"))
# Scores datasets
parenting_movies <- read_csv("PARENTING_MOVIES_PROGRESS.csv")
## Rows: 3561 Columns: 229
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (192): imdb, wiki, genre_wiki, country_wiki, language_wiki, article_wiki...
## dbl (37): ...1, latvia, israel, panama, netherlandsantilles, federalrepubli...
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
paternal_movies <- read_csv("PATERNAL_MOVIES_PROGRESS.csv")
## Rows: 3561 Columns: 229
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (192): imdb, wiki, genre_wiki, country_wiki, language_wiki, article_wiki...
## dbl (37): ...1, latvia, israel, panama, netherlandsantilles, federalrepubli...
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
longterm_movies <- read_csv("LONGTERM_MOVIES_PROGRESS.csv")
## Rows: 3561 Columns: 229
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (192): imdb, wiki, genre_wiki, country_wiki, language_wiki, article_wiki...
## dbl (37): ...1, latvia, israel, panama, netherlandsantilles, federalrepubli...
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
# PARENTING
# Rename the column with explanations + scores
parenting_movies <- parenting_movies %>%
rename(
"parenting_exp" = "parenting_score"
)
# Extract the scores in a dedicated column
parenting_movies <- parenting_movies %>%
mutate(parenting_score = str_extract(parenting_exp, "SCORE = \\d+") %>%
str_extract("\\d+") %>%
as.numeric())
print(colnames(parenting_movies))
## [1] "...1" "imdb"
## [3] "wiki" "genre_wiki"
## [5] "country_wiki" "language_wiki"
## [7] "article_wiki" "page_wiki"
## [9] "plot_wiki" "title"
## [11] "year" "date_published"
## [13] "genre_imdb" "duration"
## [15] "country_imdb" "language_imdb"
## [17] "director" "writer"
## [19] "production_company" "actors"
## [21] "description" "avg_vote"
## [23] "votes" "usa_gross_income"
## [25] "worlwide_gross_income" "metascore"
## [27] "reviews_from_users" "reviews_from_critics"
## [29] "age" "gender"
## [31] "relation" "ope"
## [33] "con" "ext"
## [35] "agr" "neu"
## [37] "plot_wiki_cleaned" "drama"
## [39] "fantasy" "musical"
## [41] "romance" "action"
## [43] "comedy" "thriller"
## [45] "biography" "family"
## [47] "sci_fi" "crime"
## [49] "history" "sport"
## [51] "adventure" "mystery"
## [53] "war" "horror"
## [55] "music" "animation"
## [57] "western" "reality_tv"
## [59] "film_noir" "documentary"
## [61] "india" "thailand"
## [63] "bangladesh" "sweden"
## [65] "france" "australia"
## [67] "usa" "uk"
## [69] "austria" "china"
## [71] "switzerland" "canada"
## [73] "unitedarabemirates" "japan"
## [75] "qatar" "fiji"
## [77] "pakistan" "iran"
## [79] "netherlands" "germany"
## [81] "southafrica" "nepal"
## [83] "eastgermany" "sovietunion"
## [85] "hungary" "kenya"
## [87] "turkey" "uganda"
## [89] "srilanka" "namibia"
## [91] "denmark" "singapore"
## [93] "luxembourg" "yugoslavia"
## [95] "indonesia" "czechrepublic"
## [97] "slovakia" "czechoslovakia"
## [99] "belgium" "serbia"
## [101] "poland" "bhutan"
## [103] "taiwan" "greece"
## [105] "newzealand" "vanuatu"
## [107] "italy" "ireland"
## [109] "finland" "laos"
## [111] "colombia" "russia"
## [113] "hongkong" "spain"
## [115] "southkorea" "estonia"
## [117] "britishvirginislands" "haiti"
## [119] "brazil" "malaysia"
## [121] "iceland" "morocco"
## [123] "cambodia" "argentina"
## [125] "portugal" "cuba"
## [127] "mexico" "westgermany"
## [129] "chile" "uruguay"
## [131] "norway" "afghanistan"
## [133] "philippines" "cyprus"
## [135] "uzbekistan" "venezuela"
## [137] "malta" "bulgaria"
## [139] "peru" "monaco"
## [141] "guatemala" "dominicanrepublic"
## [143] "romania" "vietnam"
## [145] "latvia" "kazakhstan"
## [147] "ukraine" "israel"
## [149] "belarus" "armenia"
## [151] "lithuania" "republicofnorthmacedonia"
## [153] "georgia" "albania"
## [155] "kyrgyzstan" "panama"
## [157] "netherlandsantilles" "croatia"
## [159] "slovenia" "federalrepublicofyugoslavia"
## [161] "algeria" "egypt"
## [163] "liechtenstein" "bosniaandherzegovina"
## [165] "iraq" "jordan"
## [167] "palestine" "lebanon"
## [169] "mongolia" "mozambique"
## [171] "paraguay" "svalbardandjanmayen"
## [173] "ghana" "isleofman"
## [175] "burkinafaso" "tunisia"
## [177] "senegal" "cameroon"
## [179] "tajikistan" "angola"
## [181] "martinique" "chad"
## [183] "mauritania" "liberia"
## [185] "newcaledonia" "thedemocraticrepublicofcongo"
## [187] "greenland" "macao"
## [189] "côted.ivoire" "myanmar"
## [191] "rwanda" "zaire"
## [193] "mali" "guinea"
## [195] "northkorea" "nigeria"
## [197] "azerbaijan" "malawi"
## [199] "zimbabwe" "bahrain"
## [201] "caymanislands" "puertorico"
## [203] "bahamas" "somalia"
## [205] "ethiopia" "bolivia"
## [207] "sudan" "bermuda"
## [209] "jamaica" "aruba"
## [211] "libya" "botswana"
## [213] "gibraltar" "trinidadandtobago"
## [215] "serbiaandmontenegro" "saudiarabia"
## [217] "gabon" "kuwait"
## [219] "kosovo" "zambia"
## [221] "nicaragua" "ecuador"
## [223] "montenegro" "guadeloupe"
## [225] "tanzania" "gen_love"
## [227] "gen_love_score" "year_sqrt"
## [229] "parenting_exp" "parenting_score"
# PATERNAL INVESTMENT
# Rename the column with explanations + scores
paternal_movies <- paternal_movies %>%
rename(
"paternal_exp" = "paternal_score"
)
# Extract the scores in a dedicated column
paternal_movies <- paternal_movies %>%
mutate(paternal_score = str_extract(paternal_exp, "SCORE = \\d+") %>%
str_extract("\\d+") %>%
as.numeric())
print(colnames(paternal_movies))
## [1] "...1" "imdb"
## [3] "wiki" "genre_wiki"
## [5] "country_wiki" "language_wiki"
## [7] "article_wiki" "page_wiki"
## [9] "plot_wiki" "title"
## [11] "year" "date_published"
## [13] "genre_imdb" "duration"
## [15] "country_imdb" "language_imdb"
## [17] "director" "writer"
## [19] "production_company" "actors"
## [21] "description" "avg_vote"
## [23] "votes" "usa_gross_income"
## [25] "worlwide_gross_income" "metascore"
## [27] "reviews_from_users" "reviews_from_critics"
## [29] "age" "gender"
## [31] "relation" "ope"
## [33] "con" "ext"
## [35] "agr" "neu"
## [37] "plot_wiki_cleaned" "drama"
## [39] "fantasy" "musical"
## [41] "romance" "action"
## [43] "comedy" "thriller"
## [45] "biography" "family"
## [47] "sci_fi" "crime"
## [49] "history" "sport"
## [51] "adventure" "mystery"
## [53] "war" "horror"
## [55] "music" "animation"
## [57] "western" "reality_tv"
## [59] "film_noir" "documentary"
## [61] "india" "thailand"
## [63] "bangladesh" "sweden"
## [65] "france" "australia"
## [67] "usa" "uk"
## [69] "austria" "china"
## [71] "switzerland" "canada"
## [73] "unitedarabemirates" "japan"
## [75] "qatar" "fiji"
## [77] "pakistan" "iran"
## [79] "netherlands" "germany"
## [81] "southafrica" "nepal"
## [83] "eastgermany" "sovietunion"
## [85] "hungary" "kenya"
## [87] "turkey" "uganda"
## [89] "srilanka" "namibia"
## [91] "denmark" "singapore"
## [93] "luxembourg" "yugoslavia"
## [95] "indonesia" "czechrepublic"
## [97] "slovakia" "czechoslovakia"
## [99] "belgium" "serbia"
## [101] "poland" "bhutan"
## [103] "taiwan" "greece"
## [105] "newzealand" "vanuatu"
## [107] "italy" "ireland"
## [109] "finland" "laos"
## [111] "colombia" "russia"
## [113] "hongkong" "spain"
## [115] "southkorea" "estonia"
## [117] "britishvirginislands" "haiti"
## [119] "brazil" "malaysia"
## [121] "iceland" "morocco"
## [123] "cambodia" "argentina"
## [125] "portugal" "cuba"
## [127] "mexico" "westgermany"
## [129] "chile" "uruguay"
## [131] "norway" "afghanistan"
## [133] "philippines" "cyprus"
## [135] "uzbekistan" "venezuela"
## [137] "malta" "bulgaria"
## [139] "peru" "monaco"
## [141] "guatemala" "dominicanrepublic"
## [143] "romania" "vietnam"
## [145] "latvia" "kazakhstan"
## [147] "ukraine" "israel"
## [149] "belarus" "armenia"
## [151] "lithuania" "republicofnorthmacedonia"
## [153] "georgia" "albania"
## [155] "kyrgyzstan" "panama"
## [157] "netherlandsantilles" "croatia"
## [159] "slovenia" "federalrepublicofyugoslavia"
## [161] "algeria" "egypt"
## [163] "liechtenstein" "bosniaandherzegovina"
## [165] "iraq" "jordan"
## [167] "palestine" "lebanon"
## [169] "mongolia" "mozambique"
## [171] "paraguay" "svalbardandjanmayen"
## [173] "ghana" "isleofman"
## [175] "burkinafaso" "tunisia"
## [177] "senegal" "cameroon"
## [179] "tajikistan" "angola"
## [181] "martinique" "chad"
## [183] "mauritania" "liberia"
## [185] "newcaledonia" "thedemocraticrepublicofcongo"
## [187] "greenland" "macao"
## [189] "côted.ivoire" "myanmar"
## [191] "rwanda" "zaire"
## [193] "mali" "guinea"
## [195] "northkorea" "nigeria"
## [197] "azerbaijan" "malawi"
## [199] "zimbabwe" "bahrain"
## [201] "caymanislands" "puertorico"
## [203] "bahamas" "somalia"
## [205] "ethiopia" "bolivia"
## [207] "sudan" "bermuda"
## [209] "jamaica" "aruba"
## [211] "libya" "botswana"
## [213] "gibraltar" "trinidadandtobago"
## [215] "serbiaandmontenegro" "saudiarabia"
## [217] "gabon" "kuwait"
## [219] "kosovo" "zambia"
## [221] "nicaragua" "ecuador"
## [223] "montenegro" "guadeloupe"
## [225] "tanzania" "gen_love"
## [227] "gen_love_score" "year_sqrt"
## [229] "paternal_exp" "paternal_score"
# LONG-TERM RELATIONSHIPS
# Rename the column with explanations + scores
longterm_movies <- longterm_movies %>%
rename(
"longterm_exp" = "longterm_score"
)
# Extract the scores in a dedicated column
longterm_movies <- longterm_movies %>%
mutate(longterm_score = str_extract(longterm_exp, "SCORE = \\d+") %>%
str_extract("\\d+") %>%
as.numeric())
print(colnames(longterm_movies))
## [1] "...1" "imdb"
## [3] "wiki" "genre_wiki"
## [5] "country_wiki" "language_wiki"
## [7] "article_wiki" "page_wiki"
## [9] "plot_wiki" "title"
## [11] "year" "date_published"
## [13] "genre_imdb" "duration"
## [15] "country_imdb" "language_imdb"
## [17] "director" "writer"
## [19] "production_company" "actors"
## [21] "description" "avg_vote"
## [23] "votes" "usa_gross_income"
## [25] "worlwide_gross_income" "metascore"
## [27] "reviews_from_users" "reviews_from_critics"
## [29] "age" "gender"
## [31] "relation" "ope"
## [33] "con" "ext"
## [35] "agr" "neu"
## [37] "plot_wiki_cleaned" "drama"
## [39] "fantasy" "musical"
## [41] "romance" "action"
## [43] "comedy" "thriller"
## [45] "biography" "family"
## [47] "sci_fi" "crime"
## [49] "history" "sport"
## [51] "adventure" "mystery"
## [53] "war" "horror"
## [55] "music" "animation"
## [57] "western" "reality_tv"
## [59] "film_noir" "documentary"
## [61] "india" "thailand"
## [63] "bangladesh" "sweden"
## [65] "france" "australia"
## [67] "usa" "uk"
## [69] "austria" "china"
## [71] "switzerland" "canada"
## [73] "unitedarabemirates" "japan"
## [75] "qatar" "fiji"
## [77] "pakistan" "iran"
## [79] "netherlands" "germany"
## [81] "southafrica" "nepal"
## [83] "eastgermany" "sovietunion"
## [85] "hungary" "kenya"
## [87] "turkey" "uganda"
## [89] "srilanka" "namibia"
## [91] "denmark" "singapore"
## [93] "luxembourg" "yugoslavia"
## [95] "indonesia" "czechrepublic"
## [97] "slovakia" "czechoslovakia"
## [99] "belgium" "serbia"
## [101] "poland" "bhutan"
## [103] "taiwan" "greece"
## [105] "newzealand" "vanuatu"
## [107] "italy" "ireland"
## [109] "finland" "laos"
## [111] "colombia" "russia"
## [113] "hongkong" "spain"
## [115] "southkorea" "estonia"
## [117] "britishvirginislands" "haiti"
## [119] "brazil" "malaysia"
## [121] "iceland" "morocco"
## [123] "cambodia" "argentina"
## [125] "portugal" "cuba"
## [127] "mexico" "westgermany"
## [129] "chile" "uruguay"
## [131] "norway" "afghanistan"
## [133] "philippines" "cyprus"
## [135] "uzbekistan" "venezuela"
## [137] "malta" "bulgaria"
## [139] "peru" "monaco"
## [141] "guatemala" "dominicanrepublic"
## [143] "romania" "vietnam"
## [145] "latvia" "kazakhstan"
## [147] "ukraine" "israel"
## [149] "belarus" "armenia"
## [151] "lithuania" "republicofnorthmacedonia"
## [153] "georgia" "albania"
## [155] "kyrgyzstan" "panama"
## [157] "netherlandsantilles" "croatia"
## [159] "slovenia" "federalrepublicofyugoslavia"
## [161] "algeria" "egypt"
## [163] "liechtenstein" "bosniaandherzegovina"
## [165] "iraq" "jordan"
## [167] "palestine" "lebanon"
## [169] "mongolia" "mozambique"
## [171] "paraguay" "svalbardandjanmayen"
## [173] "ghana" "isleofman"
## [175] "burkinafaso" "tunisia"
## [177] "senegal" "cameroon"
## [179] "tajikistan" "angola"
## [181] "martinique" "chad"
## [183] "mauritania" "liberia"
## [185] "newcaledonia" "thedemocraticrepublicofcongo"
## [187] "greenland" "macao"
## [189] "côted.ivoire" "myanmar"
## [191] "rwanda" "zaire"
## [193] "mali" "guinea"
## [195] "northkorea" "nigeria"
## [197] "azerbaijan" "malawi"
## [199] "zimbabwe" "bahrain"
## [201] "caymanislands" "puertorico"
## [203] "bahamas" "somalia"
## [205] "ethiopia" "bolivia"
## [207] "sudan" "bermuda"
## [209] "jamaica" "aruba"
## [211] "libya" "botswana"
## [213] "gibraltar" "trinidadandtobago"
## [215] "serbiaandmontenegro" "saudiarabia"
## [217] "gabon" "kuwait"
## [219] "kosovo" "zambia"
## [221] "nicaragua" "ecuador"
## [223] "montenegro" "guadeloupe"
## [225] "tanzania" "gen_love"
## [227] "gen_love_score" "year_sqrt"
## [229] "longterm_exp" "longterm_score"
parenting_movies_toconcat <- parenting_movies[,c("imdb", "title", "year", "parenting_exp","parenting_score")]
paternal_movies_toconcat <- paternal_movies[,c("paternal_exp","paternal_score")]
longterm_movies_toconcat <- longterm_movies[,c("longterm_exp","longterm_score")]
all_scores <- cbind(parenting_movies_toconcat, paternal_movies_toconcat, longterm_movies_toconcat)
# Overlap of new_n movies with keyword tagged movies: 1385
overlap_df <- new_n %>%
merge(keywords, by="imdb", all.x=FALSE)
print(nrow(overlap_df))
## [1] 1385
# Paternal keywords
father_kw_df <- keywords[, c("imdb", "father", "father-daughter-relationship", "father-son-relationship")]
new_n_father <- new_n %>%
merge(father_kw_df, by="imdb", all.x=TRUE) %>%
mutate(father_kw = ifelse(father | `father-daughter-relationship` | `father-son-relationship`, TRUE, FALSE)) %>%
filter(father_kw == TRUE)
print(nrow(new_n_father))
## [1] 157
# Parenting keywords
parenting_kw_df <- keywords[,c("imdb", "father", "father-daughter-relationship", "father-son-relationship", "mother", "mother-son-relationship", "mother-daughter-relationship")]
new_n_parenting <- new_n %>%
merge(parenting_kw_df, by="imdb", all.x=TRUE) %>%
mutate(parenting_kw = ifelse(father | `father-daughter-relationship` | `father-son-relationship`| mother | `mother-son-relationship` | `mother-daughter-relationship`, TRUE, FALSE)) %>%
filter(parenting_kw == TRUE)
print(nrow(new_n_parenting))
## [1] 489
# Long-term relationship keywords
longterm_kw_df <- keywords[,c("imdb", "husband-wife-relationship")]
new_n_longterm <- new_n %>%
merge(longterm_kw_df, by="imdb", all.x=TRUE) %>%
mutate(longterm_kw = ifelse(`husband-wife-relationship`, TRUE, FALSE)) %>%
filter(longterm_kw == TRUE)
print(nrow(new_n_longterm))
## [1] 105
# Other jealousy keywords
jealousy_kw_df <- keywords[,c("imdb", "jealousy", "cheating-wife", "cheating-husband", "cheating-girlfriend", "cheating")]
new_n_jealousy <- new_n %>%
merge(jealousy_kw_df, by="imdb", all.x=TRUE) %>%
mutate(jealousy_kw = ifelse(jealousy | `cheating-wife` | `cheating-husband` | `cheating-girlfriend` | cheating, TRUE, FALSE)) %>%
filter(jealousy_kw == TRUE)
print(nrow(new_n_jealousy))
## [1] 80
## Parenting
all_scores <- all_scores %>%
mutate(parenting_kw = imdb %in% new_n_parenting$imdb)
print(sum(all_scores$parenting_kw))
## [1] 489
has_parenting_kw <- all_scores$parenting_score[all_scores$parenting_kw == TRUE]
hasnot_parenting_kw <- all_scores$parenting_score[all_scores$parenting_kw == FALSE]
t.test(has_parenting_kw, hasnot_parenting_kw, mu=0)
##
## Welch Two Sample t-test
##
## data: has_parenting_kw and hasnot_parenting_kw
## t = 4.3068, df = 643.69, p-value = 1.915e-05
## alternative hypothesis: true difference in means is not equal to 0
## 95 percent confidence interval:
## 0.3003448 0.8037552
## sample estimates:
## mean of x mean of y
## 3.354906 2.802856
## Paternal investment
all_scores <- all_scores %>%
mutate(paternal_kw = imdb %in% new_n_father$imdb)
print(sum(all_scores$paternal_kw))
## [1] 157
has_paternal_kw <- all_scores$paternal_score[all_scores$paternal_kw == TRUE]
hasnot_paternal_kw <- all_scores$paternal_score[all_scores$paternal_kw == FALSE]
t.test(has_paternal_kw, hasnot_paternal_kw, mu=0)
##
## Welch Two Sample t-test
##
## data: has_paternal_kw and hasnot_paternal_kw
## t = -2.3228, df = 158.54, p-value = 0.02146
## alternative hypothesis: true difference in means is not equal to 0
## 95 percent confidence interval:
## -0.93256468 -0.07544865
## sample estimates:
## mean of x mean of y
## 1.807143 2.311150
## Long-term relationships
all_scores <- all_scores %>%
mutate(longterm_kw = imdb %in% new_n_longterm$imdb)
print(sum(all_scores$longterm_kw))
## [1] 105
has_longterm_kw <- all_scores$longterm_score[all_scores$longterm_kw == TRUE]
hasnot_longterm_kw <- all_scores$longterm_score[all_scores$longterm_kw == FALSE]
t.test(has_longterm_kw, hasnot_longterm_kw, mu=0)
##
## Welch Two Sample t-test
##
## data: has_longterm_kw and hasnot_longterm_kw
## t = -4.3469, df = 110.6, p-value = 3.081e-05
## alternative hypothesis: true difference in means is not equal to 0
## 95 percent confidence interval:
## -1.3627918 -0.5093394
## sample estimates:
## mean of x mean of y
## 3.288462 4.224527
# A function for retrieving the AIC
summary_AIC <- function(model) {
summary_output <- summary(model)
aic <- AIC(model)
summary_output$aic <- aic
print(summary_output)
}
sum_parenting_na <- sum(is.na(all_scores$parenting_score))
print(sum_parenting_na)
## [1] 561
lm1_par <- lm(parenting_score ~ as.numeric(year), data=parenting_movies) # lm ignores NAs by default
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm1_par)$aic
##
## Call:
## lm(formula = parenting_score ~ as.numeric(year), data = parenting_movies)
##
## Residuals:
## Min 1Q Median 3Q Max
## -3.1816 -1.7771 -0.8276 0.9195 7.5769
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) -30.847334 7.133165 -4.324 1.58e-05 ***
## as.numeric(year) 0.016854 0.003564 4.730 2.35e-06 ***
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 2.441 on 2997 degrees of freedom
## (562 observations effacées parce que manquantes)
## Multiple R-squared: 0.007409, Adjusted R-squared: 0.007077
## F-statistic: 22.37 on 1 and 2997 DF, p-value: 2.353e-06
## [1] 13867.19
parenting_movies_grouped <- parenting_movies %>%
filter(parenting_score != "NA") %>%
group_by(year) %>%
mutate(movies_per_year = n()) %>%
ungroup()
lm2_par <- lm(parenting_score ~ as.numeric(year) + movies_per_year, data=parenting_movies_grouped)
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm2_par)$aic
##
## Call:
## lm(formula = parenting_score ~ as.numeric(year) + movies_per_year,
## data = parenting_movies_grouped)
##
## Residuals:
## Min 1Q Median 3Q Max
## -3.3776 -1.8071 -0.8199 0.9513 7.6092
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) -50.816533 14.726448 -3.451 0.000567 ***
## as.numeric(year) 0.027014 0.007461 3.621 0.000299 ***
## movies_per_year -0.004767 0.003076 -1.550 0.121280
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 2.44 on 2996 degrees of freedom
## (1 observation effacée parce que manquante)
## Multiple R-squared: 0.008204, Adjusted R-squared: 0.007542
## F-statistic: 12.39 on 2 and 2996 DF, p-value: 4.373e-06
## [1] 13866.78
sum_paternal_na <- sum(is.na(all_scores$paternal_score))
print(sum_paternal_na)
## [1] 1107
lm1_pat <- lm(paternal_score ~ as.numeric(year), data=all_scores, na.action=na.omit) # lm ignores NAs by default
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm1_pat)$aic
##
## Call:
## lm(formula = paternal_score ~ as.numeric(year), data = all_scores,
## na.action = na.omit)
##
## Residuals:
## Min 1Q Median 3Q Max
## -2.5925 -2.1771 -0.4841 0.8048 8.1299
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) -33.873372 8.558273 -3.958 7.77e-05 ***
## as.numeric(year) 0.018061 0.004275 4.224 2.48e-05 ***
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 2.619 on 2451 degrees of freedom
## (1108 observations effacées parce que manquantes)
## Multiple R-squared: 0.007229, Adjusted R-squared: 0.006823
## F-statistic: 17.85 on 1 and 2451 DF, p-value: 2.483e-05
## [1] 11688.83
paternal_movies_grouped <- all_scores %>%
filter(paternal_score != "NA") %>%
group_by(year) %>%
mutate(movies_per_year = n()) %>%
ungroup()
lm2_pat <- lm(paternal_score ~ as.numeric(year) + movies_per_year, data=paternal_movies_grouped)
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm2_pat)$aic
##
## Call:
## lm(formula = paternal_score ~ as.numeric(year) + movies_per_year,
## data = paternal_movies_grouped)
##
## Residuals:
## Min 1Q Median 3Q Max
## -2.6738 -2.1906 -0.4399 0.7960 8.1183
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) -42.453862 17.709200 -2.397 0.0166 *
## as.numeric(year) 0.022429 0.008975 2.499 0.0125 *
## movies_per_year -0.002562 0.004629 -0.553 0.5800
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 2.619 on 2450 degrees of freedom
## (1 observation effacée parce que manquante)
## Multiple R-squared: 0.007353, Adjusted R-squared: 0.006542
## F-statistic: 9.074 on 2 and 2450 DF, p-value: 0.0001185
## [1] 11690.52
sum_longterm_na <- sum(is.na(all_scores$longterm_score))
print(sum_longterm_na)
## [1] 179
lm1_long <- lm(longterm_score ~ as.numeric(year), data=all_scores, na.action=na.omit) # lm ignores NAs by default
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm1_long)$aic
##
## Call:
## lm(formula = longterm_score ~ as.numeric(year), data = all_scores,
## na.action = na.omit)
##
## Residuals:
## Min 1Q Median 3Q Max
## -4.314 -2.087 -1.073 1.872 5.941
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) -9.566029 6.337073 -1.510 0.131
## as.numeric(year) 0.006875 0.003166 2.171 0.030 *
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 2.305 on 3379 degrees of freedom
## (180 observations effacées parce que manquantes)
## Multiple R-squared: 0.001393, Adjusted R-squared: 0.001098
## F-statistic: 4.715 on 1 and 3379 DF, p-value: 0.02997
## [1] 15244.84
longterm_movies_grouped <- all_scores %>%
filter(longterm_score != "NA") %>%
group_by(year) %>%
mutate(movies_per_year = n()) %>%
ungroup()
lm2_long <- lm(longterm_score ~ as.numeric(year) + movies_per_year, data=longterm_movies_grouped)
## Warning in eval(predvars, data, env): NAs introduits lors de la conversion
## automatique
summary_AIC(lm2_long)$aic
##
## Call:
## lm(formula = longterm_score ~ as.numeric(year) + movies_per_year,
## data = longterm_movies_grouped)
##
## Residuals:
## Min 1Q Median 3Q Max
## -4.297 -2.085 -1.075 1.873 5.943
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) -8.0620063 13.3562317 -0.604 0.546
## as.numeric(year) 0.0061094 0.0067678 0.903 0.367
## movies_per_year 0.0003179 0.0024847 0.128 0.898
##
## Residual standard error: 2.305 on 3378 degrees of freedom
## (1 observation effacée parce que manquante)
## Multiple R-squared: 0.001398, Adjusted R-squared: 0.0008071
## F-statistic: 2.365 on 2 and 3378 DF, p-value: 0.0941
## [1] 15246.83
parenting_movies_to_plot <- parenting_movies_grouped %>%
group_by(year) %>%
mutate(mean_parenting_score = (mean(parenting_score))) %>%
ungroup()
parenting_plot <- parenting_movies_to_plot %>%
ggplot(aes(x=as.numeric(year), y=mean_parenting_score)) + geom_point(color="#4394b1", size=2, alpha=0.4) +
geom_smooth(method = "lm", color = "#045c7b", se = TRUE) +
labs(title="Evolution of Parenting scores") +
xlab("Year") +
ylab("Mean parenting score") +
theme_bw() +
theme(
plot.title = element_text(face="bold", size=14, hjust = 0.5), # Adjust title appearance
axis.title.x = element_text(face="bold", size=12), # Adjust x-axis label appearance
axis.title.y = element_text(face="bold", size=12), # Adjust y-axis label appearance
axis.text.x = element_text(size=10), # Adjust x-axis tick labels appearance
axis.text.y = element_text(size=10), # Adjust y-axis tick labels appearance
legend.position = "none"
)
parenting_plot
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## `geom_smooth()` using formula = 'y ~ x'
## Warning: Removed 1 rows containing non-finite values (`stat_smooth()`).
## Warning: Removed 1 rows containing missing values (`geom_point()`).
### Paternal investment plot
paternal_movies_to_plot <- paternal_movies_grouped %>%
group_by(year) %>%
mutate(mean_paternal_score = (mean(paternal_score))) %>%
ungroup()
paternal_plot <- paternal_movies_to_plot %>%
ggplot(aes(x=as.numeric(year), y=mean_paternal_score)) + geom_point(color="darkgreen", size=2, alpha=0.4) +
geom_smooth(method = "lm", color = "black", se = TRUE) +
labs(title="Evolution of paternal investment scores") +
xlab("Year") +
ylab("Mean paternal investment score") +
theme_bw() +
theme(
plot.title = element_text(face="bold", size=14, hjust = 0.5), # Adjust title appearance
axis.title.x = element_text(face="bold", size=12), # Adjust x-axis label appearance
axis.title.y = element_text(face="bold", size=12), # Adjust y-axis label appearance
axis.text.x = element_text(size=10), # Adjust x-axis tick labels appearance
axis.text.y = element_text(size=10), # Adjust y-axis tick labels appearance
legend.position = "none"
)
paternal_plot
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## `geom_smooth()` using formula = 'y ~ x'
## Warning: Removed 1 rows containing non-finite values (`stat_smooth()`).
## Warning: Removed 1 rows containing missing values (`geom_point()`).
### Long-term relationship plot
longterm_movies_to_plot <- longterm_movies_grouped %>%
group_by(year) %>%
mutate(mean_longterm_score = (mean(longterm_score))) %>%
ungroup()
longterm_plot <- longterm_movies_to_plot %>%
ggplot(aes(x=as.numeric(year), y=mean_longterm_score)) + geom_point(color="purple", size=2, alpha=0.4) +
geom_smooth(method = "lm", color = "black", se = TRUE) +
labs(title="Evolution of long-term relationship scores") +
xlab("Year") +
ylab("Mean long-term relationship score") +
theme_bw() +
theme(
plot.title = element_text(face="bold", size=14, hjust = 0.5), # Adjust title appearance
axis.title.x = element_text(face="bold", size=12), # Adjust x-axis label appearance
axis.title.y = element_text(face="bold", size=12), # Adjust y-axis label appearance
axis.text.x = element_text(size=10), # Adjust x-axis tick labels appearance
axis.text.y = element_text(size=10), # Adjust y-axis tick labels appearance
legend.position = "none"
)
longterm_plot
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## Warning in FUN(X[[i]], ...): NAs introduits lors de la conversion automatique
## `geom_smooth()` using formula = 'y ~ x'
## Warning: Removed 1 rows containing non-finite values (`stat_smooth()`).
## Warning: Removed 1 rows containing missing values (`geom_point()`).
write.csv(all_scores, "all_scores.csv", row.names=FALSE)