An investigation of cloud seeding and its effect on rainfall.
## Beginning steps - reading in files, summary of data, preparing library
clouds <- read.csv("clouds.csv")
summary(clouds)
## X seeding time sne
## Min. : 1.00 Length:24 Min. : 0.00 Min. :1.300
## 1st Qu.: 6.75 Class :character 1st Qu.:15.75 1st Qu.:2.612
## Median :12.50 Mode :character Median :32.50 Median :3.250
## Mean :12.50 Mean :35.33 Mean :3.169
## 3rd Qu.:18.25 3rd Qu.:55.25 3rd Qu.:3.962
## Max. :24.00 Max. :83.00 Max. :4.650
## cloudcover prewetness echomotion rainfall
## Min. : 2.200 Min. :0.0180 Length:24 Min. : 0.280
## 1st Qu.: 3.750 1st Qu.:0.1405 Class :character 1st Qu.: 2.342
## Median : 5.250 Median :0.2220 Mode :character Median : 4.335
## Mean : 7.246 Mean :0.3271 Mean : 4.403
## 3rd Qu.: 7.175 3rd Qu.:0.3297 3rd Qu.: 5.575
## Max. :37.900 Max. :1.2670 Max. :12.850
library("ggplot2")
library(ggplot2)
## Data visualizations used: boxplot and histogram.
ggplot(clouds, aes(x = seeding, y = rainfall)) +
geom_boxplot() +
labs(title = "Rainfall by Seeding Boxplot", x = "Seeding", y = "Rainfall")
ggplot(clouds, aes(x = rainfall, fill = seeding)) +
geom_histogram(binwidth = 0.5, alpha = 1, position = "identity") +
labs(title = "Rainfall by Seeding Histogram", x = "Rainfall", y = "Frequency")
## T-test to determine standard deviation
t_test_result <- t.test(rainfall ~ seeding, data = clouds)
t_test_result
##
## Welch Two Sample t-test
##
## data: rainfall by seeding
## t = -0.3574, df = 20.871, p-value = 0.7244
## alternative hypothesis: true difference in means between group no and group yes is not equal to 0
## 95 percent confidence interval:
## -3.154691 2.229691
## sample estimates:
## mean in group no mean in group yes
## 4.171667 4.634167
## According to these results, seeding does not have a significant difference on rainfall.
## Use a Multiple Linear Regression model to look at other variables
model <- lm(rainfall ~ seeding + cloudcover + prewetness + echomotion + sne, data = clouds)
summary(model)
##
## Call:
## lm(formula = rainfall ~ seeding + cloudcover + prewetness + echomotion +
## sne, data = clouds)
##
## Residuals:
## Min 1Q Median 3Q Max
## -5.1158 -1.7078 -0.2422 1.3368 6.4827
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) 6.37680 2.43432 2.620 0.0174 *
## seedingyes 1.12011 1.20725 0.928 0.3658
## cloudcover 0.01821 0.11508 0.158 0.8761
## prewetness 2.55109 2.70090 0.945 0.3574
## echomotionstationary 2.59855 1.54090 1.686 0.1090
## sne -1.27530 0.68015 -1.875 0.0771 .
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 2.855 on 18 degrees of freedom
## Multiple R-squared: 0.3403, Adjusted R-squared: 0.157
## F-statistic: 1.857 on 5 and 18 DF, p-value: 0.1524
anova(model)
## Analysis of Variance Table
##
## Response: rainfall
## Df Sum Sq Mean Sq F value Pr(>F)
## seeding 1 1.283 1.2834 0.1575 0.69613
## cloudcover 1 15.738 15.7377 1.9313 0.18157
## prewetness 1 0.003 0.0027 0.0003 0.98557
## echomotion 1 29.985 29.9853 3.6798 0.07108 .
## sne 1 28.649 28.6491 3.5158 0.07711 .
## Residuals 18 146.677 8.1487
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
## SNE seems to have the most influence out of the variables.
## Further exploration with additional models - Seeded
model_seeding <- lm(rainfall ~ sne, data = subset(clouds, seeding == "yes"))
summary(model_seeding)
##
## Call:
## lm(formula = rainfall ~ sne, data = subset(clouds, seeding ==
## "yes"))
##
## Residuals:
## Min 1Q Median 3Q Max
## -3.0134 -1.3297 -0.3276 0.6171 4.3867
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) 12.0202 2.9774 4.037 0.00237 **
## sne -2.2180 0.8722 -2.543 0.02921 *
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 2.27 on 10 degrees of freedom
## Multiple R-squared: 0.3927, Adjusted R-squared: 0.332
## F-statistic: 6.467 on 1 and 10 DF, p-value: 0.02921
## Further exploraion with additional models - Non-seeded
model_nonseeded <- lm(rainfall ~ sne, data = subset(clouds, seeding == "no"))
summary(model_nonseeded)
##
## Call:
## lm(formula = rainfall ~ sne, data = subset(clouds, seeding ==
## "no"))
##
## Residuals:
## Min 1Q Median 3Q Max
## -5.4892 -2.1762 0.2958 1.4902 7.3616
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) 7.319 3.160 2.317 0.043 *
## sne -1.046 0.995 -1.052 0.318
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 3.502 on 10 degrees of freedom
## Multiple R-squared: 0.09957, Adjusted R-squared: 0.009528
## F-statistic: 1.106 on 1 and 10 DF, p-value: 0.3177
## Comparison of seeded vs non-seeded
plot_seeded <- ggplot(subset(clouds, seeding == "yes"), aes(x = sne, y = rainfall)) +
geom_point() +
geom_smooth(method = "lm", se = FALSE) +
labs(title = "Rainfall vs SNE, SEEDED", x = "SNE", y = "RAINFALL")
plot_seeded
## `geom_smooth()` using formula = 'y ~ x'
plot_nonseeded <- ggplot(subset(clouds, seeding == "no"), aes(x = sne, y = rainfall)) +
geom_point() +
geom_smooth(method = "lm", se = FALSE) +
labs(title = "Rainfall vs. SNE, NONSEEDED", x = "SNE", y = "RAINFALL")
plot_nonseeded
## `geom_smooth()` using formula = 'y ~ x'
## Open two new plotting windows for side by side comparison
dev.new()
print(plot_seeded)
## `geom_smooth()` using formula = 'y ~ x'
dev.new()
print(plot_nonseeded)
## `geom_smooth()` using formula = 'y ~ x'
## CONCLUSION: The comparison demonstrated that in seeded experiments, SNE increasing increases rainfall. In non-seeded experiments, however, there isn't much variation influenced by SNE.