Code
knitr::opts_chunk$set(
echo = TRUE,
message = FALSE,
warning = FALSE
)knitr::opts_chunk$set(
echo = TRUE,
message = FALSE,
warning = FALSE
)#| label: load-full-data
# Load the dataset
data <- read.csv("Final_Project_Data.csv",
stringsAsFactors = FALSE)
# Clean column names by replacing spaces or dots with underscores
names(data) <- gsub("[\\. ]+", "_", names(data))
# Inspect the structure and dimensions of the dataset
str(data)'data.frame': 155 obs. of 8 variables:
$ Country_Code: chr "EGY" "EGY" "EGY" "EGY" ...
$ Year : int 1995 1996 1997 1998 1999 2000 2001 2002 2003 2004 ...
$ FDI_GDP : num 1 0.9 1.1 1.3 1.2 1.2 0.5 0.8 0.3 1.6 ...
$ GDP_USD : num 1.36e+11 1.43e+11 1.51e+11 1.59e+11 1.69e+11 ...
$ IND_VA_GDP : num 30.2 29.5 29 28.6 28.4 30.8 30.9 32.6 33.4 34.7 ...
$ NODA_RCV : num 3.4 3.2 2.5 2.3 1.8 1.4 1.3 1.5 1.3 1.9 ...
$ REN_ERG_CONS: num 9.4 9 8.6 8.5 8.8 8.3 7.8 7.3 7.4 7.6 ...
$ Trade_GDP : num 50.2 46.9 43.7 41.9 38.4 ...
dim(data)[1] 155 8
# Display the first 10 observations
head(data, 10) Country_Code Year FDI_GDP GDP_USD IND_VA_GDP NODA_RCV REN_ERG_CONS
1 EGY 1995 1.0 136218680413 30.2 3.4 9.4
2 EGY 1996 0.9 143014263359 29.5 3.2 9.0
3 EGY 1997 1.1 150869114035 29.0 2.5 8.6
4 EGY 1998 1.3 159280817658 28.6 2.3 8.5
5 EGY 1999 1.2 168922784450 28.4 1.8 8.8
6 EGY 2000 1.2 179683172290 30.8 1.4 8.3
7 EGY 2001 0.5 186035425185 30.9 1.3 7.8
8 EGY 2002 0.8 190482051409 32.6 1.5 7.3
9 EGY 2003 0.3 196565009504 33.4 1.3 7.4
10 EGY 2004 1.6 204608590453 34.7 1.9 7.6
Trade_GDP
1 50.24510
2 46.94856
3 43.73825
4 41.92763
5 38.36151
6 39.01794
7 39.81043
8 40.98707
9 46.17964
10 57.81991
The dataset contains 155 observations covering five countries over the period 1995–2025. Each observation represents a country-year combination. The variables contain information on foreign direct investment, real GDP, industrial value added, net official development assistance, renewable energy consumption, and trade trade openness. The first ten observations are displayed to inspect the structure of the data without printing the entire dataset.
A scalar is a single numerical value. It may represent one particular observation, such as a country’s GDP, FDI share, or renewable-energy consumption in a given year. The following example uses Sudan in 2021 to demonstrate basic scalar operations.
The 2021 observation for Sudan is first extracted from the dataset. Individual variables from this observation are then stored as scalars. This allows basic scalar operations to be demonstrated using values from the empirical dataset rather than manually entered numbers.
# Extract the observation for Sudan in 2021
sdn_2021 <- data[
data$Country_Code == "SDN" & data$Year == 2021,
]
# Display the selected observation
sdn_2021 Country_Code Year FDI_GDP GDP_USD IND_VA_GDP NODA_RCV REN_ERG_CONS
120 SDN 2021 1.52754 48525909156 22.5 11.62119 61
Trade_GDP
120 4.127549
# Store individual values as scalars
gdp_sdn <- sdn_2021$GDP_USD
ind_va_pct <- sdn_2021$IND_VA_GDP
fdi_pct <- sdn_2021$FDI_GDP
trade_pct <- sdn_2021$Trade_GDP
ren_share <- sdn_2021$REN_ERG_CONS
# 1. Approximate real industrial value added
real_ind_va <- gdp_sdn * (ind_va_pct / 100)
# 2. FDI-to-trade ratio
fdi_trade_ratio <- fdi_pct / trade_pct
# 3. Natural logarithm of real GDP
ln_real_gdp <- log(gdp_sdn)
# 4. Modulo and integer division
mod_share <- round(ren_share) %% 10
decade_share <- round(ren_share) %/% 10
# 5. Rounding operations
real_ind_va_bil <- round(real_ind_va / 1e9, 2)
ceil_ratio <- ceiling(fdi_trade_ratio * 100)
floor_ratio <- floor(fdi_trade_ratio * 100)
# Display results
real_ind_va[1] 10918329560
fdi_trade_ratio[1] 0.3700841
ln_real_gdp[1] 24.60536
mod_share[1] 1
decade_share[1] 6
real_ind_va_bil[1] 10.92
ceil_ratio[1] 38
floor_ratio[1] 37
The extracted observations for Sudan in 2021 are stored as scalars. Real GDP is measured in constant 2015 US$, while industrial value added, FDI, trade, and renewable energy consumption are expressed as percentage indicators.
Multiplying real GDP by industry’s share of GDP gives an approximate real industrial value added of about 10.92 billion constant 2015 US$. The FDI-to-trade ratio is approximately 0.37, while the natural logarithm of real GDP is approximately 24.61.
Character variables contain textual information rather than numerical values. String operations are useful for constructing observation identifiers, variable labels, model formulas, and data headers.
# Define panel identifiers
country_code <- "SDN"
sample_year <- "2021"
# 1. Combine strings with spaces
panel_desc <- paste("Country:", country_code,
"Year:", sample_year)
# 2. Construct a unique country-year identifier
panel_id <- paste0(country_code, "_", sample_year)
# 3. Construct a regression formula from separate strings
dependent_var <- "REN_ERG_CONS"
independent_vars <- "FDI_GDP + GDP_USD + IND_VA_GDP + NODA_RCV + Trade_GDP"
reg_formula <- paste(dependent_var, "~", independent_vars)
# 4. Construct a comma-separated variable header
csv_header <- paste(
"Country_Code", "Year", "FDI_GDP", "GDP_USD",
"IND_VA_GDP", "NODA_RCV", "REN_ERG_CONS", "Trade_GDP",
sep = ","
)
# Display results
panel_desc[1] "Country: SDN Year: 2021"
panel_id[1] "SDN_2021"
reg_formula[1] "REN_ERG_CONS ~ FDI_GDP + GDP_USD + IND_VA_GDP + NODA_RCV + Trade_GDP"
csv_header[1] "Country_Code,Year,FDI_GDP,GDP_USD,IND_VA_GDP,NODA_RCV,REN_ERG_CONS,Trade_GDP"
# Create an observation-variable key
obs_key <- "SDN_2021_REN_ERG_CONS"
# Count characters
nchar(obs_key)[1] 21
# Extract selected parts of the string
iso_code <- substr(obs_key, 1, 3)
year_val <- substr(obs_key, 5, 8)
dep_var <- substr(obs_key, 10, 21)
iso_code[1] "SDN"
year_val[1] "2021"
dep_var[1] "REN_ERG_CONS"
The paste0() function creates the country-year identifier SDN_2021, while paste() constructs a model formula from separate dependent- and independent-variable strings. The functions nchar() and substr() are then used to inspect and extract information from a composite observation key.
The following examples use illustrative thresholds to demonstrate comparisons and the logical operators NOT (!), AND (&), OR (|), and exclusive OR (xor). These thresholds are chosen for demonstration and are not intended to represent formal economic classifications.
# Create logical values using illustrative thresholds
high_renew <- ren_share > 50
high_fdi <- fdi_pct > 5
# Basic logical operators
!high_renew[1] FALSE
high_renew & high_fdi[1] FALSE
high_renew | high_fdi[1] TRUE
xor(high_renew, high_fdi)[1] TRUE
# Numerical comparisons
fdi_pct > 2[1] FALSE
ren_share >= 50[1] TRUE
trade_pct == 20.3[1] FALSE
trade_pct != 0[1] TRUE
# Combined conditions
(ren_share > 50) & (fdi_pct > 3)[1] FALSE
(trade_pct < 15) | (ren_share < 50)[1] TRUE
# Store several logical checks in a vector
regime_checks <- c(
FDI_above_3 = fdi_pct > 3,
RenErg_above_50 = ren_share > 50,
Trade_above_25 = trade_pct > 25,
IndVA_above_15 = ind_va_pct > 15
)
regime_checks FDI_above_3 RenErg_above_50 Trade_above_25 IndVA_above_15
FALSE TRUE FALSE TRUE
For Sudan in 2021, renewable energy consumption exceeds the illustrative 50% threshold, while FDI does not exceed either 3% or 5% of GDP. The logical vector summarizes several conditions simultaneously. This demonstrates how numerical observations can be converted into TRUE and FALSE values for filtering or classification.
The dataset is loaded and extracted the complete 31-year time series observations for Sudan across our study variables.
# Extract Sudan's time-series vectors
years_sdn <- data$Year[data$Country_Code == "SDN"]
fdi_sdn <- data$FDI_GDP[data$Country_Code == "SDN"]
trade_sdn <- data$Trade_GDP[data$Country_Code == "SDN"]
ren_sdn <- data$REN_ERG_CONS[data$Country_Code == "SDN"]
# Inspect vector length and first observations
length(fdi_sdn)[1] 31
head(fdi_sdn)[1] 0.1 0.0 0.8 3.3 3.5 3.2
head(trade_sdn)[1] 14.77240 20.02984 17.85861 21.87515 24.71437 29.40423
# 1. Positional indexing
fdi_sdn[1] # First FDI observation[1] 0.1
fdi_sdn[length(fdi_sdn)] # Last FDI observation[1] NA
# 2. Conditional filtering
high_fdi_vals <- fdi_sdn[fdi_sdn > 3]
high_fdi_years <- years_sdn[fdi_sdn > 3]
high_fdi_vals [1] 3.300000 3.500000 3.200000 3.700000 3.932155 6.317823 5.670906 4.438799
[9] 4.069107 3.344176 3.500045 3.152349 6.142124 3.923121 3.341353 3.512695
[17] NA NA
high_fdi_years [1] 1998 1999 2000 2001 2002 2003 2004 2005 2006 2009 2010 2011 2012 2013 2015
[16] 2018 NA NA
# 3. Vector arithmetic
fdi_sdn_prop <- fdi_sdn / 100
ren_sdn_prop <- ren_sdn / 100
head(fdi_sdn_prop)[1] 0.001 0.000 0.008 0.033 0.035 0.032
head(ren_sdn_prop)[1] 0.815 0.836 0.805 0.819 0.808 0.804
# 4. Sorting
sort(fdi_sdn, decreasing = TRUE, na.last = NA) [1] 6.317823 6.142124 5.670906 4.438799 4.069107 3.932155 3.923121 3.700000
[9] 3.512695 3.500045 3.500000 3.344176 3.341353 3.300000 3.200000 3.152349
[17] 2.651934 2.580439 2.552270 2.549810 2.530916 2.526985 2.495328 1.527540
[25] 1.373994 1.110004 0.800000 0.100000 0.000000
# 5. Summary statistics for renewable energy
mean(ren_sdn, na.rm = TRUE)[1] 69.28519
median(ren_sdn, na.rm = TRUE)[1] 64.3
sd(ren_sdn, na.rm = TRUE)[1] 8.745491
# 6. Missing-value checks
is.na(ren_sdn) [1] FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE
[13] FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE FALSE
[25] FALSE FALSE FALSE TRUE TRUE TRUE TRUE
sum(is.na(ren_sdn))[1] 4
# Without na.rm = TRUE, the result is NA
mean(ren_sdn)[1] NA
# Count missing observations for every variable
colSums(is.na(data))Country_Code Year FDI_GDP GDP_USD IND_VA_GDP NODA_RCV
0 0 5 0 0 10
REN_ERG_CONS Trade_GDP
19 16
Positional indexing can be used to retrieve individual observations, while logical indexing identifies years in which FDI exceeded an illustrative threshold of 3% of GDP. Vector arithmetic can transform all observations simultaneously. Dividing the percentage variables by 100, for example, converts them from percentage units into proportions. Sudan’s renewable-energy consumption has a mean of approximately 69.3% and a median of approximately 64.3% across the available observations. Its standard deviation is approximately 8.75 percentage points, indicating variation over time. The series also contains four missing observations. Consequently, mean(ren_sdn) returns NA, whereas mean(ren_sdn, na.rm = TRUE) calculates the mean using the available observations.
Missing values are not equally distributed across the variables. FDI_GDP contains 5 missing observations, NODA_RCV contains 10, REN_ERG_CONS contains 19, and Trade_GDP contains 16. Missing observations need to be considered when calculating statistics because many R functions return NA unless missing values are removed explicitly.
Vectorized arithmetic allows the same mathematical operation to be applied to every observation in a variable. The following transformations construct three additional variables from the dataset.
real_industry_output): Approximates industrial value added in constant 2015 US$ by multiplying real GDP by industry’s percentage share of GDPfdi_trade_ratio): Demonstrates vector division by comparing FDI net inflows (% of GDP) with trade (% of GDP). It is constructed here primarily as a programming exercise rather than as a standard economic indicator.log_gdp_constant): Applies the natural logarithm to real GDP, reducing the numerical scale of the variable.# Read the dataset
# 1. Approximate real industrial value added in USD
data$IND_VA_USD <- data$GDP_USD *
(data$IND_VA_GDP / 100)
# 2. FDI-to-trade ratio
data$FDI_Trade_Ratio <- data$FDI_GDP /
data$Trade_GDP
# 3. Natural logarithm of GDP
data$Log_GDP <- log(data$GDP_USD)
# Display selected original and transformed variables
head(
data[, c(
"Country_Code",
"Year",
"GDP_USD",
"IND_VA_GDP",
"IND_VA_USD",
"FDI_Trade_Ratio",
"Log_GDP"
)],
10
) Country_Code Year GDP_USD IND_VA_GDP IND_VA_USD FDI_Trade_Ratio
1 EGY 1995 136218680413 30.2 41138041485 0.019902439
2 EGY 1996 143014263359 29.5 42189207691 0.019169916
3 EGY 1997 150869114035 29.0 43752043070 0.025149613
4 EGY 1998 159280817658 28.6 45554313850 0.031005809
5 EGY 1999 168922784450 28.4 47974070784 0.031281356
6 EGY 2000 179683172290 30.8 55342417065 0.030755087
7 EGY 2001 186035425185 30.9 57484946382 0.012559524
8 EGY 2002 190482051409 32.6 62097148759 0.019518352
9 EGY 2003 196565009504 33.4 65652713174 0.006496369
10 EGY 2004 204608590453 34.7 70999180887 0.027672131
Log_GDP
1 25.63753
2 25.68621
3 25.73968
4 25.79393
5 25.85271
6 25.91446
7 25.94920
8 25.97282
9 26.00426
10 26.04436
These calculations demonstrate vectorized operations because R performs each transformation across all rows of the dataset. IND_VA_USD converts the industrial value-added share of GDP in constant 2015 US$. FDI_Trade_Ratio compares two GDP-relative indicators, while Log_GDP reduces the numerical scale of the Real GDP.
A cross-sectional matrix is constructed for 2011. The rows represent the five countries in the sample (EGY, ETH, KEN, SDN, and UGA), while the columns contain FDI, industrial value added, and renewable energy consumption.
# Filter cross-sectional observations for 2011
data_2011 <- data[data$Year == 2011, ]
# Extract three numeric variables
X_macro <- as.matrix(
data_2011[, c(
"FDI_GDP",
"IND_VA_GDP",
"REN_ERG_CONS"
)]
)
# Assign row and column names
rownames(X_macro) <- data_2011$Country_Code
colnames(X_macro) <- c(
"FDI",
"Ind_VA",
"Ren_ERG"
)
X_macro FDI Ind_VA Ren_ERG
EGY -0.200000 36.0000 5.4
ETH 2.000000 9.7000 93.6
KEN 3.100000 19.7000 74.7
SDN 3.152349 20.4000 63.7
UGA 3.208606 27.5583 93.3
# Matrix dimensions
dim(X_macro)[1] 5 3
# Access matrix elements
X_macro["SDN", "Ren_ERG"][1] 63.7
X_macro[1, 2][1] 36
X_macro["EGY", ] FDI Ind_VA Ren_ERG
-0.2 36.0 5.4
X_macro[, "FDI"] EGY ETH KEN SDN UGA
-0.200000 2.000000 3.100000 3.152349 3.208606
X_macro[1:2, 2:3] Ind_VA Ren_ERG
EGY 36.0 5.4
ETH 9.7 93.6
The resulting matrix has five rows and three columns. Matrix indexing can retrieve a single observation, an entire country’s row, a complete variable column, or a smaller sub-matrix.
Two 2 × 2 matrices are extracted from the 2011 cross-sectional data. Matrix A contains FDI and industrial value added for Egypt and Sudan, while matrix B contains the same indicators for Kenya and Uganda. The matrices contain actual observations from the dataset.
# Create two illustrative 2 x 2 matrices from the 2011 data
# Matrix A: Egypt and Sudan
A <- X_macro[
c("EGY", "SDN"),
c("FDI", "Ind_VA")
]
# Matrix B: Kenya and Uganda
B <- X_macro[
c("KEN", "UGA"),
c("FDI", "Ind_VA")
]
A FDI Ind_VA
EGY -0.200000 36.0
SDN 3.152349 20.4
B FDI Ind_VA
KEN 3.100000 19.7000
UGA 3.208606 27.5583
# Matrix addition
A + B FDI Ind_VA
EGY 2.900000 55.7000
SDN 6.360955 47.9583
# Element-wise multiplication
A * B FDI Ind_VA
EGY -0.62000 709.2000
SDN 10.11464 562.1892
# Matrix multiplication
A %*% B FDI Ind_VA
EGY 114.88982 988.1586
SDN 75.22784 624.2905
# Transpose of A
t(A) EGY SDN
FDI -0.2 3.152349
Ind_VA 36.0 20.400000
# Determinant of A
det(A)[1] -117.5645
# Inverse of A
solve(A) EGY SDN
FDI -0.17352170 0.306214764
Ind_VA 0.02681377 0.001701193
Since matrices A and B have identical 2 × 2 dimensions, they can be added and multiplied element by element. The operator %% instead performs conventional matrix multiplication. These results should be interpreted as demonstrations of matrix algebra rather than economically meaningful composite indicators. Transposing A interchanges its rows and columns. The determinant provides information about whether the matrix is singular. Since the determinant of A is non-zero, the matrix is invertible and solve(A) returns its inverse. Matrix addition and element-wise multiplication operate on corresponding entries of matrices with equal dimensions. In contrast, %% performs matrix multiplication. The transpose interchanges rows and columns, while det() calculates the determinant. Because matrix A has a non-zero determinant, its inverse exists and can be calculated using solve().
col_combined <- cbind(FDI = X_macro[, "FDI"],
Ren_ERG = X_macro[, "Ren_ERG"])
col_combined FDI Ren_ERG
EGY -0.200000 5.4
ETH 2.000000 93.6
KEN 3.100000 74.7
SDN 3.152349 63.7
UGA 3.208606 93.3
# Calculate the mean of each matrix column
apply(X_macro, 2, mean) FDI Ind_VA Ren_ERG
2.252191 22.671659 66.140000
# Calculate the sum of each country's three indicators
apply(X_macro, 1, sum) EGY ETH KEN SDN UGA
41.20000 105.30000 97.50000 87.25235 124.06690
cbind() combines vectors as columns of a new matrix. In apply(), MARGIN = 2 applies a function to columns, so the first apply() calculation produces the mean of each indicator across the five countries. MARGIN = 1 applies the function across rows.
In the following example, a covariance matrix is constructed using three indicators for Egypt: foreign direct investment net inflows (FDI_GDP), industrial value added (IND_VA_GDP), and trade openness (Trade_GDP). Eigenvalue decomposition is then used to examine how the total variation in these three indicators is distributed across the resulting components.
# Extract three indicators for Egypt
egy_indicators <- data[
data$Country_Code == "EGY",
c("FDI_GDP", "IND_VA_GDP", "Trade_GDP")
]
# Check whether these variables contain missing values
colSums(is.na(egy_indicators)) FDI_GDP IND_VA_GDP Trade_GDP
0 0 0
# Construct the covariance matrix
Sigma_egy <- cov(egy_indicators)
round(Sigma_egy, 3) FDI_GDP IND_VA_GDP Trade_GDP
FDI_GDP 8.026 1.549 14.404
IND_VA_GDP 1.549 9.754 6.451
Trade_GDP 14.404 6.451 105.946
# Compute eigenvalues and eigenvectors
eig_macro <- eigen(
Sigma_egy,
symmetric = TRUE
)
eigenvalues <- eig_macro$values
eigenvectors <- eig_macro$vectors
colnames(eigenvectors) <- c(
"Component_1",
"Component_2",
"Component_3"
)
rownames(eigenvectors) <- c(
"FDI_GDP",
"IND_VA_GDP",
"Trade_GDP"
)
eigenvalues[1] 108.462607 9.419989 5.842989
round(eigenvectors, 4) Component_1 Component_2 Component_3
FDI_GDP -0.1427 -0.1627 0.9763
IND_VA_GDP -0.0668 -0.9826 -0.1735
Trade_GDP -0.9875 0.0899 -0.1293
# Verify that the sum of eigenvalues equals the trace
total_variance <- sum(diag(Sigma_egy))
sum_eigenvalues <- sum(eigenvalues)
c(
total_variance = total_variance,
sum_eigenvalues = sum_eigenvalues
) total_variance sum_eigenvalues
123.7256 123.7256
# Percentage of total variance associated with each eigenvalue
var_explained <- (
eigenvalues / total_variance
) * 100
names(var_explained) <- c(
"Component_1",
"Component_2",
"Component_3"
)
round(var_explained, 2)Component_1 Component_2 Component_3
87.66 7.61 4.72
The covariance matrix describes how FDI, industrial value added, and trade vary individually and together for Egypt over the sample period. The eigenvalues are approximately 108.46, 9.42, and 5.84. Relative to their sum, the component associated with the largest eigenvalue accounts for approximately 87.7% of the total variance, while the second and third components account for approximately 7.6% and 4.7%, respectively. The sum of the eigenvalues is equal to the trace of the covariance matrix, providing a numerical check of the decomposition. However, this result should be interpreted cautiously. The analysis uses a covariance matrix, meaning variables with greater numerical variance can have greater influence on the eigenvalues and eigenvectors. Trade_GDP has considerably greater variance than the other two indicators. Therefore, the first component is strongly influenced by variation in trade openness.