This report covers the daily prices of all listed companies in the Taiwan stock market over the last two years (2024/10/07 to 2026/10/05). The data file contains one row per company per trading day with the open, high, low and close prices (NTD) and trading volume (in thousands of shares).
The report:
library(tidyverse) # readr, dplyr, ggplot2, tidyr
library(lubridate) # dates
library(scales) # axis formatting
library(knitr) # kable tables
library(DT) # interactive tables
library(plotly) # interactive charts
The CSV file is stored in the data/ folder next to this
.Rmd file.
data_file <- "data/Taiwan_Stocks_2024_2026_FINAL.csv"
if (!file.exists(data_file)) {
stop("Data file not found: ", data_file,
"\nCurrent working directory is: ", getwd(),
"\nPut the CSV inside a 'data' folder next to this .Rmd file.")
}
raw <- read_csv(
data_file,
col_types = cols(
COID = col_character(), # keep stock codes as text
COID_Name = col_character(),
Date = col_character(),
.default = col_double()
)
)
glimpse(raw)
## Rows: 957,924
## Columns: 8
## $ COID <chr> "1101", "1101", "1101", "1101", "1101", "1101", "1101"…
## $ COID_Name <chr> "TCC", "TCC", "TCC", "TCC", "TCC", "TCC", "TCC", "TCC"…
## $ Date <chr> "2024/10/07", "2024/10/08", "2024/10/09", "2024/10/11"…
## $ `Open(NTD)` <dbl> 31.4261, 30.9148, 30.5893, 30.4499, 30.2639, 30.3104, …
## $ `High(NTD)` <dbl> 31.4726, 31.1007, 30.6358, 30.6823, 30.4034, 30.4034, …
## $ `Low(NTD)` <dbl> 30.9612, 30.5893, 30.1709, 30.1245, 30.1709, 30.1709, …
## $ `Close(NTD)` <dbl> 31.1472, 30.7288, 30.1709, 30.2174, 30.2639, 30.2174, …
## $ `Volume(1000S)` <dbl> 13191, 14236, 11396, 9445, 6059, 8102, 10435, 7485, 10…
stocks <- raw %>%
rename(
coid = COID,
name = COID_Name,
date = Date,
open = `Open(NTD)`,
high = `High(NTD)`,
low = `Low(NTD)`,
close = `Close(NTD)`,
volume = `Volume(1000S)`
) %>%
mutate(date = ymd(date)) %>%
distinct(coid, date, .keep_all = TRUE) %>% # drop duplicate stock-days
arrange(coid, date)
head(stocks, 10) %>% kable(caption = "First rows of the cleaned data")
| coid | name | date | open | high | low | close | volume |
|---|---|---|---|---|---|---|---|
| 1101 | TCC | 2024-10-07 | 31.4261 | 31.4726 | 30.9612 | 31.1472 | 13191 |
| 1101 | TCC | 2024-10-08 | 30.9148 | 31.1007 | 30.5893 | 30.7288 | 14236 |
| 1101 | TCC | 2024-10-09 | 30.5893 | 30.6358 | 30.1709 | 30.1709 | 11396 |
| 1101 | TCC | 2024-10-11 | 30.4499 | 30.6823 | 30.1245 | 30.2174 | 9445 |
| 1101 | TCC | 2024-10-14 | 30.2639 | 30.4034 | 30.1709 | 30.2639 | 6059 |
| 1101 | TCC | 2024-10-15 | 30.3104 | 30.4034 | 30.1709 | 30.2174 | 8102 |
| 1101 | TCC | 2024-10-16 | 30.2174 | 30.2174 | 29.8920 | 29.8920 | 10435 |
| 1101 | TCC | 2024-10-17 | 29.9385 | 30.3104 | 29.9385 | 30.2174 | 7485 |
| 1101 | TCC | 2024-10-18 | 30.3104 | 30.4499 | 29.9385 | 30.4499 | 10399 |
| 1101 | TCC | 2024-10-21 | 30.5893 | 30.5893 | 30.0315 | 30.0315 | 9212 |
quality <- tibble(
Check = c("Total rows",
"Number of companies / securities",
"First date",
"Last date",
"Number of trading days",
"Missing values (all columns)",
"Rows with close <= 0",
"Rows where high < low"),
Result = c(
comma(nrow(stocks)),
comma(n_distinct(stocks$coid)),
as.character(min(stocks$date)),
as.character(max(stocks$date)),
comma(n_distinct(stocks$date)),
comma(sum(is.na(stocks))),
comma(sum(stocks$close <= 0, na.rm = TRUE)),
comma(sum(stocks$high < stocks$low, na.rm = TRUE))
)
)
kable(quality, caption = "Data quality summary")
| Check | Result |
|---|---|
| Total rows | 957,924 |
| Number of companies / securities | 1,987 |
| First date | 2024-10-07 |
| Last date | 2026-10-05 |
| Number of trading days | 485 |
| Missing values (all columns) | 0 |
| Rows with close <= 0 | 0 |
| Rows where high < low | 0 |
daily_count <- stocks %>% count(date, name = "n_stocks")
ggplot(daily_count, aes(date, n_stocks)) +
geom_line(color = "#2c7fb8") +
labs(title = "Number of Stocks Traded per Day",
x = NULL, y = "Number of stocks") +
theme_minimal()
Trading value is approximated as close x volume x 1000
(NTD).
market_value <- stocks %>%
mutate(value = close * volume * 1000) %>%
group_by(date) %>%
summarise(total_value = sum(value, na.rm = TRUE), .groups = "drop")
ggplot(market_value, aes(date, total_value / 1e9)) +
geom_col(fill = "#41b6c4") +
labs(title = "Total Daily Trading Value (all stocks)",
x = NULL, y = "NTD billions") +
theme_minimal()
Daily returns are computed for every stock, averaged across stocks, and compounded into an index starting at 100. Extreme daily moves (above 50% in absolute value, usually caused by new listings or data gaps) are excluded.
returns <- stocks %>%
group_by(coid) %>%
arrange(date, .by_group = TRUE) %>%
mutate(ret = close / lag(close) - 1) %>%
ungroup() %>%
filter(!is.na(ret), abs(ret) < 0.5)
market_index <- returns %>%
group_by(date) %>%
summarise(mkt_ret = mean(ret), .groups = "drop") %>%
arrange(date) %>%
mutate(index = 100 * cumprod(1 + mkt_ret))
ggplot(market_index, aes(date, index)) +
geom_line(color = "#d95f0e", linewidth = 0.9) +
geom_hline(yintercept = 100, linetype = "dashed", color = "grey50") +
labs(title = "Equal-Weighted Taiwan Market Index (base = 100)",
x = NULL, y = "Index") +
theme_minimal()
stock_summary <- stocks %>%
group_by(coid, name) %>%
summarise(
n_days = n(),
first_date = min(date),
last_date = max(date),
first_close = first(close),
last_close = last(close),
total_return = last_close / first_close - 1,
avg_volume = mean(volume),
avg_value = mean(close * volume * 1000),
.groups = "drop"
)
stock_summary %>%
mutate(total_return = round(total_return * 100, 1),
avg_volume = round(avg_volume),
avg_value = round(avg_value / 1e6, 1)) %>%
select(Code = coid, Name = name, Days = n_days,
`First close` = first_close, `Last close` = last_close,
`Return (%)` = total_return,
`Avg volume (1000s)` = avg_volume,
`Avg value (NTD m)` = avg_value) %>%
datatable(filter = "top", rownames = FALSE,
caption = "Search and sort all stocks")
top_value <- stock_summary %>%
slice_max(avg_value, n = 10) %>%
mutate(label = paste(coid, name))
ggplot(top_value, aes(reorder(label, avg_value), avg_value / 1e9)) +
geom_col(fill = "#2c7fb8") +
coord_flip() +
labs(title = "Top 10 Stocks by Average Daily Trading Value",
x = NULL, y = "NTD billions per day") +
theme_minimal()
Only stocks that traded for at least 90% of all trading days are included, so that returns are comparable over the full two years.
full_period <- stock_summary %>%
filter(n_days >= 0.9 * max(n_days))
best <- full_period %>% slice_max(total_return, n = 10)
worst <- full_period %>% slice_min(total_return, n = 10)
bind_rows(Best = best, Worst = worst, .id = "group") %>%
mutate(label = paste(coid, name),
group = factor(group, levels = c("Best", "Worst"))) %>%
ggplot(aes(reorder(label, total_return), total_return, fill = group)) +
geom_col() +
coord_flip() +
scale_y_continuous(labels = percent) +
scale_fill_manual(values = c(Best = "#1a9850", Worst = "#d73027")) +
labs(title = "Top 10 Best and Worst Performers (Oct 2024 - Oct 2026)",
x = NULL, y = "Total return", fill = NULL) +
theme_minimal()
tsmc <- stocks %>% filter(coid == "2330")
plot_ly(tsmc, x = ~date, type = "candlestick",
open = ~open, high = ~high, low = ~low, close = ~close) %>%
layout(title = "TSMC (2330) Daily Prices (NTD)",
xaxis = list(rangeslider = list(visible = FALSE)),
yaxis = list(title = "Price (NTD)"))
Prices are rebased to 100 on each stock’s first trading day.
top5 <- stock_summary %>% slice_max(avg_value, n = 5) %>% pull(coid)
stocks %>%
filter(coid %in% top5) %>%
group_by(coid, name) %>%
mutate(rebased = close / first(close) * 100,
label = paste(coid, name)) %>%
ungroup() %>%
ggplot(aes(date, rebased, color = label)) +
geom_line(linewidth = 0.8) +
labs(title = "Price Performance of the Five Most Traded Stocks (start = 100)",
x = NULL, y = "Rebased price", color = NULL) +
theme_minimal() +
theme(legend.position = "bottom")
ggplot(returns %>% filter(abs(ret) < 0.11), aes(ret)) +
geom_histogram(bins = 120, fill = "#2c7fb8", color = "white") +
scale_x_continuous(labels = percent) +
labs(title = "Distribution of Daily Stock Returns (all stocks)",
x = "Daily return", y = "Count") +
theme_minimal()
sessionInfo()
## R version 4.6.1 (2026-06-24 ucrt)
## Platform: x86_64-w64-mingw32/x64
## Running under: Windows 11 x64 (build 26200)
##
## Matrix products: default
## LAPACK version 3.12.1
##
## locale:
## [1] LC_COLLATE=English_United States.utf8
## [2] LC_CTYPE=English_United States.utf8
## [3] LC_MONETARY=English_United States.utf8
## [4] LC_NUMERIC=C
## [5] LC_TIME=English_United States.utf8
##
## time zone: Asia/Ulaanbaatar
## tzcode source: internal
##
## attached base packages:
## [1] stats graphics grDevices utils datasets methods base
##
## other attached packages:
## [1] plotly_4.12.1 DT_0.34.0 knitr_1.52 scales_1.4.0
## [5] lubridate_1.9.5 forcats_1.0.1 stringr_1.6.0 dplyr_1.2.1
## [9] purrr_1.2.2 readr_2.2.0 tidyr_1.3.2 tibble_3.3.1
## [13] ggplot2_4.0.3 tidyverse_2.0.0
##
## loaded via a namespace (and not attached):
## [1] sass_0.4.10 generics_0.1.4 stringi_1.8.9
## [4] hms_1.1.4 digest_0.6.39 magrittr_2.0.5
## [7] evaluate_1.0.5 grid_4.6.1 timechange_0.4.0
## [10] RColorBrewer_1.1-3 fastmap_1.2.0 jsonlite_2.0.0
## [13] httr_1.4.9 crosstalk_1.2.2 viridisLite_0.4.3
## [16] jquerylib_0.1.4 cli_3.6.6 crayon_1.5.3
## [19] rlang_1.3.0 bit64_4.8.6 withr_3.0.3
## [22] cachem_1.1.0 yaml_2.3.12 otel_0.2.0
## [25] parallel_4.6.1 tools_4.6.1 tzdb_0.5.0
## [28] vctrs_0.7.3 R6_2.6.1 lifecycle_1.0.5
## [31] htmlwidgets_1.6.4 bit_4.6.0 vroom_1.7.1
## [34] pkgconfig_2.0.3 pillar_1.11.1 bslib_0.12.0
## [37] gtable_0.3.6 glue_1.8.1 data.table_1.18.6.1
## [40] xfun_0.60 tidyselect_1.2.1 rstudioapi_0.19.0
## [43] farver_2.1.2 htmltools_0.5.9 labeling_0.4.3
## [46] rmarkdown_2.32 compiler_4.6.1 S7_0.2.2