1 Introduction

This report covers the daily prices of all listed companies in the Taiwan stock market over the last two years (2024/10/07 to 2026/10/05). The data file contains one row per company per trading day with the open, high, low and close prices (NTD) and trading volume (in thousands of shares).

The report:

  1. loads and cleans the downloaded data,
  2. checks its quality,
  3. summarizes the whole market, and
  4. visualizes individual stocks and the overall market trend.

2 Load Packages

library(tidyverse)   # readr, dplyr, ggplot2, tidyr
library(lubridate)   # dates
library(scales)      # axis formatting
library(knitr)       # kable tables
library(DT)          # interactive tables
library(plotly)      # interactive charts

3 Load the Data

The CSV file is stored in the data/ folder next to this .Rmd file.

data_file <- "data/Taiwan_Stocks_2024_2026_FINAL.csv"

if (!file.exists(data_file)) {
  stop("Data file not found: ", data_file,
       "\nCurrent working directory is: ", getwd(),
       "\nPut the CSV inside a 'data' folder next to this .Rmd file.")
}

raw <- read_csv(
  data_file,
  col_types = cols(
    COID       = col_character(),   # keep stock codes as text
    COID_Name  = col_character(),
    Date       = col_character(),
    .default   = col_double()
  )
)

glimpse(raw)
## Rows: 957,924
## Columns: 8
## $ COID            <chr> "1101", "1101", "1101", "1101", "1101", "1101", "1101"…
## $ COID_Name       <chr> "TCC", "TCC", "TCC", "TCC", "TCC", "TCC", "TCC", "TCC"…
## $ Date            <chr> "2024/10/07", "2024/10/08", "2024/10/09", "2024/10/11"…
## $ `Open(NTD)`     <dbl> 31.4261, 30.9148, 30.5893, 30.4499, 30.2639, 30.3104, …
## $ `High(NTD)`     <dbl> 31.4726, 31.1007, 30.6358, 30.6823, 30.4034, 30.4034, …
## $ `Low(NTD)`      <dbl> 30.9612, 30.5893, 30.1709, 30.1245, 30.1709, 30.1709, …
## $ `Close(NTD)`    <dbl> 31.1472, 30.7288, 30.1709, 30.2174, 30.2639, 30.2174, …
## $ `Volume(1000S)` <dbl> 13191, 14236, 11396, 9445, 6059, 8102, 10435, 7485, 10…

4 Clean the Data

stocks <- raw %>%
  rename(
    coid   = COID,
    name   = COID_Name,
    date   = Date,
    open   = `Open(NTD)`,
    high   = `High(NTD)`,
    low    = `Low(NTD)`,
    close  = `Close(NTD)`,
    volume = `Volume(1000S)`
  ) %>%
  mutate(date = ymd(date)) %>%
  distinct(coid, date, .keep_all = TRUE) %>%   # drop duplicate stock-days
  arrange(coid, date)

head(stocks, 10) %>% kable(caption = "First rows of the cleaned data")
First rows of the cleaned data
coid name date open high low close volume
1101 TCC 2024-10-07 31.4261 31.4726 30.9612 31.1472 13191
1101 TCC 2024-10-08 30.9148 31.1007 30.5893 30.7288 14236
1101 TCC 2024-10-09 30.5893 30.6358 30.1709 30.1709 11396
1101 TCC 2024-10-11 30.4499 30.6823 30.1245 30.2174 9445
1101 TCC 2024-10-14 30.2639 30.4034 30.1709 30.2639 6059
1101 TCC 2024-10-15 30.3104 30.4034 30.1709 30.2174 8102
1101 TCC 2024-10-16 30.2174 30.2174 29.8920 29.8920 10435
1101 TCC 2024-10-17 29.9385 30.3104 29.9385 30.2174 7485
1101 TCC 2024-10-18 30.3104 30.4499 29.9385 30.4499 10399
1101 TCC 2024-10-21 30.5893 30.5893 30.0315 30.0315 9212

4.1 Data quality checks

quality <- tibble(
  Check = c("Total rows",
            "Number of companies / securities",
            "First date",
            "Last date",
            "Number of trading days",
            "Missing values (all columns)",
            "Rows with close <= 0",
            "Rows where high < low"),
  Result = c(
    comma(nrow(stocks)),
    comma(n_distinct(stocks$coid)),
    as.character(min(stocks$date)),
    as.character(max(stocks$date)),
    comma(n_distinct(stocks$date)),
    comma(sum(is.na(stocks))),
    comma(sum(stocks$close <= 0, na.rm = TRUE)),
    comma(sum(stocks$high < stocks$low, na.rm = TRUE))
  )
)

kable(quality, caption = "Data quality summary")
Data quality summary
Check Result
Total rows 957,924
Number of companies / securities 1,987
First date 2024-10-07
Last date 2026-10-05
Number of trading days 485
Missing values (all columns) 0
Rows with close <= 0 0
Rows where high < low 0

5 Market Overview

5.1 Number of stocks traded each day

daily_count <- stocks %>% count(date, name = "n_stocks")

ggplot(daily_count, aes(date, n_stocks)) +
  geom_line(color = "#2c7fb8") +
  labs(title = "Number of Stocks Traded per Day",
       x = NULL, y = "Number of stocks") +
  theme_minimal()

5.2 Total market trading value

Trading value is approximated as close x volume x 1000 (NTD).

market_value <- stocks %>%
  mutate(value = close * volume * 1000) %>%
  group_by(date) %>%
  summarise(total_value = sum(value, na.rm = TRUE), .groups = "drop")

ggplot(market_value, aes(date, total_value / 1e9)) +
  geom_col(fill = "#41b6c4") +
  labs(title = "Total Daily Trading Value (all stocks)",
       x = NULL, y = "NTD billions") +
  theme_minimal()

5.3 Equal-weighted market index

Daily returns are computed for every stock, averaged across stocks, and compounded into an index starting at 100. Extreme daily moves (above 50% in absolute value, usually caused by new listings or data gaps) are excluded.

returns <- stocks %>%
  group_by(coid) %>%
  arrange(date, .by_group = TRUE) %>%
  mutate(ret = close / lag(close) - 1) %>%
  ungroup() %>%
  filter(!is.na(ret), abs(ret) < 0.5)

market_index <- returns %>%
  group_by(date) %>%
  summarise(mkt_ret = mean(ret), .groups = "drop") %>%
  arrange(date) %>%
  mutate(index = 100 * cumprod(1 + mkt_ret))

ggplot(market_index, aes(date, index)) +
  geom_line(color = "#d95f0e", linewidth = 0.9) +
  geom_hline(yintercept = 100, linetype = "dashed", color = "grey50") +
  labs(title = "Equal-Weighted Taiwan Market Index (base = 100)",
       x = NULL, y = "Index") +
  theme_minimal()

6 Stock-Level Summary

stock_summary <- stocks %>%
  group_by(coid, name) %>%
  summarise(
    n_days       = n(),
    first_date   = min(date),
    last_date    = max(date),
    first_close  = first(close),
    last_close   = last(close),
    total_return = last_close / first_close - 1,
    avg_volume   = mean(volume),
    avg_value    = mean(close * volume * 1000),
    .groups = "drop"
  )

stock_summary %>%
  mutate(total_return = round(total_return * 100, 1),
         avg_volume   = round(avg_volume),
         avg_value    = round(avg_value / 1e6, 1)) %>%
  select(Code = coid, Name = name, Days = n_days,
         `First close` = first_close, `Last close` = last_close,
         `Return (%)` = total_return,
         `Avg volume (1000s)` = avg_volume,
         `Avg value (NTD m)` = avg_value) %>%
  datatable(filter = "top", rownames = FALSE,
            caption = "Search and sort all stocks")

6.1 Most actively traded stocks

top_value <- stock_summary %>%
  slice_max(avg_value, n = 10) %>%
  mutate(label = paste(coid, name))

ggplot(top_value, aes(reorder(label, avg_value), avg_value / 1e9)) +
  geom_col(fill = "#2c7fb8") +
  coord_flip() +
  labs(title = "Top 10 Stocks by Average Daily Trading Value",
       x = NULL, y = "NTD billions per day") +
  theme_minimal()

6.2 Best and worst performers

Only stocks that traded for at least 90% of all trading days are included, so that returns are comparable over the full two years.

full_period <- stock_summary %>%
  filter(n_days >= 0.9 * max(n_days))

best  <- full_period %>% slice_max(total_return, n = 10)
worst <- full_period %>% slice_min(total_return, n = 10)

bind_rows(Best = best, Worst = worst, .id = "group") %>%
  mutate(label = paste(coid, name),
         group = factor(group, levels = c("Best", "Worst"))) %>%
  ggplot(aes(reorder(label, total_return), total_return, fill = group)) +
  geom_col() +
  coord_flip() +
  scale_y_continuous(labels = percent) +
  scale_fill_manual(values = c(Best = "#1a9850", Worst = "#d73027")) +
  labs(title = "Top 10 Best and Worst Performers (Oct 2024 - Oct 2026)",
       x = NULL, y = "Total return", fill = NULL) +
  theme_minimal()

7 Individual Stock Examples

7.1 TSMC (2330): candlestick chart

tsmc <- stocks %>% filter(coid == "2330")

plot_ly(tsmc, x = ~date, type = "candlestick",
        open = ~open, high = ~high, low = ~low, close = ~close) %>%
  layout(title = "TSMC (2330) Daily Prices (NTD)",
         xaxis = list(rangeslider = list(visible = FALSE)),
         yaxis = list(title = "Price (NTD)"))

7.2 Comparing the five largest stocks

Prices are rebased to 100 on each stock’s first trading day.

top5 <- stock_summary %>% slice_max(avg_value, n = 5) %>% pull(coid)

stocks %>%
  filter(coid %in% top5) %>%
  group_by(coid, name) %>%
  mutate(rebased = close / first(close) * 100,
         label   = paste(coid, name)) %>%
  ungroup() %>%
  ggplot(aes(date, rebased, color = label)) +
  geom_line(linewidth = 0.8) +
  labs(title = "Price Performance of the Five Most Traded Stocks (start = 100)",
       x = NULL, y = "Rebased price", color = NULL) +
  theme_minimal() +
  theme(legend.position = "bottom")

7.3 Distribution of daily returns

ggplot(returns %>% filter(abs(ret) < 0.11), aes(ret)) +
  geom_histogram(bins = 120, fill = "#2c7fb8", color = "white") +
  scale_x_continuous(labels = percent) +
  labs(title = "Distribution of Daily Stock Returns (all stocks)",
       x = "Daily return", y = "Count") +
  theme_minimal()

8 Conclusion

  • The dataset contains 957,924 daily observations for 1,987 securities between 2024-10-07 and 2026-10-05.
  • The equal-weighted market index ended the period at about 118.7 (base = 100).
  • Trading activity is highly concentrated: the ten most traded stocks account for a large share of total market value.
  • Daily returns are centered near zero, with fat tails and a visible limit at about +/-10%, consistent with the daily price limit of the Taiwan market.

9 Session Information

sessionInfo()
## R version 4.6.1 (2026-06-24 ucrt)
## Platform: x86_64-w64-mingw32/x64
## Running under: Windows 11 x64 (build 26200)
## 
## Matrix products: default
##   LAPACK version 3.12.1
## 
## locale:
## [1] LC_COLLATE=English_United States.utf8 
## [2] LC_CTYPE=English_United States.utf8   
## [3] LC_MONETARY=English_United States.utf8
## [4] LC_NUMERIC=C                          
## [5] LC_TIME=English_United States.utf8    
## 
## time zone: Asia/Ulaanbaatar
## tzcode source: internal
## 
## attached base packages:
## [1] stats     graphics  grDevices utils     datasets  methods   base     
## 
## other attached packages:
##  [1] plotly_4.12.1   DT_0.34.0       knitr_1.52      scales_1.4.0   
##  [5] lubridate_1.9.5 forcats_1.0.1   stringr_1.6.0   dplyr_1.2.1    
##  [9] purrr_1.2.2     readr_2.2.0     tidyr_1.3.2     tibble_3.3.1   
## [13] ggplot2_4.0.3   tidyverse_2.0.0
## 
## loaded via a namespace (and not attached):
##  [1] sass_0.4.10         generics_0.1.4      stringi_1.8.9      
##  [4] hms_1.1.4           digest_0.6.39       magrittr_2.0.5     
##  [7] evaluate_1.0.5      grid_4.6.1          timechange_0.4.0   
## [10] RColorBrewer_1.1-3  fastmap_1.2.0       jsonlite_2.0.0     
## [13] httr_1.4.9          crosstalk_1.2.2     viridisLite_0.4.3  
## [16] jquerylib_0.1.4     cli_3.6.6           crayon_1.5.3       
## [19] rlang_1.3.0         bit64_4.8.6         withr_3.0.3        
## [22] cachem_1.1.0        yaml_2.3.12         otel_0.2.0         
## [25] parallel_4.6.1      tools_4.6.1         tzdb_0.5.0         
## [28] vctrs_0.7.3         R6_2.6.1            lifecycle_1.0.5    
## [31] htmlwidgets_1.6.4   bit_4.6.0           vroom_1.7.1        
## [34] pkgconfig_2.0.3     pillar_1.11.1       bslib_0.12.0       
## [37] gtable_0.3.6        glue_1.8.1          data.table_1.18.6.1
## [40] xfun_0.60           tidyselect_1.2.1    rstudioapi_0.19.0  
## [43] farver_2.1.2        htmltools_0.5.9     labeling_0.4.3     
## [46] rmarkdown_2.32      compiler_4.6.1      S7_0.2.2