MDS, t-SNE and UMAP

This vignette demonstrates dimensionality reduction methods available in tidymatrix: MDS, t-SNE, and UMAP.

library(tidymatrix)
library(dplyr)

Creating Example Data

set.seed(42)
n_genes <- 100
n_samples <- 20

# Create expression matrix
expression <- matrix(rnorm(n_genes * n_samples), nrow = n_genes, ncol = n_samples)

# Add structure
expression[1:30, 1:10] <- expression[1:30, 1:10] + 2
expression[31:60, 11:20] <- expression[31:60, 11:20] + 2

# Metadata
gene_data <- data.frame(
  gene_id = paste0("Gene_", 1:n_genes),
  category = rep(c("TypeA", "TypeB", "TypeC"), length.out = n_genes)
)

sample_data <- data.frame(
  sample_id = paste0("Sample_", 1:n_samples),
  condition = rep(c("Control", "Treatment"), each = 10),
  batch = rep(c("B1", "B2"), 10)
)

tm <- tidymatrix(expression, gene_data, sample_data)

Classical Multidimensional Scaling (MDS)

MDS is a fast, deterministic method for dimensionality reduction:

tm_mds <- tm |>
  activate(columns) |>
  compute_mds(k = 2, name = "sample_mds")

# Check added columns
cat("MDS coordinates added:\n")
#> MDS coordinates added:
print(names(tm_mds$col_data))
#> [1] "sample_id"    "condition"    "batch"        "sample_mds_1" "sample_mds_2"

# View the results
head(tm_mds$col_data)
#>   sample_id condition batch sample_mds_1 sample_mds_2
#> 1  Sample_1   Control    B1    -8.617083  0.002052457
#> 2  Sample_2   Control    B2    -7.004946  3.105104315
#> 3  Sample_3   Control    B1    -7.633199 -0.746203884
#> 4  Sample_4   Control    B2    -8.176460 -2.586303525
#> 5  Sample_5   Control    B1    -6.682964 -4.795454877
#> 6  Sample_6   Control    B2    -7.932464 -6.706727458

Accessing Stored MDS Object

# Get full MDS result
mds_obj <- get_analysis(tm_mds, "sample_mds")
cat("MDS object components:", names(mds_obj), "\n")
#> MDS object components: dist mds

t-SNE (t-distributed Stochastic Neighbor Embedding)

t-SNE excels at revealing local structure in high-dimensional data:

set.seed(42)
tm_tsne <- tm |>
  activate(columns) |>
  compute_tsne(dims = 2, name = "sample_tsne", perplexity = 5, verbose = FALSE)

# View results
head(tm_tsne$col_data)
#>   sample_id condition batch sample_tsne_1 sample_tsne_2
#> 1  Sample_1   Control    B1      31.18318      64.22357
#> 2  Sample_2   Control    B2      43.44043      62.46839
#> 3  Sample_3   Control    B1      49.95858      77.22997
#> 4  Sample_4   Control    B2      36.97408      70.83221
#> 5  Sample_5   Control    B1      32.58468      75.68345
#> 6  Sample_6   Control    B2      38.67922      83.34465

Note: t-SNE is stochastic. Use set.seed() for reproducibility.

t-SNE Parameters

The perplexity parameter controls local vs global structure:

# Lower perplexity emphasizes local structure
set.seed(42)
tm_tsne_low <- tm |>
  activate(columns) |>
  compute_tsne(dims = 2, name = "tsne_low", perplexity = 3,
               verbose = FALSE, check_duplicates = FALSE)

# Higher perplexity preserves more global structure. Rtsne requires
# 3 * perplexity < n - 1, so with 20 samples the ceiling is 6.
set.seed(42)
tm_tsne_high <- tm |>
  activate(columns) |>
  compute_tsne(dims = 2, name = "tsne_high", perplexity = 6,
               verbose = FALSE, check_duplicates = FALSE)

cat("Different perplexity values create different embeddings\n")
#> Different perplexity values create different embeddings

UMAP (Uniform Manifold Approximation and Projection)

UMAP is faster than t-SNE and often better preserves global structure:

tm_umap <- tm |>
  activate(columns) |>
  compute_umap(n_components = 2, name = "sample_umap",
               n_neighbors = 10, random_state = 42)

# View results
head(tm_umap$col_data)
#>   sample_id condition batch sample_umap_1 sample_umap_2
#> 1  Sample_1   Control    B1     -7.517308    -1.1444799
#> 2  Sample_2   Control    B2     -7.788300    -0.7665247
#> 3  Sample_3   Control    B1     -8.399422    -0.9482957
#> 4  Sample_4   Control    B2     -7.976027    -1.3868782
#> 5  Sample_5   Control    B1     -8.264245    -1.4191702
#> 6  Sample_6   Control    B2     -8.706839    -1.6707032

UMAP Parameters

tm_umap_custom <- tm |>
  activate(columns) |>
  compute_umap(n_components = 2, n_neighbors = 5, min_dist = 0.05,
               random_state = 42, name = "umap_tight")

cat("Custom UMAP parameters create tighter clustering\n")
#> Custom UMAP parameters create tighter clustering

Comparing Methods

All three methods can be applied to the same data:

tm_all <- tm |>
  activate(columns) |>
  compute_mds(k = 2, name = "mds")

# Add t-SNE and UMAP if packages available
if (requireNamespace("Rtsne", quietly = TRUE)) {
  set.seed(42)
  tm_all <- tm_all |>
    compute_tsne(dims = 2, name = "tsne", perplexity = 5,
                 verbose = FALSE, check_duplicates = FALSE)
}

if (requireNamespace("umap", quietly = TRUE)) {
  tm_all <- tm_all |>
    compute_umap(n_components = 2, name = "umap",
                 n_neighbors = 10, random_state = 42)
}

# View all coordinates
head(tm_all$col_data)
#>   sample_id condition batch     mds_1        mds_2   tsne_1   tsne_2    umap_1
#> 1  Sample_1   Control    B1 -8.617083  0.002052457 31.18318 64.22357 -7.517308
#> 2  Sample_2   Control    B2 -7.004946  3.105104315 43.44043 62.46839 -7.788300
#> 3  Sample_3   Control    B1 -7.633199 -0.746203884 49.95858 77.22997 -8.399422
#> 4  Sample_4   Control    B2 -8.176460 -2.586303525 36.97408 70.83221 -7.976027
#> 5  Sample_5   Control    B1 -6.682964 -4.795454877 32.58468 75.68345 -8.264245
#> 6  Sample_6   Control    B2 -7.932464 -6.706727458 38.67922 83.34465 -8.706839
#>       umap_2
#> 1 -1.1444799
#> 2 -0.7665247
#> 3 -0.9482957
#> 4 -1.3868782
#> 5 -1.4191702
#> 6 -1.6707032

Gene-Level Dimensionality Reduction

Dimensionality reduction works on rows (genes) too:

tm_genes <- tm |>
  activate(rows) |>
  compute_mds(k = 2, name = "gene_mds")

# Filter genes by their embedding
interesting_genes <- tm_genes |>
  activate(rows) |>
  filter(gene_mds_1 > median(gene_mds_1))
#> Warning: Removed 1 stored analysis object(s) due to filter: gene_mds
#> Metadata columns are preserved.

cat("Genes in right half of MDS space:", nrow(interesting_genes$matrix), "\n")
#> Genes in right half of MDS space: 50

Multiple Dimensions

Request more than 2 dimensions for comprehensive analysis:

tm_3d <- tm |>
  activate(columns) |>
  compute_mds(k = 3, name = "mds3d")

# Check all three dimensions
cat("MDS columns:", grep("mds3d", names(tm_3d$col_data), value = TRUE), "\n")
#> MDS columns: mds3d_1 mds3d_2 mds3d_3

Using with to_long()

Combine dimensionality reduction with to_long() for visualization:

# After computing dimensionality reduction
tm |>
  activate(columns) |>
  compute_mds(k = 2) |>
  to_long() |>
  ggplot(aes(x = column_mds_1, y = column_mds_2, color = condition)) +
  geom_point()

Summary

tidymatrix provides three dimensionality reduction methods:

Method Speed Deterministic Best For
MDS Fast Yes Quick exploration, no dependencies
t-SNE Slow No (use seed) Local structure, visualization
UMAP Medium No (use seed) Balanced local/global, faster than t-SNE

When to use each: - MDS: Quick exploration, always available - t-SNE: Publication-quality visualizations, small to medium datasets - UMAP: Large datasets, need for global structure

All methods integrate seamlessly with tidymatrix operations and can be applied to both rows and columns.