This vignette demonstrates dimensionality reduction methods available in tidymatrix: MDS, t-SNE, and UMAP.
library(tidymatrix)
library(dplyr)
#>
#> Attaching package: 'dplyr'
#> The following objects are masked from 'package:stats':
#>
#> filter, lag
#> The following objects are masked from 'package:base':
#>
#> intersect, setdiff, setequal, unionCreating Example Data
set.seed(42)
n_genes <- 100
n_samples <- 20
# Create expression matrix
expression <- matrix(rnorm(n_genes * n_samples), nrow = n_genes, ncol = n_samples)
# Add structure
expression[1:30, 1:10] <- expression[1:30, 1:10] + 2
expression[31:60, 11:20] <- expression[31:60, 11:20] + 2
# Metadata
gene_data <- data.frame(
gene_id = paste0("Gene_", 1:n_genes),
category = rep(c("TypeA", "TypeB", "TypeC"), length.out = n_genes)
)
sample_data <- data.frame(
sample_id = paste0("Sample_", 1:n_samples),
condition = rep(c("Control", "Treatment"), each = 10),
batch = rep(c("B1", "B2"), 10)
)
tm <- tidymatrix(expression, gene_data, sample_data)Classical Multidimensional Scaling (MDS)
MDS is a fast, deterministic method for dimensionality reduction:
tm_mds <- tm |>
activate(columns) |>
compute_mds(k = 2, name = "sample_mds")
# Check added columns
cat("MDS coordinates added:\n")
#> MDS coordinates added:
print(names(tm_mds$col_data))
#> [1] "sample_id" "condition" "batch" "sample_mds_1" "sample_mds_2"
# View the results
head(tm_mds$col_data)
#> sample_id condition batch sample_mds_1 sample_mds_2
#> 1 Sample_1 Control B1 -8.617083 0.002052457
#> 2 Sample_2 Control B2 -7.004946 3.105104315
#> 3 Sample_3 Control B1 -7.633199 -0.746203884
#> 4 Sample_4 Control B2 -8.176460 -2.586303525
#> 5 Sample_5 Control B1 -6.682964 -4.795454877
#> 6 Sample_6 Control B2 -7.932464 -6.706727458Accessing Stored MDS Object
# Get full MDS result
mds_obj <- get_analysis(tm_mds, "sample_mds")
cat("MDS object components:", names(mds_obj), "\n")
#> MDS object components: dist mdst-SNE (t-distributed Stochastic Neighbor Embedding)
t-SNE excels at revealing local structure in high-dimensional data:
set.seed(42)
tm_tsne <- tm |>
activate(columns) |>
compute_tsne(dims = 2, name = "sample_tsne", perplexity = 5, verbose = FALSE)
# View results
head(tm_tsne$col_data)
#> sample_id condition batch sample_tsne_1 sample_tsne_2
#> 1 Sample_1 Control B1 70.78583 13.41886
#> 2 Sample_2 Control B2 72.04628 17.51309
#> 3 Sample_3 Control B1 68.59004 22.29858
#> 4 Sample_4 Control B2 68.59600 16.28628
#> 5 Sample_5 Control B1 67.10259 14.63846
#> 6 Sample_6 Control B2 64.11826 16.00315Note: t-SNE is stochastic. Use set.seed() for
reproducibility.
t-SNE Parameters
The perplexity parameter controls local vs global
structure:
# Lower perplexity emphasizes local structure
set.seed(42)
tm_tsne_low <- tm |>
activate(columns) |>
compute_tsne(dims = 2, name = "tsne_low", perplexity = 3,
verbose = FALSE, check_duplicates = FALSE)
# Higher perplexity preserves more global structure. Rtsne requires
# 3 * perplexity < n - 1, so with 20 samples the ceiling is 6.
set.seed(42)
tm_tsne_high <- tm |>
activate(columns) |>
compute_tsne(dims = 2, name = "tsne_high", perplexity = 6,
verbose = FALSE, check_duplicates = FALSE)
cat("Different perplexity values create different embeddings\n")
#> Different perplexity values create different embeddingsUMAP (Uniform Manifold Approximation and Projection)
UMAP is faster than t-SNE and often better preserves global structure:
tm_umap <- tm |>
activate(columns) |>
compute_umap(n_components = 2, name = "sample_umap",
n_neighbors = 10, random_state = 42)
# View results
head(tm_umap$col_data)
#> sample_id condition batch sample_umap_1 sample_umap_2
#> 1 Sample_1 Control B1 -11.23182 -1.2179000
#> 2 Sample_2 Control B2 -11.32073 -0.8089717
#> 3 Sample_3 Control B1 -10.43600 -1.8461883
#> 4 Sample_4 Control B2 -10.77704 -0.8251424
#> 5 Sample_5 Control B1 -10.98153 -1.9383749
#> 6 Sample_6 Control B2 -11.26206 -2.3579226UMAP Parameters
-
n_neighbors: Controls local vs global structure (larger = more global) -
min_dist: Minimum distance between points (smaller = tighter clusters)
tm_umap_custom <- tm |>
activate(columns) |>
compute_umap(n_components = 2, n_neighbors = 5, min_dist = 0.05,
random_state = 42, name = "umap_tight")
cat("Custom UMAP parameters create tighter clustering\n")
#> Custom UMAP parameters create tighter clusteringComparing Methods
All three methods can be applied to the same data:
tm_all <- tm |>
activate(columns) |>
compute_mds(k = 2, name = "mds")
# Add t-SNE and UMAP if packages available
if (requireNamespace("Rtsne", quietly = TRUE)) {
set.seed(42)
tm_all <- tm_all |>
compute_tsne(dims = 2, name = "tsne", perplexity = 5,
verbose = FALSE, check_duplicates = FALSE)
}
if (requireNamespace("umap", quietly = TRUE)) {
tm_all <- tm_all |>
compute_umap(n_components = 2, name = "umap",
n_neighbors = 10, random_state = 42)
}
# View all coordinates
head(tm_all$col_data)
#> sample_id condition batch mds_1 mds_2 tsne_1 tsne_2 umap_1
#> 1 Sample_1 Control B1 -8.617083 0.002052457 70.78583 13.41886 -11.23182
#> 2 Sample_2 Control B2 -7.004946 3.105104315 72.04628 17.51309 -11.32073
#> 3 Sample_3 Control B1 -7.633199 -0.746203884 68.59004 22.29858 -10.43600
#> 4 Sample_4 Control B2 -8.176460 -2.586303525 68.59600 16.28628 -10.77704
#> 5 Sample_5 Control B1 -6.682964 -4.795454877 67.10259 14.63846 -10.98153
#> 6 Sample_6 Control B2 -7.932464 -6.706727458 64.11826 16.00315 -11.26206
#> umap_2
#> 1 -1.2179000
#> 2 -0.8089717
#> 3 -1.8461883
#> 4 -0.8251424
#> 5 -1.9383749
#> 6 -2.3579226Gene-Level Dimensionality Reduction
Dimensionality reduction works on rows (genes) too:
tm_genes <- tm |>
activate(rows) |>
compute_mds(k = 2, name = "gene_mds")
# Filter genes by their embedding
interesting_genes <- tm_genes |>
activate(rows) |>
filter(gene_mds_1 > median(gene_mds_1))
#> Warning: Removed 1 stored analysis object(s) due to filter: gene_mds
#> Metadata columns are preserved.
cat("Genes in right half of MDS space:", nrow(interesting_genes$matrix), "\n")
#> Genes in right half of MDS space: 50Multiple Dimensions
Request more than 2 dimensions for comprehensive analysis:
tm_3d <- tm |>
activate(columns) |>
compute_mds(k = 3, name = "mds3d")
# Check all three dimensions
cat("MDS columns:", grep("mds3d", names(tm_3d$col_data), value = TRUE), "\n")
#> MDS columns: mds3d_1 mds3d_2 mds3d_3Using with to_long()
Combine dimensionality reduction with to_long() for visualization:
# After computing dimensionality reduction
tm |>
activate(columns) |>
compute_mds(k = 2) |>
to_long() |>
ggplot(aes(x = column_mds_1, y = column_mds_2, color = condition)) +
geom_point()Summary
tidymatrix provides three dimensionality reduction methods:
| Method | Speed | Deterministic | Best For |
|---|---|---|---|
| MDS | Fast | Yes | Quick exploration, no dependencies |
| t-SNE | Slow | No (use seed) | Local structure, visualization |
| UMAP | Medium | No (use seed) | Balanced local/global, faster than t-SNE |
When to use each: - MDS: Quick exploration, always available - t-SNE: Publication-quality visualizations, small to medium datasets - UMAP: Large datasets, need for global structure
All methods integrate seamlessly with tidymatrix operations and can be applied to both rows and columns.