| name | bio-rna-quantification-tximport-workflow |
| description | Import transcript-level quantifications from Salmon/kallisto into R for gene-level analysis with DESeq2/edgeR using tximport or tximeta. Use when importing transcript counts into R for DESeq2/edgeR. |
| tool_type | r |
| primary_tool | tximport |
tximport Workflow
Import transcript-level estimates from Salmon, kallisto, or other quantifiers into R for gene-level differential expression analysis.
Basic tximport
library(tximport)
files <- c(
sample1 = 'sample1_quant/quant.sf',
sample2 = 'sample2_quant/quant.sf',
sample3 = 'sample3_quant/quant.sf'
)
tx2gene <- read.csv('tx2gene.csv')
txi <- tximport(files, type = 'salmon', tx2gene = tx2gene)
Creating tx2gene Mapping
From GTF (using GenomicFeatures)
library(GenomicFeatures)
txdb <- makeTxDbFromGFF('annotation.gtf')
k <- keys(txdb, keytype = 'TXNAME')
tx2gene <- select(txdb, k, 'GENEID', 'TXNAME')
From Ensembl (using biomaRt)
library(biomaRt)
mart <- useMart('ensembl', dataset = 'hsapiens_gene_ensembl')
tx2gene <- getBM(
attributes = c('ensembl_transcript_id_version', 'ensembl_gene_id_version'),
mart = mart
)
colnames(tx2gene) <- c('TXNAME', 'GENEID')
From Salmon quant.sf
quant <- read.table('sample1_quant/quant.sf', header = TRUE)
tx2gene <- data.frame(
TXNAME = quant$Name,
GENEID = gsub('\\..*', '', quant$Name)
)
Import Types
Gene-Level Summarization (Default)
txi <- tximport(files, type = 'salmon', tx2gene = tx2gene)
Transcript-Level (No Summarization)
txi <- tximport(files, type = 'salmon', txOut = TRUE)
Scaled TPM (for visualization)
txi <- tximport(files, type = 'salmon', tx2gene = tx2gene,
countsFromAbundance = 'scaledTPM')
Source-Specific Import
Salmon
txi <- tximport(files, type = 'salmon', tx2gene = tx2gene)
kallisto
txi <- tximport(files, type = 'kallisto', tx2gene = tx2gene)
RSEM
txi <- tximport(files, type = 'rsem', tx2gene = tx2gene)
StringTie
txi <- tximport(files, type = 'stringtie', tx2gene = tx2gene)
Using with DESeq2
library(DESeq2)
coldata <- data.frame(
condition = factor(c('control', 'control', 'treated', 'treated')),
row.names = names(files)
)
dds <- DESeqDataSetFromTximport(txi, colData = coldata, design = ~ condition)
dds <- dds[rowSums(counts(dds)) >= 10, ]
dds <- DESeq(dds)
res <- results(dds)
Using with edgeR
library(edgeR)
cts <- txi$counts
normMat <- txi$length
normMat <- normMat / exp(rowMeans(log(normMat)))
o <- log(calcNormFactors(cts / normMat)) + log(colSums(cts / normMat))
y <- DGEList(cts)
y$offset <- t(t(log(normMat)) + o)
y <- estimateDisp(y, design)
tximeta: Metadata-Aware Import
tximeta automatically attaches transcript and gene information from the original annotation.
library(tximeta)
makeLinkedTxome(
indexDir = 'salmon_index',
source = 'Ensembl',
organism = 'Homo sapiens',
release = '110',
genome = 'GRCh38',
fasta = 'transcripts.fa',
gtf = 'annotation.gtf'
)
coldata <- data.frame(
files = files,
names = names(files),
condition = c('control', 'control', 'treated', 'treated')
)
se <- tximeta(coldata)
gse <- summarizeToGene(se
dds DESeqDataSetgse design condition
tximport Output Structure
names(txi)
Handling Version Numbers
tx2gene$TXNAME <- gsub('\\.\\d+$', '', tx2gene$TXNAME)
txi <- tximport(files, type = 'salmon', tx2gene = tx2gene,
ignoreTxVersion = TRUE, ignoreAfterBar = TRUE)
Related Skills
- rna-quantification/alignment-free-quant - Upstream Salmon/kallisto
- differential-expression/deseq2-basics - DESeq2 analysis
- differential-expression/edger-basics - edgeR analysis
- genome-intervals/gtf-gff-handling - GTF annotation parsing