| name | bio-metabolomics-metabolite-annotation |
| description | Metabolite identification from m/z and retention time. Covers database matching, MS/MS spectral matching, and confidence level assignment. Use when assigning compound identities to detected features in untargeted metabolomics. |
| tool_type | mixed |
| primary_tool | HMDB |
Metabolite Annotation
Database Matching by m/z
library(MetaboAnalystR)
features <- read.csv('feature_table.csv')
search_hmdb <- function(mz, adduct = '[M+H]+', ppm = 10) {
adduct_masses <- list(
'[M+H]+' = 1.007276,
'[M+Na]+' = 22.989218,
'[M-H]-' = -1.007276,
'[M+Cl]-' = 34.969402
)
neutral_mass <- mz - adduct_masses[[adduct]]
matches <- QueryHMDB(neutral_mass, ppm)
return(matches)
}
annotations <- lapply(features$mz, function(m) search_hmdb(m, '[M+H]+', 10))
MS/MS Spectral Matching
from matchms import calculate_scores
from matchms.importing import load_from_mgf
from matchms.similarity import CosineGreedy
queries = list(load_from_mgf('sample_msms.mgf'))
references = list(load_from_mgf('reference_library.mgf'))
similarity = CosineGreedy(tolerance=0.01)
scores = calculate_scores(references, queries, similarity)
for query_idx, query in enumerate(queries):
best_match_idx = scores.scores[:, query_idx].argmax()
best_score = scores.scores[best_match_idx, query_idx]
if best_score > 0.7:
ref = references[best_match_idx]
print(f'{query.get("precursor_mz")}: {ref.get("compound_name")} (score={best_score:.2f})')
SIRIUS + CSI:FingerID
sirius \
--input sample.ms \
--output sirius_results \
--database hmdb \
formula \
fingerid
MetFrag In Silico Fragmentation
library(metfRag)
settings <- list(
DatabaseSearchRelativeMassDeviation = 10,
FragmentPeakMatchAbsoluteMassDeviation = 0.01,
FragmentPeakMatchRelativeMassDeviation = 10,
MetFragDatabaseType = 'HMDB',
NeutralPrecursorMass = 147.0532
)
results <- run.metfrag(settings, spectrum_file = 'query_spectrum.txt')
RT Prediction for Validation
from deepchem.models import GraphConvModel
import pandas as pd
def validate_annotation(observed_rt, smiles, rt_model):
'''Check if observed RT matches prediction'''
predicted_rt = rt_model.predict(smiles)
rt_error = abs(observed_rt - predicted_rt)
if rt_error < 30:
return 'confident'
elif rt_error < 60:
return 'probable'
else:
return 'unlikely'
Confidence Levels (MSI)
assign_confidence <- function(annotation) {
if (!is.null(annotation$authentic_standard)) {
return(1)
} else if (!is.null(annotation$msms_match) && annotation$msms_score > 0.8) {
return(2)
} else if (!is.null(annotation$formula_match)) {
return(3)
annotationmass_match
featuresconfidence_level sapplyannotations assign_confidence
CAMERA Adduct Annotation
library(CAMERA)
xsa <- xsAnnotate(xcms_set)
xsa <- groupFWHM(xsa, perfwhm = 0.6)
xsa <- findIsotopes(xsa, mzabs = 0.01, ppm = 10)
xsa <- findAdducts(xsa, polarity = 'positive',
rules = c('[M+H]+', '[M+Na]+', '[M+K]+', '[M+NH4]+'))
annotated <- getPeaklist(xsa)
annotated$adduct
annotated$isotopes
annotated$pcgroup
Batch Annotation Pipeline
library(tidyverse)
annotate_features <- function(feature_table, ppm = 10, polarity = 'positive') {
results <- feature_table %>%
rowwise() %>%
mutate(
mass_h = ifelse(polarity == 'positive', mz - 1.007276, mz + 1.007276),
hmdb_match = list(query_hmdb(mass_h, ppm)),
kegg_match = list(query_kegg(mass_h, ppm)),
best_match = get_best_match(hmdb_match, kegg_match
compound_name best_matchname
compound_id best_matchid
mass_error_ppm mz best_matchmz mz
results
query_hmdb mass ppm
Export Annotated Results
annotation_report <- features %>%
select(feature_id, mz, rt, compound_name, compound_id,
formula, confidence_level, mass_error_ppm, adduct) %>%
arrange(confidence_level, desc(intensity))
write.csv(annotation_report, 'annotated_features.csv', row.names = FALSE)
cat('Annotation summary:\n')
cat(' Level 1 (confirmed):', sum(annotation_report$confidence_level == 1), '\n')
cat(' Level 2 (MS/MS match):', sum(annotation_report$confidence_level == 2), '\n')
cat annotation_reportconfidence_level
cat annotation_reportconfidence_level
cat annotation_reportconfidence_level
Related Skills
- xcms-preprocessing - Generate feature table
- pathway-mapping - Map annotated metabolites to pathways
- proteomics/spectral-libraries - Similar spectral matching concepts