PEARL: integrative multi-omics classification and omics feature discovery via deep graph learning.
The 3 matches
- [1] § 2 Materials and methods › 2.4 Feature integration ↔ models.py, lines 189–225 · score 0.79 · binary classification, directly concatenates, multi class, refined features, activation, MLP
- [2] § 2 Materials and methods › 2.1 Overview of PEARL ↔ models.py, lines 189–225 · score 0.70 · refined features, unifies, max, min, component, concatenation
- [3] § 2 Materials and methods › 2.3 Feature refinement ↔ models.py, lines 50–117 · score 0.52 · ReLU, activation function, dropout, layer
Paper
Loaded from Europe PMC by your browser, not stored by OSCR: doi.org · Europe PMC
The paper is loaded when this pane is shown.
The authors' code
Python · 235 lines · 9 KB · no license · 3 matches
- import torch
- import torch.nn as nn
- import torch.nn.functional as F
- from torch_geometric.nn import SSGConv
- device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
- def xavier_init(m):
- if type(m) == nn.Linear:
- nn.init.xavier_normal_(m.weight)
- if m.bias is not None:
- m.bias.data.fill_(0.0)
- class SSGGraphConvolution(nn.Module):
- def __init__(self, in_features, out_features, K, alpha):
- super(SSGGraphConvolution, self).__init__()
- self.conv = SSGConv(in_features, out_features, K=K, alpha=alpha)
- def forward(self, x, adj):
- if adj.is_sparse:
- edge_index, edge_weight = adj._indices(), adj._values()
- else:
- edge_index = adj.nonzero().t().contiguous()
- edge_weight = adj[edge_index[0], edge_index[1]]
- return self.conv(x, edge_index, edge_weight)
- class GCN_E(nn.Module):
- def __init__(self, in_dim, hidden_dim, out_dim, K, alpha):
- super(GCN_E, self).__init__()
- self.gc1 = SSGGraphConvolution(in_dim, hidden_dim, K=K, alpha=alpha)
- self.gc2 = SSGGraphConvolution(hidden_dim, out_dim, K=K, alpha=alpha)
- self.dropout = nn.Dropout(0.5)
- def forward(self, x, adj):
- x = F.relu(self.gc1(x, adj))
- x = self.dropout(x)
- x = self.gc2(x, adj)
- return x
- class Classifier_1(nn.Module):
- def __init__(self, in_dim, out_dim):
- super().__init__()
- self.clf = nn.Sequential(nn.Linear(in_dim, out_dim))
- self.clf.apply(xavier_init)
- def forward(self, x):
- x = self.clf(x)
- return x
- class CombinedPoolingMLP(nn.Module):
- def __init__(self, num_view, num_cls, hidden_dims=[64], activation='relu',
- normalization='batchnorm', dropout_rates=[0.7], residual=True,
- input_normalization=False, output_activation=None):
- super().__init__()
- self.num_cls = num_cls
- self.num_view = num_view
- self.input_normalization = input_normalization
- self.output_activation = output_activation
- if input_normalization:
- self.input_norm = nn.BatchNorm1d(num_cls * 4)
- input_dim = num_cls * 4
- layers = []
- in_dim = input_dim
- for i, hidden_dim in enumerate(hidden_dims):
- layers.append(nn.Linear(in_dim, hidden_dim))
- layers.append(self.get_activation(activation))
- if normalization == 'batchnorm':
- layers.append(nn.BatchNorm1d(hidden_dim))
- elif normalization == 'layernorm':
- layers.append(nn.LayerNorm(hidden_dim))
- dropout_rate = dropout_rates[i] if i < len(dropout_rates) else dropout_rates[-1]
- layers.append(nn.Dropout(dropout_rate))
- if residual and in_dim == hidden_dim:
- layers.append(ResidualConnection())
- in_dim = hidden_dim
- layers.append(nn.Linear(in_dim, num_cls))
- if self.output_activation:
- layers.append(self.get_activation(self.output_activation))
- self.mlp = nn.Sequential(*layers)
- def get_activation(self, activation):
- if activation == 'relu':
- return nn.ReLU()
- elif activation == 'leaky_relu':
- return nn.LeakyReLU()
- elif activation == 'elu':
- return nn.ELU()
- elif activation == 'gelu':
- return nn.GELU()
- elif activation == 'tanh':
- return nn.Tanh()
- elif activation == 'sigmoid':
- return nn.Sigmoid()
- else:
- raise ValueError(f"Unsupported activation function: {activation}")
- def forward(self, in_list):
- stacked = torch.stack(in_list)
- max_pooled = torch.max(stacked, dim=0)[0]
- min_pooled = torch.min(stacked, dim=0)[0]
- avg_pooled = torch.mean(stacked, dim=0)
- x = torch.cat([max_pooled, min_pooled, avg_pooled, max_pooled - min_pooled], dim=1)
- if self.input_normalization:
- x = self.input_norm(x)
- return self.mlp(x)
- class EnhancedConcatMLPIntegration(nn.Module):
- def __init__(self, num_view, num_cls, hidden_dims=[64], activation='relu',
- normalization='layernorm', dropout_rates=[0.7], residual=True,
- input_normalization=False, output_activation=None):
- super().__init__()
- self.num_cls = num_cls
- self.num_view = num_view
- self.input_dim = num_view * num_cls
- self.input_normalization = input_normalization
- self.output_activation = output_activation
- if input_normalization:
- self.input_norm = nn.BatchNorm1d(self.input_dim)
- layers = []
- in_dim = self.input_dim
- for i, hidden_dim in enumerate(hidden_dims):
- layers.append(nn.Linear(in_dim, hidden_dim))
- layers.append(self.get_activation(activation))
- if normalization == 'batchnorm':
- layers.append(nn.BatchNorm1d(hidden_dim))
- elif normalization == 'layernorm':
- layers.append(nn.LayerNorm(hidden_dim))
- dropout_rate = dropout_rates[i] if i < len(dropout_rates) else dropout_rates[-1]
- layers.append(nn.Dropout(dropout_rate))
- if residual and in_dim == hidden_dim:
- layers.append(ResidualConnection())
- in_dim = hidden_dim
- layers.append(nn.Linear(in_dim, num_cls))
- if self.output_activation:
- layers.append(self.get_activation(self.output_activation))
- self.mlp = nn.Sequential(*layers)
- def get_activation(self, activation):
- if activation == 'relu':
- return nn.ReLU()
- elif activation == 'leaky_relu':
- return nn.LeakyReLU()
- elif activation == 'elu':
- return nn.ELU()
- elif activation == 'gelu':
- return nn.GELU()
- elif activation == 'tanh':
- return nn.Tanh()
- elif activation == 'sigmoid':
- return nn.Sigmoid()
- else:
- raise ValueError(f"Unsupported activation function: {activation}")
- def forward(self, in_list):
- x = torch.cat(in_list, dim=1)
- if self.input_normalization:
- x = self.input_norm(x)
- return self.mlp(x)
- class ResidualConnection(nn.Module):
- def forward(self, x):
- return x + self.branch(x)
- def branch(self, x):
- return x
- def init_model_dict(num_view, num_class, dim_list, dim_he_list, dim_hc,
- aggregation='combined_pooling',
- K=1, alpha=0.7, hidden_dims=[64], activation='relu',
- normalization='layernorm', dropout_rates=[0.7], residual=True,
- input_normalization=False, output_activation=None):
- """Initialize PEARL model components.
- Args:
- aggregation (str): Feature integration strategy for multi-view fusion.
- - 'combined_pooling': CombinedPoolingMLP — unifies max, min, mean,
- and difference pooling across views (recommended for multi-class).
- - 'concatenation': EnhancedConcatMLPIntegration — directly concatenates
- refined features from all views (recommended for binary classification).
- """
- model_dict = {}
- for i in range(num_view):
- model_dict["E{:}".format(i+1)] = GCN_E(dim_list[i], dim_he_list[0], dim_he_list[-1], K=K, alpha=alpha).to(device)
- model_dict["C{:}".format(i+1)] = Classifier_1(dim_he_list[-1], num_class).to(device)
- if num_view >= 2:
- if aggregation == 'combined_pooling':
- model_dict["C"] = CombinedPoolingMLP(
- num_view, num_class, hidden_dims=hidden_dims, activation=activation,
- normalization=normalization, dropout_rates=dropout_rates, residual=residual,
- input_normalization=input_normalization, output_activation=output_activation
- ).to(device)
- elif aggregation == 'concatenation':
- model_dict["C"] = EnhancedConcatMLPIntegration(
- num_view, num_class, hidden_dims=hidden_dims, activation=activation,
- normalization=normalization, dropout_rates=dropout_rates, residual=residual,
- input_normalization=input_normalization, output_activation=output_activation
- ).to(device)
- else:
- raise ValueError(
- f"Unsupported aggregation: '{aggregation}'. "
- f"Choose 'combined_pooling' or 'concatenation'."
- )
- return model_dict
- def init_optim(num_view, model_dict, lr_e=1e-4, lr_c=1e-4):
- optim_dict = {}
- for i in range(num_view):
- optim_dict["C{:}".format(i+1)] = torch.optim.Adam(
- list(model_dict["E{:}".format(i+1)].parameters()) + list(model_dict["C{:}".format(i+1)].parameters()),
- lr=lr_e)
- if num_view >= 2:
- optim_dict["C"] = torch.optim.Adam(model_dict["C"].parameters(), lr=lr_c)
- return optim_dict
models.py at commit 0316950, no license · at the source
Overview
- Carolina Health Informatics Program, University of North Carolina at Chapel Hill, Chapel Hill, NC 27599, United States
- Department of Biostatistics, University of North Carolina at Chapel Hill, Chapel Hill, NC 27599, United States
- Department of Genetics, University of North Carolina at Chapel Hill, Chapel Hill, NC 27599, United States
- Channing Division of Network Medicine, Department of Medicine, Brigham and Women’s Hospital, Harvard Medical School, Boston, MA 02115, United States
- Center for Computational and Genomic Medicine, Children’s Hospital of Philadelphia, Phildaelphia, PA 19104, United States
- Department of Pathology and Laboratory Medicine, University of Pennsylvania Perelman School of Medicine, Philadelphia, PA 19104, United States
- School of Data Science and Society, University of North Carolina at Chapel Hill, Chapel Hill, NC 27599, United States
- Department of Mathematics, University of North Carolina at Chapel Hill, Chapel Hill, NC 27599, United States
Abstract
Motivation: Integrating multi-omics data provides valuable insights into biological processes by capturing information across multiple molecular layers, enabling a comprehensive understanding of complex diseases and driving advancements in precision medicine. However, existing computational methods for multi-omics integration face significant challenges, such as low reliability and poor generalizability, due to the high dimensionality and low sample size nature of omics data.
Results: To address these challenges, we present PEARL (Pearson-Enhanced spectrAl gRaph convoLutional networks), a novel deep graph learning method for biomedical classification and functional important omics features identification. PEARL leverages a simple yet effective learning architecture to achieve superior and robust performance in high-dimensional, low-sample-size multi-omics settings. Our results demonstrate that PEARL significantly outperforms existing state-of-the-art methods on both synthetic and real biomedical datasets. Furthermore, applied to Alzheimer’s disease (AD) brain multi-omics data, features prioritized by PEARL lead to functionally important genes that demonstrate significant enrichment in AD-related pathways. These findings highlight PEARL’s practical utility in biomedical research and its potential to enhance biological interpretability in multi-omics studies.
Availability and implementation: The source code of our computational framework is available at https://
Reproduced under the paper's license (CC BY), from the paper cited above.
Repository
Its files are read in the Code ↔ Paper reader above, with 3 matches between paragraphs and lines of code.
zqq121017/PEARL
031695035891bfcf212bd78ab0a948404d5360e0, 31 March 2026Availability: 1 check, the latest on 27 September 2026: the link answers
- 27 September 2026: the link answers
6 files
- feat_importance.py, Python, 143 lines
- main_biomarker.py, Python, 33 lines
- models.py, Python, 235 lines, 3 matches
- train_test.py, Python, 197 lines
- utils.py, Python, 150 lines
- README.md, Text, 86 lines
Availability and implementation
The source code of our computational framework is available at https://
Reproduced under the paper's license (CC BY), from the paper cited above.
Tracing map
Proposed by the machine: these links were found in the paper and verified at the source, without human review. The map will receive a Zenodo DOI once one of the paper's authors has validated it with their ORCID.
What the map holds:
- 1 repository of the authors' code, each at its verified commit, with its license and how the link was found in the paper;
- 5 scripts, each with its path and the digest of its content;
- 3 matches between paragraphs of the paper and lines of the code (method lexical-v1);
- neither the text of the paper nor the code itself.
Its JSON (tracing-map.json) is deposited on Zenodo with its DOI once the map is validated.
Data
No dataset and no data link were found in the paper.
Data availability
The ROSMAP dataset was obtained from AMP-AD Knowledge Portal (https://
Reproduced under the paper's license (CC BY), from the paper cited above.
Versions
The history of this record: each version stored by the harvester or made by a correction of its authors or of the maintainers of its code, and what changed in its facts. The texts of the paper (its abstract, its availability statements) are not part of it; versions that changed only those are not listed.
Version 1, 27 September 2026: the first record
Recorded: type, language, journal, volume, issue, pages, dates, 6 authors, 9 MeSH terms, 59 references.
Cite
This paper
Zhao, Q., Du, J., Zhou, M., Wang, X.-W., Sun, Q., & Chen, C. (2026). PEARL: integrative multi-omics classification and omics feature discovery via deep graph learning. Bioinformatics (Oxford, England), 42(6), btag253. https://
BibTeX
@article{zhao2026pearl,
author = {Zhao, Quan and Du, Jiawen and Zhou, Muqing and Wang, Xu-Wen and Sun, Quan and Chen, Can},
title = {{PEARL: integrative multi-omics classification and omics feature discovery via deep graph learning}},
journal = {Bioinformatics (Oxford, England)},
year = {2026},
month = jun,
volume = {42},
number = {6},
pages = {btag253},
publisher = {Oxford University Press},
issn = {1367-4803},
doi = {10.1093/
url = {https://
pmid = {42108553},
pmcid = {PMC13224962}
}
RIS
TY - JOUR
AU - Zhao, Quan
AU - Du, Jiawen
AU - Zhou, Muqing
AU - Wang, Xu-Wen
AU - Sun, Quan
AU - Chen, Can
TI - PEARL: integrative multi-omics classification and omics feature discovery via deep graph learning
T2 - Bioinformatics (Oxford, England)
J2 - Bioinformatics
PY - 2026
DA - 2026/
VL - 42
IS - 6
SP - btag253
SN - 1367-4803
PB - Oxford University Press
DO - 10.1093/
UR - https://
LA - en
ER -
CSL-JSON
{
"id": "10.1093/
"type": "article-journal",
"title": "PEARL: integrative multi-omics classification and omics feature discovery via deep graph learning",
"container-title": "Bioinformatics (Oxford, England)",
"author": [
{
"family": "Zhao",
"given": "Quan"
},
{
"family": "Du",
"given": "Jiawen"
},
{
"family": "Zhou",
"given": "Muqing"
},
{
"family": "Wang",
"given": "Xu-Wen"
},
{
"family": "Sun",
"given": "Quan"
},
{
"family": "Chen",
"given": "Can"
}
],
"container-title-short":
"volume": "42",
"issue": "6",
"page": "btag253",
"DOI": "10.1093/
"PMID": "42108553",
"PMCID": "PMC13224962",
"ISSN": "1367-4803",
"publisher": "Oxford University Press",
"URL": "https://
"language": "en",
"issued": {
"date-parts": [
[
2026,
6,
1
]
]
}
}
The tracing map gets a citation of its own once an author has validated it and it has a DOI.
Similar papers
The papers with a page that share the most with this one: the tools found in their code, their categories, datasets, cited references and authors, the rarest counting most.
- [1] doi:10.3389/fsysb.2026.1873899 [code]
- A systems microbiology framework for reproducible multi-dataset omics integration with application to long COVID.Journal: Frontiers in systems biologyIn common: PyTorch Geometric, PyTorch, scikit-learn, 3 other tools, methods / tools, genetics / omics, 1 reference
- [2] doi:10.1038/s41467-026-74694-6 [code]
- Semi-supervised Omics Factor Analysis (SOFA) disentangles known and latent sources of variation in multi-omic data.Journal: Nature communicationsIn common: PyTorch, scikit-learn, pandas, 2 other tools, methods / tools, genetics / omics, 2 references
- [3] doi:10.1093/bioinformatics/btag540 [code]
- Deciphering spatial heterogeneity by multimodal spatial transcriptomics modelling with SpatialModal.Journal: Bioinformatics (Oxford, England)In common: PyTorch Geometric, PyTorch, scikit-learn, 3 other tools, methods / tools, Alzheimer's / dementia, genetics / omics
- [4] doi:10.1093/bib/bbag259 [code]
- PVAED: prior-guided variational autoencoders with diffusion denoising for interpretable single-cell representation learning.Journal: Briefings in bioinformaticsIn common: PyTorch, scikit-learn, pandas, 2 other tools, genetics / omics, 2 references
- [5] doi:10.1038/s41467-026-71391-2 [code]
- Accelerating Leigh syndrome drug discovery through deep learning screening in brain organoids.Journal: Nature communicationsIn common: PyTorch Geometric, PyTorch, scikit-learn, 3 other tools, 1 reference
- [6] doi:10.1038/s41592-026-03194-8 [code]
- Beyond benchmarking: an expert-guided consensus approach to spatially aware clustering.Journal: Nature methodsIn common: PyTorch Geometric, PyTorch, scikit-learn, 3 other tools, methods / tools, genetics / omics
- [7] doi:10.1093/bib/bbag298 [code]
- Empowering multifaceted analysis of spatial transcriptomics data with RGAST.Journal: Briefings in bioinformaticsIn common: PyTorch Geometric, PyTorch, scikit-learn, 3 other tools, methods / tools, genetics / omics
- [8] doi:10.1002/advs.75969 [code]
- Accurately Deciphering Tissue Heterogeneity From Spatial Multi-Modal and Multi-Omics With STransformer.Journal: Advanced science (Weinheim, Baden-Wurttemberg, Germany)In common: PyTorch Geometric, PyTorch, scikit-learn, 3 other tools, Alzheimer's / dementia, genetics / omics
- [9] doi:10.1371/journal.pcbi.1014327 [code]
- Supervised deep learning with gene functional annotation for cell classification.Journal: PLoS computational biologyIn common: PyTorch Geometric, PyTorch, scikit-learn, 3 other tools, Alzheimer's / dementia, genetics / omics
- [10] doi:10.1038/s41746-026-02735-x [code]
- Multimodal interpretable deep learning for transcriptome-informed precision oncology and drug mechanism analysis.Journal: NPJ digital medicineIn common: PyTorch, scikit-learn, pandas, 2 other tools, genetics / omics, 2 references
Contribute
The authors of this paper can claim it, correct its record and validate its tracing map, and the maintainers of its code (its owner, or a public member of its organization) correct what it says of their repository; anyone signed in can ask for its removal. Every request goes to OSCR's own machine, which answers it; your account page follows them.
Sign in with ORCID to claim this paper as one of its authors, correct its record or validate its tracing map: when the paper's metadata lists your ORCID iD, you are recognized at once. Maintainers of its code: sign in with GitHub, then claim the repository on your account page.
Claim this paper
Correct its record
Say what each link of this record is, remove the ones that are not the paper's, add the ones that are missing. The correction becomes a new version of the record, in its Versions section.
Validate its tracing map
You validate the map as this page shows it: 1 repository of the authors' code, each at its verified commit and with its license, 5 scripts, and 3 matches between paragraphs and code (see the Code and Map sections). It then receives a DOI on Zenodo, with you (your ORCID iD) and OSCR as its creators; the code itself is not deposited.
The map's fingerprint: sha256:1add68176fdc0c71…
Add the badge to its README
The badge links the code to this page. Copy one of these into the README of the paper's code: only you decide where it goes, and nothing is changed for you.
Markdown
[, paste the snippet at the top, then “Commit changes…” and, to review it first, “Create a new branch and start a pull request”. You open the pull request; OSCR asks for no permission.
Request its removal
To ask OSCR to remove this record, the copies of its authors' scripts or its tracing map, use the removal request page: signed in, you say who you are, what to remove and why, then review and confirm the request. Published rules decide every request (how).
Discussion, reproductions, activity
Discussion: questions and error reports about this paper and its code, from signed-in readers and its authors. It opens with sign-in.
Reproductions: reports from readers who ran the authors' code: what they reproduced, with which environment, commit and data. It opens with sign-in.
Activity: what happens around this paper: new versions of its record, its map's validation, discussions and reproductions. It opens with sign-in.
