Discovery-Driven Plasma Proteomics Identifies a Multi-Protein Signature for Amyloid PET Positivity: A Machine Learning Analysis of the Bio-Hermes Cohort.
The 1 match
- [1] § 4. Materials and Methods › 4.5. Machine-Learning Pipeline ↔ Workflow-RScript/SMLPipelineR.R, lines 21–35 · score 0.58 · Confusion matrices, balanced accuracy, sensitivity, metrics
Paper
Loaded from Europe PMC by your browser, not stored by OSCR: doi.org · Europe PMC
The paper is loaded when this pane is shown.
The authors' code
R · 316 lines · 12 KB · MIT · 1 match
- #!/usr/bin/env Rscript
- # =====================================================================
- # Omics ML Pipeline (General, Tabular Data)
- # Author: Your Name
- # Version: 0.1.0
- # Dependencies: caret, randomForest, xgboost, nnet, tidyverse, pROC, VennDiagram
- # =====================================================================
- suppressPackageStartupMessages({
- library(tidyverse)
- library(caret)
- library(randomForest)
- library(xgboost)
- library(nnet)
- library(pROC)
- library(VennDiagram)
- })
- # ----------- Utility --------------------------------------------------
- metric_from_cm <- function(cm) {
- # caret::confusionMatrix stores metrics in $byClass and $overall
- byc <- as.list(cm$byClass)
- ov <- as.list(cm$overall)
- tibble::tibble(
- Accuracy = as.numeric(ov[["Accuracy"]]),
- Kappa = as.numeric(ov[["Kappa"]]),
- Sensitivity = as.numeric(byc[["Sensitivity"]]),
- Specificity = as.numeric(byc[["Specificity"]]),
- `Balanced Accuracy` = as.numeric(byc[["Balanced Accuracy"]]),
- `Pos Pred Value` = as.numeric(byc[["Pos Pred Value"]]),
- `Neg Pred Value` = as.numeric(byc[["Neg Pred Value"]]),
- `Mcnemar P` = suppressWarnings(as.numeric(ov[["Mcnemar's Test P-Value"]]))
- )
- }
- ensure_dir <- function(path) {
- if (!dir.exists(path)) dir.create(path, recursive = TRUE, showWarnings = FALSE)
- }
- # ----------- Pipeline Functions --------------------------------------
- load_data <- function(path, label_col, id_col = NULL) {
- message("Loading data: ", path)
- ext <- tools::file_ext(path)
- df <- switch(tolower(ext),
- "csv" = readr::read_csv(path, show_col_types = FALSE),
- "tsv" = readr::read_tsv(path, show_col_types = FALSE),
- "txt" = readr::read_delim(path, delim = "\t", show_col_types = FALSE),
- stop("Unsupported file extension: ", ext))
- if (!label_col %in% names(df)) stop("label_col not found in data.")
- if (!is.null(id_col) && !id_col %in% names(df)) stop("id_col not found in data.")
- df <- df %>% mutate(!!label_col := as.factor(.data[[label_col]])) %>% droplevels()
- df
- }
- preprocess_data <- function(df, label_col, mode = c("complete_case", "median_impute"),
- log1p = FALSE, center_scale = TRUE) {
- mode <- match.arg(mode)
- y <- df[[label_col]]
- x <- df %>% select(-all_of(label_col))
- # Keep only numeric predictors for modeling
- numeric_mask <- purrr::map_lgl(x, is.numeric)
- x_num <- x[, numeric_mask, drop = FALSE]
- # Optional log1p
- if (log1p) x_num <- mutate_all(x_num, ~log1p(.x))
- # Missing handling
- if (mode == "complete_case") {
- keep <- stats::complete.cases(x_num) & !is.na(y)
- x_num <- x_num[keep, , drop = FALSE]
- y <- y[keep]
- } else {
- # Median impute via caret preProcess on predictors only
- pp <- caret::preProcess(x_num, method = "medianImpute")
- x_num <- predict(pp, x_num)
- }
- # Center/scale if requested
- if (center_scale) {
- pp2 <- caret::preProcess(x_num, method = c("center", "scale"))
- x_num <- predict(pp2, x_num)
- }
- out <- bind_cols(x_num, tibble::tibble(!!label_col := y)) %>% drop_na(all_of(label_col))
- list(data = out, predictors = colnames(x_num))
- }
- make_splits <- function(df, label_col, ratios = c(0.6, 0.7, 0.8, 0.9), seed = 123) {
- set.seed(seed)
- splits <- list()
- for (r in ratios) {
- idx <- caret::createDataPartition(df[[label_col]], p = r, list = FALSE)
- train_idx <- as.vector(idx)
- test_idx <- setdiff(seq_len(nrow(df)), train_idx)
- splits[[paste0(round(r*100), "_split")]] <- list(train = train_idx, test = test_idx)
- }
- splits
- }
- feature_screen <- function(df, label_col, alpha = 0.05, max_keep = NA) {
- # Two-group t-tests for numeric predictors
- y <- df[[label_col]]
- x <- df %>% select(-all_of(label_col))
- numeric_mask <- purrr::map_lgl(x, is.numeric)
- x <- x[, numeric_mask, drop = FALSE]
- res <- purrr::map_df(colnames(x), function(feat) {
- a <- x[[feat]][y == levels(y)[1]]
- b <- x[[feat]][y == levels(y)[2]]
- tt <- try(stats::t.test(a, b), silent = TRUE)
- if (inherits(tt, "try-error")) return(tibble::tibble(feature = feat, p = NA_real_, stat = NA_real_))
- tibble::tibble(feature = feat, p = tt$p.value, stat = unname(tt$statistic))
- }) %>% arrange(p)
- res$padj <- p.adjust(res$p, method = "BH")
- keep <- res %>% filter(padj < alpha)
- if (!is.na(max_keep)) keep <- keep %>% slice_head(n = max_keep)
- list(screen_table = res, keep_features = keep$feature)
- }
- train_one <- function(train_df, label_col, method = c("rf","xgbTree","nnet"), tuneLength = 10, seed = 123) {
- method <- match.arg(method)
- set.seed(seed)
- ctrl <- caret::trainControl(method = "repeatedcv", number = 5, repeats = 2,
- classProbs = TRUE, summaryFunction = twoClassSummary,
- savePredictions = "final")
- # Ensure positive class is the first level (caret uses first level as "event" for ROC)
- y <- train_df[[label_col]]
- if (length(levels(y)) != 2) stop("Outcome must be binary factor.")
- # Relevel so that the first level is the 'positive' class (customizable here)
- # By default, make the first level the one with lower frequency to emphasize recall;
- # adjust as needed for your use-case.
- levs <- levels(y)
- counts <- table(y)
- pos <- names(sort(counts))[1]
- train_df[[label_col]] <- relevel(train_df[[label_col]], ref = pos)
- fit <- caret::train(
- reformulate(termlabels = setdiff(names(train_df), label_col), response = label_col),
- data = train_df,
- method = method,
- metric = "ROC",
- trControl = ctrl,
- tuneLength = tuneLength
- )
- fit
- }
- evaluate_one <- function(fit, test_df, label_col) {
- # Predictions
- probs <- predict(fit, newdata = test_df, type = "prob")
- pred <- predict(fit, newdata = test_df, type = "raw")
- # Ensure same positive level as training
- positive_class <- fit$levels[1]
- cm <- caret::confusionMatrix(pred, test_df[[label_col]], positive = positive_class)
- metrics <- metric_from_cm(cm) %>% mutate(Model = fit$method, Positive = positive_class)
- list(confusion = cm, metrics = metrics, probs = probs, pred = pred)
- }
- var_importance <- function(fit, top_n = 20) {
- imp <- try(caret::varImp(fit, scale = TRUE), silent = TRUE)
- if (inherits(imp, "try-error")) return(tibble::tibble(Feature = character(), Importance = numeric()))
- vi <- imp$importance %>% tibble::rownames_to_column("Feature") %>% arrange(desc(Overall))
- if (!is.null(top_n)) vi <- vi %>% slice_head(n = top_n)
- vi
- }
- consensus_signature <- function(imp_list, top_n = 10) {
- top_sets <- lapply(imp_list, function(df) head(df$Feature, top_n))
- names(top_sets) <- names(imp_list)
- inter_all <- Reduce(intersect, top_sets)
- # Also return pairwise intersections
- list(top_sets = top_sets, intersect_all = inter_all)
- }
- # ----------- Orchestrator --------------------------------------------
- run_pipeline <- function(data_path,
- label_col,
- id_col = NULL,
- out_dir = "outputs",
- screen_alpha = 0.05,
- screen_top = NA,
- preprocess_mode = c("complete_case", "median_impute"),
- log1p = FALSE,
- center_scale = TRUE,
- split_ratios = c(0.6, 0.7, 0.8, 0.9),
- models = c("rf","xgbTree","nnet"),
- tuneLength = 10,
- seed = 123) {
- ensure_dir(out_dir)
- df0 <- load_data(data_path, label_col = label_col, id_col = id_col)
- # Preprocess
- pp <- preprocess_data(df0, label_col = label_col, mode = preprocess_mode,
- log1p = log1p, center_scale = center_scale)
- df <- pp$data
- predictors <- pp$predictors
- # Feature screening
- fs <- feature_screen(df, label_col = label_col, alpha = screen_alpha, max_keep = screen_top)
- keep_feats <- if (length(fs$keep_features) > 0) fs$keep_features else predictors
- readr::write_csv(fs$screen_table, file.path(out_dir, "feature_screening.csv"))
- # Use screened features
- df_use <- df %>% select(all_of(c(keep_feats, label_col)))
- # Splits
- splits <- make_splits(df_use, label_col = label_col, ratios = split_ratios, seed = seed)
- # Storage
- all_metrics <- list()
- all_importance <- list()
- # Loop over splits and models
- for (sp in names(splits)) {
- tr_idx <- splits[[sp]]$train
- te_idx <- splits[[sp]]$test
- train_df <- df_use[tr_idx, , drop = FALSE]
- test_df <- df_use[te_idx, , drop = FALSE]
- for (m in models) {
- fit <- train_one(train_df, label_col = label_col, method = m, tuneLength = tuneLength, seed = seed)
- ev <- evaluate_one(fit, test_df, label_col = label_col)
- vi <- var_importance(fit, top_n = 50)
- # Save artifacts
- model_tag <- paste(sp, m, sep = "_")
- saveRDS(fit, file = file.path(out_dir, paste0("model_", model_tag, ".rds")))
- readr::write_csv(ev$metrics %>% mutate(Split = sp), file.path(out_dir, paste0("metrics_", model_tag, ".csv")))
- readr::write_csv(vi, file.path(out_dir, paste0("varimp_", model_tag, ".csv")))
- all_metrics[[model_tag]] <- ev$metrics %>% mutate(Split = sp, Model = m)
- all_importance[[model_tag]] <- vi
- }
- }
- # Aggregate metrics
- metrics_df <- dplyr::bind_rows(all_metrics)
- readr::write_csv(metrics_df, file.path(out_dir, "metrics_all_models.csv"))
- # Consensus signature (per model family across best split or all splits)
- # Here we compute per family using all splits:
- families <- unique(metrics_df$Model)
- consensus <- list()
- for (fam in families) {
- fam_ims <- all_importance[grepl(paste0("_", fam, "$"), names(all_importance))]
- consensus[[fam]] <- consensus_signature(fam_ims, top_n = 10)
- # Save simple text summary
- sink(file.path(out_dir, paste0("consensus_", fam, ".txt")))
- cat("Model family:", fam, "\n")
- cat("Top-10 sets per split:\n")
- print(consensus[[fam]]$top_sets)
- cat("\nIntersection across splits:\n")
- print(consensus[[fam]]$intersect_all)
- sink()
- }
- # Optional: simple Venn diagram for first three sets of a family (if available)
- for (fam in names(consensus)) {
- ts <- consensus[[fam]]$top_sets
- if (length(ts) >= 3) {
- first_three <- ts[1:3]
- venn.plot <- VennDiagram::venn.diagram(
- x = first_three,
- filename = NULL,
- fill = c("#FEE08B", "#D53E4F", "#3288BD"),
- alpha = 0.5,
- cat.cex = 0.8,
- cex = 0.8,
- main = paste("Top-10 Feature Overlap -", fam)
- )
- grDevices::png(filename = file.path(out_dir, paste0("venn_", fam, ".png")), width = 1400, height = 1000, res = 150)
- grid::grid.draw(venn.plot)
- grDevices::dev.off()
- }
- }
- message("Pipeline complete. Outputs written to: ", out_dir)
- invisible(list(metrics = metrics_df, consensus = consensus))
- }
- # ----------- CLI Entry Point -----------------------------------------
- if (sys.nframe() == 0) {
- # Example CLI usage:
- # Rscript pipeline.R --data data/omics.csv --label outcome --out outputs
- args <- commandArgs(trailingOnly = TRUE)
- arg_list <- list()
- if (length(args) > 0) {
- for (i in seq(1, length(args), by = 2)) {
- key <- gsub("^--", "", args[i])
- val <- args[i + 1]
- arg_list[[key]] <- val
- }
- }
- data_path <- arg_list[["data"]]
- label_col <- arg_list[["label"]]
- out_dir <- arg_list[["out"]]
- if (is.null(data_path) || is.null(label_col)) {
- stop("Usage: Rscript pipeline.R --data <path.csv> --label <label_col> [--out outputs]")
- }
- if (is.null(out_dir)) out_dir <- "outputs"
- # Run with defaults
- run_pipeline(
- data_path = data_path,
- label_col = label_col,
- out_dir = out_dir,
- screen_alpha = 0.05,
- preprocess_mode = "complete_case",
- log1p = FALSE,
- center_scale = TRUE,
- split_ratios = c(0.6, 0.7, 0.8, 0.9),
- models = c("rf","xgbTree","nnet"),
- tuneLength = 10,
- seed = 123
- )
- }
SMLPipelineR.R at commit 0f5b910, under MIT · at the source
Overview
- School of Cardiovascular and Metabolic Health, University of Glasgow, Glasgow G12 8TA, UK; (S.L.)
- School of Biology, University of St. Andrews, St. Andrews KY16 9AJ, UK
Abstract
Alzheimer’s disease is a progressive neurodegenerative disorder in which early detection remains limited by the cost and invasiveness of positron emission tomography and cerebrospinal fluid testing. We evaluated whether plasma proteomic profiles could distinguish amyloid PET-positive from amyloid PET-negative individuals using the Bio-Hermes cohort. After quality control and missing-data filtering, 988 participants and 295 proteins were analysed; 31 proteins showing group differences were used for supervised classification. Random Forest, Gradient Boosting, and Neural Network models were trained across four train/
Reproduced under the paper's license (CC BY), from the paper cited above.
Repository
Its files are read in the Code ↔ Paper reader above, with 1 match between paragraphs and lines of code.
stelioslamprou37/SML_PipelineR
0f5b9109fb54d17fc2e932361dd9ba1d788863cd, 2 September 2025Availability: 1 check, the latest on 27 September 2026: the link answers
- 27 September 2026: the link answers
3 files
- Workflow-RScript/
SMLPipelineR.R , R, 316 lines, 1 match - LICENSE, License, 21 lines
- README.md, Text, 123 lines
The paper's code and data availability statement is in the Data section.
Tracing map
Proposed by the machine: these links were found in the paper and verified at the source, without human review. The map will receive a Zenodo DOI once one of the paper's authors has validated it with their ORCID.
What the map holds:
- 1 repository of the authors' code, each at its verified commit, with its license and how the link was found in the paper;
- 1 script, each with its path and the digest of its content;
- 1 match between paragraphs of the paper and lines of the code (method lexical-v1);
- neither the text of the paper nor the code itself.
Its JSON (tracing-map.json) is deposited on Zenodo with its DOI once the map is validated.
Data
No dataset and no data link were found in the paper.
Data Availability Statement
The Bio-Hermes plasma proteomic dataset and associated phenotypic data are available through the AD Workbench platform of the Alzheimer’s Disease Data Initiative (https://
Reproduced under the paper's license (CC BY), from the paper cited above.
Versions
The history of this record: each version stored by the harvester or made by a correction of its authors or of the maintainers of its code, and what changed in its facts. The texts of the paper (its abstract, its availability statements) are not part of it; versions that changed only those are not listed.
Version 1, 27 September 2026: the first record
Recorded: type, language, journal, volume, issue, pages, dates, 4 authors, 6 keywords, 12 MeSH terms, 2 funders, 60 references.
Cite
This paper
Lamprou, S., Mavromati, K., Gunn-Moore, F. J., & Quinn, T. J. (2026). Discovery-Driven Plasma Proteomics Identifies a Multi-Protein Signature for Amyloid PET Positivity: A Machine Learning Analysis of the Bio-Hermes Cohort. International journal of molecular sciences, 27(12), 5533. https://
BibTeX
@article{lamprou2026disc
author = {Lamprou, Stelios and Mavromati, Kalliopi and Gunn-Moore, Frank J. and Quinn, Terry J.},
title = {{Discovery-Driven Plasma Proteomics Identifies a Multi-Protein Signature for Amyloid PET Positivity: A Machine Learning Analysis of the Bio-Hermes Cohort}},
journal = {International journal of molecular sciences},
year = {2026},
month = jun,
volume = {27},
number = {12},
pages = {5533},
publisher = {Multidisciplinary Digital Publishing Institute (MDPI)},
issn = {1422-0067},
doi = {10.3390/
url = {https://
pmid = {42353247},
pmcid = {PMC13299064}
}
RIS
TY - JOUR
AU - Lamprou, Stelios
AU - Mavromati, Kalliopi
AU - Gunn-Moore, Frank J.
AU - Quinn, Terry J.
TI - Discovery-Driven Plasma Proteomics Identifies a Multi-Protein Signature for Amyloid PET Positivity: A Machine Learning Analysis of the Bio-Hermes Cohort
T2 - International journal of molecular sciences
J2 - Int J Mol Sci
PY - 2026
DA - 2026/
VL - 27
IS - 12
SP - 5533
SN - 1422-0067
PB - Multidisciplinary Digital Publishing Institute (MDPI)
DO - 10.3390/
UR - https://
LA - en
ER -
CSL-JSON
{
"id": "10.3390/
"type": "article-journal",
"title": "Discovery-Driven Plasma Proteomics Identifies a Multi-Protein Signature for Amyloid PET Positivity: A Machine Learning Analysis of the Bio-Hermes Cohort",
"container-title": "International journal of molecular sciences",
"author": [
{
"family": "Lamprou",
"given": "Stelios"
},
{
"family": "Mavromati",
"given": "Kalliopi"
},
{
"family": "Gunn-Moore",
"given": "Frank J."
},
{
"family": "Quinn",
"given": "Terry J."
}
],
"container-title-short":
"volume": "27",
"issue": "12",
"page": "5533",
"DOI": "10.3390/
"PMID": "42353247",
"PMCID": "PMC13299064",
"ISSN": "1422-0067",
"publisher": "Multidisciplinary Digital Publishing Institute (MDPI)",
"URL": "https://
"language": "en",
"issued": {
"date-parts": [
[
2026,
6,
18
]
]
}
}
The tracing map gets a citation of its own once an author has validated it and it has a DOI.
Similar papers
The papers with a page that share the most with this one: the tools found in their code, their categories, datasets, cited references and authors, the rarest counting most.
- [1] doi:10.3390/ijms27156925 [code]
- XGBoost-SHAP Interpretable Modeling Identifies and Validates an Eight-Gene Biomarker for Hepatic Encephalopathy Risk Prediction in Cirrhosis.Journal: International journal of molecular sciencesIn common: randomForest, pROC, caret, 2 other tools, genetics / omics
- [2] doi:10.21037/jtd-2026-0997 [code]
- Machine learning models based on XGBoost algorithm to predict prognosis of lung cancer brain metastases.Journal: Journal of thoracic diseaseIn common: randomForest, pROC, caret, 2 other tools
- [3] doi:10.1038/s41467-026-77170-3 [code]
- DNA methylation profiling identifies long-range epigenetic silencing of clustered protocadherins as a key determinant of meningioma progression.Journal: Nature communicationsIn common: randomForest, pROC, caret, 1 other tool, genetics / omics, cellular / molecular
- [4] doi:10.7717/peerj.21426 [code]
- Integrated transcriptomic identification and validation reveal key autophagy-associated biomarkers in sleep deprivation.Journal: PeerJIn common: randomForest, pROC, caret, 1 other tool, genetics / omics, cellular / molecular
- [5] doi:10.1177/13872877261471049 [code]
- Predicting future brain atrophy based on longitudinal MRI.Journal: Journal of Alzheimer's disease : JADIn common: randomForest, pROC, caret, 1 other tool, Alzheimer's / dementia
- [6] doi:10.1093/braincomms/fcag236 [code]
- Dynamic, state-dependent characteristics of cognitive fluctuations in Lewy body dementia: a magnetoencephalography study.Journal: Brain communicationsIn common: randomForest, pROC, caret, 1 other tool, Alzheimer's / dementia
- [7] doi:10.1038/s42003-026-10957-8 [code]
- Brain defence by the extracellular matrix protein Cochlin.Journal: Communications biologyIn common: randomForest, caret, XGBoost, 1 other tool, cellular / molecular
- [8] doi:10.1016/j.celrep.2026.117852 [code]
- Graph theory identifies altered prefrontal microcircuit organization in Shank3 mice, a mouse Model of autism.Journal: Cell reportsIn common: randomForest, pROC, caret, 1 other tool
- [9] doi:10.1002/alz.71711 [code]
- Exploring longitudinal relationships among Alzheimer's disease biomarkers.Journal: Alzheimer's & dementia : the journal of the Alzheimer's AssociationIn common: tidyverse, PET / SPECT, Alzheimer's / dementia, 3 references
- [10] doi:10.1002/alz.71567 [code]
- Associations of dementia polyexposure scores to Alzheimer's disease endophenotypes in a diverse population.Journal: Alzheimer's & dementia : the journal of the Alzheimer's AssociationIn common: randomForest, pROC, tidyverse, PET / SPECT, Alzheimer's / dementia
Contribute
The authors of this paper can claim it, correct its record and validate its tracing map, and the maintainers of its code (its owner, or a public member of its organization) correct what it says of their repository; anyone signed in can ask for its removal. Every request goes to OSCR's own machine, which answers it; your account page follows them.
Sign in with ORCID to claim this paper as one of its authors, correct its record or validate its tracing map: when the paper's metadata lists your ORCID iD, you are recognized at once. Maintainers of its code: sign in with GitHub, then claim the repository on your account page.
Claim this paper
Correct its record
Say what each link of this record is, remove the ones that are not the paper's, add the ones that are missing. The correction becomes a new version of the record, in its Versions section.
Validate its tracing map
You validate the map as this page shows it: 1 repository of the authors' code, each at its verified commit and with its license, 1 script, and 1 match between paragraphs and code (see the Code and Map sections). It then receives a DOI on Zenodo, with you (your ORCID iD) and OSCR as its creators; the code itself is not deposited.
The map's fingerprint: sha256:1f593b94d57b4cbd…
Add the badge to its README
The badge links the code to this page. Copy one of these into the README of the paper's code: only you decide where it goes, and nothing is changed for you.
Markdown
[, paste the snippet at the top, then “Commit changes…” and, to review it first, “Create a new branch and start a pull request”. You open the pull request; OSCR asks for no permission.
Request its removal
To ask OSCR to remove this record, the copies of its authors' scripts or its tracing map, use the removal request page: signed in, you say who you are, what to remove and why, then review and confirm the request. Published rules decide every request (how).
Discussion, reproductions, activity
Discussion: questions and error reports about this paper and its code, from signed-in readers and its authors. It opens with sign-in.
Reproductions: reports from readers who ran the authors' code: what they reproduced, with which environment, commit and data. It opens with sign-in.
Activity: what happens around this paper: new versions of its record, its map's validation, discussions and reproductions. It opens with sign-in.
