SIMLINK enables accurate variant pathogenicity prediction through modeling the gene-variant-feature association structure.
The 3 matches
- [1] § 2 Materials and methods › 2.1 The SIMLINK method › 2.1.2 Model architecture ↔ layers.py, lines 235–268 · score 0.89 · TransH, QuatE, RotatE, TransE, DistMult, TransD
- [2] § 3 Results and discussion › 3.1 Comparison with state-of-the-art pathogenicity prediction methods ↔ utils.py, lines 241–289 · score 0.84 · PrDSM, SilVA, TraP, fathmm MKL, usDSM, DANN
- [3] § 3 Results and discussion › 3.1 Comparison with state-of-the-art pathogenicity prediction methods ↔ utils.py, lines 241–289 · score 0.51 · fathmm MKL, Eigen, DANN, MVP, CADD, error
Paper
Loaded from Europe PMC by your browser, not stored by OSCR: doi.org · Europe PMC
The paper is loaded when this pane is shown.
The authors' code
Python · 347 lines · 13 KB · no license · 2 matches
- import numpy as np
- import pickle as pkl
- import scipy.sparse as sp
- import sys
- import tensorflow as tf
- import math
- import os
- import random
- from collections import Counter
- import logging
- import pandas as pd
- import shutil
- from sklearn.impute import SimpleImputer
- from sklearn.linear_model import LinearRegression
- from sklearn.metrics import mean_squared_error, r2_score, accuracy_score, roc_auc_score
- import joblib
- import argparse
- flags = tf.app.flags
- FLAGS = flags.FLAGS
- def create_exp_dir(path, scripts_to_save=None):
- path_split = path.split("/")
- path_i = "."
- for one_path in path_split:
- path_i += "/" + one_path
- if not os.path.exists(path_i):
- os.mkdir(path_i)
- print('Experiment dir : {}'.format(path_i))
- if scripts_to_save is not None:
- os.mkdir(os.path.join(path, 'scripts'))
- for script in scripts_to_save:
- dst_file = os.path.join(path, 'scripts', os.path.basename(script))
- shutil.copyfile(script, dst_file)
- def inverse_sum(adj):
- adj = sp.coo_matrix(adj)
- rowsum = np.array(adj.sum(1))
- d_inv_sqrt = np.power(rowsum, -1).flatten()
- d_inv_sqrt[np.isinf(d_inv_sqrt)] = 0.
- return d_inv_sqrt.reshape((-1, 1))
- def preprocess_adj(adj):
- ent_adj_invsum = inverse_sum(adj[0])
- rel_adj_invsum = inverse_sum(adj[1])
- return [ent_adj_invsum, rel_adj_invsum, adj[2]]
- def construct_feed_dict(features, support, placeholders):
- feed_dict = dict()
- feed_dict.update({placeholders['features']: features})
- if isinstance(support[0], list):
- for i in range(len(support)):
- feed_dict.update({placeholders['support'][i][j]: support[i][j] \
- for j in range(len(support[i]))})
- else:
- feed_dict.update({placeholders['support'][i]: support[i] \
- for i in range(len(support))})
- return feed_dict
- def loadfile(file, num=1):
- '''
- num: number of elements per row
- '''
- print('loading file ' + file)
- ret = []
- with open(file, "r", encoding='utf-8') as rf:
- for line in rf:
- th = line[:-1].split('\t')
- x = []
- for i in range(num):
- x.append(int(th[i]))
- ret.append(tuple(x))
- return ret
- def get_ent2id(files):
- ent2id = {}
- for file in files:
- with open(file, 'r', encoding='utf-8') as rf:
- for line in rf:
- th = line[:-1].split('\t')
- ent2id[th[1]] = int(th[0])
- return ent2id
- def get_extended_adj_auto(e, KG):
- nei_list = []
- ent_row, rel_row = [], []
- ent_col, rel_col = [], []
- ent_data, rel_data = [], []
- count = 0
- for tri in KG:
- nei_list.append([tri[0], tri[1], tri[2]])
- ent_row.append(tri[0])
- ent_col.append(count)
- ent_data.append(1.)
- ent_row.append(tri[2])
- ent_col.append(count)
- ent_data.append(1.)
- rel_row.append(tri[1])
- rel_col.append(count)
- rel_data.append(1.)
- count += 1
- ent_adj_ind = sp.coo_matrix((ent_data, (ent_row, ent_col)), shape=(e, count))
- rel_adj_ind = sp.coo_matrix((rel_data, (rel_row, rel_col)), shape=(max(rel_row)+1, count))
- return [ent_adj_ind, rel_adj_ind, np.array(nei_list)]
- def load_data_class(FLAGS):
- def analysis(A, y, train, test):
- for A_i in A:
- print(A_i.nonzero())
- exit()
- def to_KG(A):
- KG = []
- count = 0
- for A_i in A:
- idx = A_i.nonzero()
- for head, tail in zip(idx[0], idx[1]):
- KG.append([head, count, tail])
- if len(idx[0]) > 0:
- count += 1
- # print(KG[:100])
- return KG
- dirname = os.path.dirname(os.path.realpath(sys.argv[0]))
- raw_file = dirname + '/kgdata/class/' + FLAGS.dataset + '.pickle'
- pro_file = dirname + '/kgdata/class/' + FLAGS.dataset + 'pro.pickle'
- if not os.path.exists(pro_file):
- with open(dirname + '/kgdata/class/' + FLAGS.dataset + '.pickle', 'rb') as f:
- data = pkl.load(f)
- A = data['A']
- KG = to_KG(A)
- num_ent = A[0].shape[0]
- data["A"] = KG
- data["e"] = num_ent
- # analysis(A, y, train, test)
- with open(dirname + '/kgdata/class/' + FLAGS.dataset + 'pro.pickle', 'wb') as handle:
- pkl.dump(data, handle, protocol=pkl.HIGHEST_PROTOCOL)
- with open(dirname + '/kgdata/class/' + FLAGS.dataset + 'pro.pickle', 'rb') as f:
- data = pkl.load(f)
- KG = data["A"]
- # y: csr_sparse_matrix
- y = sp.csr_matrix(data['y']).astype(np.float32)
- train = data['train_idx']
- test = data['test_idx']
- num_ent = data["e"]
- if FLAGS.dataset in ["train_clinvar_2022_all_test_clinvar_20230326", "train_clinvar_2022_all_test_usDSM"]:
- random.shuffle(train)
- temp_train = train[:int(0.9*len(train))]
- valid = train[int(0.9*len(train)):]
- train = temp_train
- test = test
- logging.info("train {}, valid {}, test {}".format(len(train), len(valid), len(test)))
- else:
- valid = None
- adj = get_extended_adj_auto(num_ent, KG)
- return adj, num_ent, train, test, valid, y
- def load_data_align(FLAGS):
- names = [['ent_ids_1', 'ent_ids_2'], ['triples_1', 'triples_2'], ['ref_ent_ids']]
- if FLAGS.rel_align:
- names[1][1] = "triples_2_relaligned"
- for fns in names:
- for i in range(len(fns)):
- fns[i] = 'data/'+FLAGS.dataset+'/'+fns[i]
- Ent_files, Tri_files, align_file = names
- num_ent = len(set(loadfile(Ent_files[0], 1)) | set(loadfile(Ent_files[1], 1)))
- align_labels = loadfile(align_file[0], 2)
- num_align_labels = len(align_labels)
- np.random.shuffle(align_labels)
- if not FLAGS.valid:
- train = np.array(align_labels[:num_align_labels // 10 * FLAGS.seed])
- valid = None
- else:
- train = np.array(align_labels[:int(num_align_labels // 10 * (FLAGS.seed-1))])
- valid = align_labels[int(num_align_labels // 10 * (FLAGS.seed-1)): num_align_labels // 10 * FLAGS.seed]
- test = align_labels[num_align_labels // 10 * FLAGS.seed:]
- KG = loadfile(Tri_files[0], 3) + loadfile(Tri_files[1], 3)
- ent2id = get_ent2id([Ent_files[0], Ent_files[1]])
- adj = get_extended_adj_auto(num_ent, KG)
- return adj, num_ent, train, test, valid
- def load_data_rel_align(FLAGS):
- names = [['ent_ids_1', 'ent_ids_2'], ['triples_1', 'triples_2'], ['ref_ent_ids']]
- for fns in names:
- for i in range(len(fns)):
- fns[i] = 'data/'+FLAGS.dataset+'/'+fns[i]
- Ent_files, Tri_files, align_file = names
- num_ent = len(set(loadfile(Ent_files[0], 1)) | set(loadfile(Ent_files[1], 1)))
- align_labels = loadfile(align_file[0], 2)
- num_align_labels = len(align_labels)
- np.random.shuffle(align_labels)
- if not FLAGS.valid:
- train = np.array(align_labels[:num_align_labels // 10 * FLAGS.seed])
- valid = None
- else:
- train = np.array(align_labels[:int(num_align_labels // 10 * (FLAGS.seed-1))])
- valid = align_labels[int(num_align_labels // 10 * (FLAGS.seed-1)): num_align_labels // 10 * FLAGS.seed]
- test = align_labels[num_align_labels // 10 * FLAGS.seed:]
- KG = loadfile(Tri_files[0], 3) + loadfile(Tri_files[1], 3)
- ent2id = get_ent2id([Ent_files[0], Ent_files[1]])
- adj = get_extended_adj_auto(num_ent, KG)
- rel_align_labels = loadfile('data/'+FLAGS.dataset+"/ref_rel_ids", 2)
- num_rel_align_labels = len(rel_align_labels)
- np.random.shuffle(rel_align_labels)
- if not FLAGS.valid:
- train_rel = np.array(rel_align_labels[:num_rel_align_labels // 10 * FLAGS.rel_seed])
- valid_rel = None
- else:
- train_rel = np.array(rel_align_labels[:int(num_rel_align_labels // 10 * (FLAGS.rel_seed-1))])
- valid_rel = rel_align_labels[int(num_rel_align_labels // 10 * (FLAGS.rel_seed-1)): num_rel_align_labels // 10 * FLAGS.rel_seed]
- test_rel = rel_align_labels[num_rel_align_labels // 10 * FLAGS.rel_seed:]
- return adj, num_ent, train, test, valid, train_rel, test_rel, valid_rel
- def get_batch(datax, datay, batch_size):
- input_queue = tf.train.slice_input_producer([datax, datay], num_epochs=None, shuffle=False, capacity=32 )
- x_batch, y_batch = tf.train.batch(input_queue, batch_size=batch_size, num_threads=1, capacity=32, allow_smaller_final_batch=False)
- return x_batch, y_batch
- def load_and_preprocess_data(train_file_path, test_file_path, mode):
- """
- 加载和预处理训练集和测试集数据。
- 【修改】: 此函数现在返回在训练集上训练好的imputer对象,
- 以便在后续的预测中保持数据处理的一致性。
- """
- # 读取训练集和测试集CSV文件
- train_df = pd.read_csv(train_file_path)
- test_df = pd.read_csv(test_file_path)
- if mode == "missense":
- feature_columns = [
- 'BayesDel_addAF_rankscore', 'BayesDel_noAF_rankscore', 'CADD_raw_rankscore',
- 'CADD_raw_rankscore_hg19', 'ClinPred_rankscore', 'DANN_rankscore',
- 'DEOGEN2_rankscore', 'Eigen_PC_raw_coding_rankscore', 'Eigen_raw_coding_rankscore',
- 'FATHMM_converted_rankscore', 'LIST_S2_rankscore', 'M_CAP_rankscore',
- 'MPC_rankscore', 'MVP_rankscore', 'MetaLR_rankscore', 'MetaRNN_rankscore',
- 'MetaSVM_rankscore', 'MutPred_rankscore', 'MutationAssessor_rankscore',
- 'MutationTaster_converted_rankscore', 'PROVEAN_converted_rankscore',
- 'Polyphen2_HDIV_rankscore', 'Polyphen2_HVAR_rankscore', 'PrimateAI_rankscore',
- 'REVEL_rankscore', 'SIFT4G_converted_rankscore', 'SIFT_converted_rankscore',
- 'VEST4_rankscore', 'fathmm_MKL_coding_rankscore', 'fathmm_XF_coding_rankscore',
- 'phastCons100way_vertebrate_rankscore', 'phyloP100way_vertebrate_rankscore']
- else:
- feature_columns = ['usDSM', 'CADD_raw_rankscore', 'TraP', 'SilVA', 'fathmm_MKL_coding_rankscore', 'PrDSM', 'DANN_rankscore']
- # 使用两个数据集中都存在的列
- common_columns = [col for col in feature_columns if col in train_df.columns and col in test_df.columns]
- print(f"使用的共同特征数量: {len(common_columns)}")
- # 提取特征和标签
- X_train_raw = train_df[common_columns].apply(pd.to_numeric, errors='coerce')
- X_test_raw = test_df[common_columns].apply(pd.to_numeric, errors='coerce')
- # 检查目标变量列
- if 'True Label' not in train_df.columns or 'True Label' not in test_df.columns:
- raise ValueError("训练集和测试集中都必须包含 'True Label' 列")
- # 【优化】使用numpy.where进行向量化操作,比循环更高效
- y_train = np.where(train_df['True Label'] < 0.5, -1, 1)
- y_test = np.where(test_df['True Label'] < 0.5, -1, 1)
- # 【核心修改】创建imputer,在训练集上fit,然后分别转换训练集和测试集
- imputer = SimpleImputer(strategy='mean')
- X_train = imputer.fit_transform(X_train_raw)
- X_test = imputer.transform(X_test_raw) # 注意:这里只用transform
- # 【核心修改】返回训练好的imputer
- return X_train, X_test, y_train, y_test, common_columns, imputer
- def train_linear_model(X_train, X_test, y_train, y_test):
- """
- 训练线性回归模型并在测试集上评估。
- 【无修改】此函数逻辑正确。
- """
- # 创建并训练模型
- model = LinearRegression()
- model.fit(X_train, y_train)
- # 在训练集和测试集上预测
- y_train_pred = model.predict(X_train)
- y_test_pred = model.predict(X_test)
- # 评估模型
- train_mse = mean_squared_error(y_train, y_train_pred)
- test_mse = mean_squared_error(y_test, y_test_pred)
- train_r2 = r2_score(y_train, y_train_pred)
- test_r2 = r2_score(y_test, y_test_pred)
- print(f"模型评估结果:")
- print(f"训练集均方误差 (MSE): {train_mse:.4f}")
- print(f"测试集均方误差 (MSE): {test_mse:.4f}")
- print(f"训练集 R2 分数: {train_r2:.4f}")
- print(f"测试集 R2 分数: {test_r2:.4f}")
- # 返回模型本身,真实的测试标签,和对测试集的预测
- return model, y_test, y_test_pred
- def predict_new_data(model, imputer, new_data_file, feature_columns):
- """
- 【修改版】
- 对无标签的新数据进行预测,并输出0到1之间的概率分数。
- """
- # 加载新数据
- new_df = pd.read_csv(new_data_file)
- # 检查新数据是否包含所有必需的特征列
- if not all(col in new_df.columns for col in feature_columns):
- missing = [col for col in feature_columns if col not in new_df.columns]
- raise ValueError(f"新数据文件中缺少以下必需的特征列: {missing}")
- # 提取特征数据并处理
- X_new_raw = new_df[feature_columns].apply(pd.to_numeric, errors='coerce')
- X_new = imputer.transform(X_new_raw)
- # 步骤1: 从线性模型获取原始预测分数(和之前一样)
- raw_scores = model.predict(X_new)
- # 步骤2: 【核心修改】应用Sigmoid函数将原始分数转换为0-1之间的概率
- # Sigmoid(x) = 1 / (1 + exp(-x))
- predicted_probabilities = 1 / (1 + np.exp(-raw_scores))
- # 步骤3: 【核心修改】只返回概率分数
- return predicted_probabilities
utils.py at commit 3e26596, no license · at the source
Overview
- School of Computer Science and Engineering, Central South University, Changsha, Hunan 410083, P.R. China
- Hunan Provincial Key Lab on Bioinformatics, Changsha, Hunan 410083, P.R. China
- Department of Mathematics, Hong Kong University of Science and Technology, Hong Kong, P.R. China
Abstract
Motivation: Predicting variant pathogenicity is crucial for clinical genetics. Existing approaches face two primary limitations. First, biologically, data for pathogenicity prediction often lacks explicit modeling of the gene-variant-feature association structure. A single gene can harbor multiple variants, and each variant can be characterized by multiple features intrinsically associated with its parent gene. Current methods fail to explicitly model the gene-variant-feature association, thus limiting their performance. Second, methodologically, the variant-pathogenicity association is often assumed to comprise a linear component alongside a nonlinear one. However, current methods typically do not explicitly model the linear component, often failing to disentangle the linear component that might be better addressed with a linear approach.
Results: To overcome these limitations, we introduce simultaneous modeling of linear and nonlinear components of knowledge graph (SIMLINK). This novel approach leverages a knowledge graph to model gene-variant-feature associations and a linear model to isolate the linear component. We begin by constructing a variant-centered knowledge graph, comprising over 8 million triplets, which explicitly models the associations between genes, variants, and features. Subsequently, the linear and nonlinear components are learned using a combination of linear and graph neural networks. We train SIMLINK on ClinVar variants. Benchmarking experiments on independent test sets demonstrate its superior prediction on both missense and synonymous variants compared to state-of-the-art methods, including CADD and AlphaMissense. We evaluate the impact of allele frequencies on prediction performance. Applied to variants implicated in Autism Spectrum Disorder, SIMLINK effectively distinguished between high- and low-confidence variants, and critically, the genes harboring top-ranked variants are highly pathogenic.
Availability and implementation: The source code is freely available at https://
Reproduced under the paper's license (CC BY), from the paper cited above.
Repositories
Its files are read in the Code ↔ Paper reader above, with 3 matches between paragraphs and lines of code.
Chen-LuWang/SIMLINK
3e2659637d8950362b344b7a0db2fe5e0ca74a4d, 5 August 2026Availability: 1 check, the latest on 26 September 2026: the link answers
- 26 September 2026: the link answers
8 files
- inits.py, Python, 66 lines
- layers.py, Python, 343 lines, 1 match
- metrics.py, Python, 166 lines
- models.py, Python, 298 lines
- run_class.sh, Shell, 54 lines
- train_class.py, Python, 204 lines
- utils.py, Python, 347 lines, 2 matches
- README.md, Text, 27 lines
Zenodo 21802211
Availability: 1 check, the latest on 26 September 2026: the link answers (HTTP 200)
- 26 September 2026: the link answers (HTTP 200)
8 files
- inits.py, Python, 66 lines
- layers.py, Python, 343 lines
- metrics.py, Python, 166 lines
- models.py, Python, 298 lines
- run_class.sh, Shell, 54 lines
- train_class.py, Python, 204 lines
- utils.py, Python, 347 lines
- README.md, Text, 27 lines
Availability and implementation
The source code is freely available at https://
Reproduced under the paper's license (CC BY), from the paper cited above.
Tracing map
Proposed by the machine: these links were found in the paper and verified at the source, without human review. The map will receive a Zenodo DOI once one of the paper's authors has validated it with their ORCID.
What the map holds:
- 2 repositories of the authors' code, each at its verified commit, with its license and how the link was found in the paper;
- 14 scripts, each with its path and the digest of its content;
- 3 matches between paragraphs of the paper and lines of the code (method lexical-v1);
- neither the text of the paper nor the code itself.
Its JSON (tracing-map.json) is deposited on Zenodo with its DOI once the map is validated.
Data
No dataset and no data link were found in the paper.
Data availability
The source code of SIMLINK is freely available at GitHub (https://
Reproduced under the paper's license (CC BY), from the paper cited above.
Versions
The history of this record: each version stored by the harvester or made by a correction of its authors or of the maintainers of its code, and what changed in its facts. The texts of the paper (its abstract, its availability statements) are not part of it; versions that changed only those are not listed.
Version 1, 27 September 2026: the first record
Recorded: type, language, journal, volume, issue, pages, dates, 6 authors, 7 MeSH terms, 3 funders, 40 references.
Cite
This paper
Li, H.-D., Wang, C., Yan, D., Huang, W., Li, Z., & Wang, S. (2026). SIMLINK enables accurate variant pathogenicity prediction through modeling the gene-variant-feature association structure. Bioinformatics (Oxford, England), 42(8), btag601. https://
BibTeX
@article{li2026simlink,
author = {Li, Hong-Dong and Wang, Chenlu and Yan, Dongfang and Huang, Wenkui and Li, Zongxuan and Wang, Shaokai},
title = {{SIMLINK enables accurate variant pathogenicity prediction through modeling the gene-variant-feature association structure}},
journal = {Bioinformatics (Oxford, England)},
year = {2026},
month = aug,
volume = {42},
number = {8},
pages = {btag601},
publisher = {Oxford University Press},
issn = {1367-4803},
doi = {10.1093/
url = {https://
pmid = {42574509},
pmcid = {PMC13505626}
}
RIS
TY - JOUR
AU - Li, Hong-Dong
AU - Wang, Chenlu
AU - Yan, Dongfang
AU - Huang, Wenkui
AU - Li, Zongxuan
AU - Wang, Shaokai
TI - SIMLINK enables accurate variant pathogenicity prediction through modeling the gene-variant-feature association structure
T2 - Bioinformatics (Oxford, England)
J2 - Bioinformatics
PY - 2026
DA - 2026/
VL - 42
IS - 8
SP - btag601
SN - 1367-4803
PB - Oxford University Press
DO - 10.1093/
UR - https://
LA - en
ER -
CSL-JSON
{
"id": "10.1093/
"type": "article-journal",
"title": "SIMLINK enables accurate variant pathogenicity prediction through modeling the gene-variant-feature association structure",
"container-title": "Bioinformatics (Oxford, England)",
"author": [
{
"family": "Li",
"given": "Hong-Dong"
},
{
"family": "Wang",
"given": "Chenlu"
},
{
"family": "Yan",
"given": "Dongfang"
},
{
"family": "Huang",
"given": "Wenkui"
},
{
"family": "Li",
"given": "Zongxuan"
},
{
"family": "Wang",
"given": "Shaokai"
}
],
"container-title-short":
"volume": "42",
"issue": "8",
"page": "btag601",
"DOI": "10.1093/
"PMID": "42574509",
"PMCID": "PMC13505626",
"ISSN": "1367-4803",
"publisher": "Oxford University Press",
"URL": "https://
"language": "en",
"issued": {
"date-parts": [
[
2026,
8,
1
]
]
}
}
The tracing map gets a citation of its own once an author has validated it and it has a DOI.
Similar papers
The papers with a page that share the most with this one: the tools found in their code, their categories, datasets, cited references and authors, the rarest counting most.
- [1] doi:10.1126/sciadv.adq6577 [code]
- Autism-like phenotypes and increased NMDAR2D expression in mice with KDM5B histone lysine demethylase deficiency.Journal: Science advancesIn common: pandas, SciPy, NumPy, autism, 4 references
- [2] doi:10.1038/s41467-026-72598-z [code]
- Functional impact of genetic background on variable expressivity in neurodevelopmental disorders.Journal: Nature communicationsIn common: pandas, SciPy, NumPy, 4 references
- [3] doi:10.1101/gr.280394.124 [code]
- De novo structural variants in autism spectrum disorder disrupt distal regulatory interactions of neuronal genes.Journal: Genome researchIn common: TensorFlow, scikit-learn, pandas, 2 other tools, autism, 1 reference
- [4] doi:10.1038/s42003-026-10380-z [code]
- Identification of moderate effect size genes in autism spectrum disorder through a novel gene pairing approach.Journal: Communications biologyIn common: autism, 4 references
- [5] doi:10.1038/s41586-026-10679-1 [code]
- Cortical development dynamics across autism spectrum disorder mouse models.Journal: NatureIn common: scikit-learn, pandas, SciPy, 1 other tool, autism, 2 references
- [6] doi:10.1038/s41586-026-10515-6 [code]
- An X-linked long non-coding RNA, PTCHD1-AS, and the core features of autism.Journal: NatureIn common: pandas, SciPy, NumPy, autism, 2 references
- [7] doi:10.1016/j.xhgg.2026.100652 [code]
- CRISPR-engineered deletion of POGZ alters transcription factor binding at promoters of genes involved in synaptic signaling.Journal: HGG advancesIn common: pandas, SciPy, NumPy, autism, 2 references
- [8] doi:10.1038/s41598-026-55163-y [code]
- Autism spectrum disorder identification using machine learning models on MRI data.Journal: Scientific reportsIn common: TensorFlow, scikit-learn, pandas, 2 other tools, autism
- [9] doi:10.1016/j.xgen.2026.101284 [code]
- NERINE reveals rare variant associations in gene networks across phenotypes and implicates an SNCA-PRL-LRRK2 subnetwork in Parkinson's disease.Journal: Cell genomicsIn common: pandas, SciPy, NumPy, 2 references
- [10] doi:10.7554/elife.110588 [code]
- Opening the black box toward a modular approach to spike sorting.Journal: eLifeIn common: TensorFlow, scikit-learn, pandas, 2 other tools, methods / tools
Contribute
The authors of this paper can claim it, correct its record and validate its tracing map, and the maintainers of its code (its owner, or a public member of its organization) correct what it says of their repository; anyone signed in can ask for its removal. Every request goes to OSCR's own machine, which answers it; your account page follows them.
Sign in with ORCID to claim this paper as one of its authors, correct its record or validate its tracing map: when the paper's metadata lists your ORCID iD, you are recognized at once. Maintainers of its code: sign in with GitHub, then claim the repository on your account page.
Claim this paper
Correct its record
Say what each link of this record is, remove the ones that are not the paper's, add the ones that are missing. The correction becomes a new version of the record, in its Versions section.
Validate its tracing map
You validate the map as this page shows it: 2 repositories of the authors' code, each at its verified commit and with its license, 14 scripts, and 3 matches between paragraphs and code (see the Code and Map sections). It then receives a DOI on Zenodo, with you (your ORCID iD) and OSCR as its creators; the code itself is not deposited.
The map's fingerprint: sha256:46a8c1eb1e2ad5b5…
Add the badge to its README
The badge links the code to this page. Copy one of these into the README of the paper's code: only you decide where it goes, and nothing is changed for you.
Markdown
[, paste the snippet at the top, then “Commit changes…” and, to review it first, “Create a new branch and start a pull request”. You open the pull request; OSCR asks for no permission.
Request its removal
To ask OSCR to remove this record, the copies of its authors' scripts or its tracing map, use the removal request page: signed in, you say who you are, what to remove and why, then review and confirm the request. Published rules decide every request (how).
Discussion, reproductions, activity
Discussion: questions and error reports about this paper and its code, from signed-in readers and its authors. It opens with sign-in.
Reproductions: reports from readers who ran the authors' code: what they reproduced, with which environment, commit and data. It opens with sign-in.
Activity: what happens around this paper: new versions of its record, its map's validation, discussions and reproductions. It opens with sign-in.
