OSCR

Intracellular protein GBF1 displays significant associations with amyloid pathology in Alzheimer's disease.

Code ↔ Paper

1 match between paragraphs of the paper and lines of its authors' code, computed by the harvester (lexical-v1). Click a colored paragraph or line to see its counterpart.

The 1 match
  1. [1] § METHODS › Genetic discovery ↔ bcbio/pipeline/main.py, lines 116–202 · score 0.65 · quality control, Variant calls, pruned, jointly, GEMINI, SNPEff

Paper

Loaded from Europe PMC by your browser, not stored by OSCR: doi.org · Europe PMC

The paper is loaded when this pane is shown.

The authors' code

Python · 533 lines · 27 KB · MIT · 1 match

  1. """Main entry point for distributed next-gen sequencing pipelines.
  2. Handles running the full pipeline based on instructions
  3. """
  4. from __future__ import print_function
  5. from collections import defaultdict
  6. import copy
  7. import os
  8. import sys
  9. import resource
  10. import tempfile
  11. import toolz as tz
  12. from bcbio import log, heterogeneity, hla, structural, utils
  13. from bcbio.cwl.inspect import initialize_watcher
  14. from bcbio.distributed import prun
  15. from bcbio.distributed.transaction import tx_tmpdir
  16. from bcbio.log import logger, DEFAULT_LOG_DIR
  17. from bcbio.ngsalign import alignprep
  18. from bcbio.pipeline import datadict as dd
  19. from bcbio.pipeline import (archive, config_utils, disambiguate, region,
  20. run_info, qcsummary, rnaseq)
  21. from bcbio.provenance import profile, system
  22. from bcbio.variation import (ensemble, genotype, population, validate, joint,
  23. peddy)
  24. from bcbio.chipseq import peaks, atac
  25. def run_main(workdir, config_file=None, fc_dir=None, run_info_yaml=None,
  26. parallel=None, workflow=None):
  27. """Run variant analysis, handling command line options.
  28. """
  29. # Set environment to standard to use periods for decimals and avoid localization
  30. locale_to_use = utils.get_locale()
  31. os.environ["LC_ALL"] = locale_to_use
  32. os.environ["LC"] = locale_to_use
  33. os.environ["LANG"] = locale_to_use
  34. workdir = utils.safe_makedir(os.path.abspath(workdir))
  35. os.chdir(workdir)
  36. config, config_file = config_utils.load_system_config(config_file, workdir)
  37. parallel = log.create_base_logger(config, parallel)
  38. log.setup_local_logging(config, parallel)
  39. logger.info(f"System YAML configuration: {os.path.abspath(config_file)}.")
  40. logger.info(f"Locale set to {locale_to_use}.")
  41. if config.get("log_dir", None) is None:
  42. config["log_dir"] = os.path.join(workdir, DEFAULT_LOG_DIR)
  43. if parallel["type"] in ["local", "clusterk"]:
  44. _setup_resources()
  45. _run_toplevel(config, config_file, workdir, parallel,
  46. fc_dir, run_info_yaml)
  47. elif parallel["type"] == "ipython":
  48. assert parallel["scheduler"] is not None, "IPython parallel requires a specified scheduler (-s)"
  49. if parallel["scheduler"] != "sge":
  50. assert parallel["queue"] is not None, "IPython parallel requires a specified queue (-q)"
  51. elif not parallel["queue"]:
  52. parallel["queue"] = ""
  53. _run_toplevel(config, config_file, workdir, parallel,
  54. fc_dir, run_info_yaml)
  55. else:
  56. raise ValueError("Unexpected type of parallel run: %s" % parallel["type"])
  57. def _setup_resources():
  58. """Attempt to increase resource limits up to hard limits.
  59. This allows us to avoid out of file handle limits where we can
  60. move beyond the soft limit up to the hard limit.
  61. """
  62. target_procs = 10240
  63. cur_proc, max_proc = resource.getrlimit(resource.RLIMIT_NPROC)
  64. target_proc = min(max_proc, target_procs) if max_proc > 0 else target_procs
  65. resource.setrlimit(resource.RLIMIT_NPROC, (max(cur_proc, target_proc), max_proc))
  66. cur_hdls, max_hdls = resource.getrlimit(resource.RLIMIT_NOFILE)
  67. target_hdls = min(max_hdls, target_procs) if max_hdls > 0 else target_procs
  68. resource.setrlimit(resource.RLIMIT_NOFILE, (max(cur_hdls, target_hdls), max_hdls))
  69. def _run_toplevel(config, config_file, work_dir, parallel,
  70. fc_dir=None, run_info_yaml=None):
  71. """
  72. Run toplevel analysis, processing a set of input files.
  73. config_file -- Main YAML configuration file with system parameters
  74. fc_dir -- Directory of fastq files to process
  75. run_info_yaml -- YAML configuration file specifying inputs to process
  76. """
  77. dirs = run_info.setup_directories(work_dir, fc_dir, config, config_file)
  78. config_file = os.path.join(dirs["config"], os.path.basename(config_file))
  79. pipelines, config = _pair_samples_with_pipelines(run_info_yaml, config)
  80. system.write_info(dirs, parallel, config)
  81. with tx_tmpdir(config if parallel.get("type") == "local" else None) as tmpdir:
  82. tempfile.tempdir = tmpdir
  83. for pipeline, samples in pipelines.items():
  84. for xs in pipeline(config, run_info_yaml, parallel, dirs, samples):
  85. pass
  86. # ## Generic pipeline framework
  87. def _wres(parallel, progs, fresources=None, ensure_mem=None):
  88. """Add resource information to the parallel environment on required programs and files.
  89. Enables spinning up required machines and operating in non-shared filesystem
  90. environments.
  91. progs -- Third party tools used in processing
  92. fresources -- Required file-based resources needed. These will be transferred on non-shared
  93. filesystems.
  94. ensure_mem -- Dictionary of required minimum memory for programs used. Ensures
  95. enough memory gets allocated on low-core machines.
  96. """
  97. parallel = copy.deepcopy(parallel)
  98. parallel["progs"] = progs
  99. if fresources:
  100. parallel["fresources"] = fresources
  101. if ensure_mem:
  102. parallel["ensure_mem"] = ensure_mem
  103. return parallel
  104. def variant2pipeline(config, run_info_yaml, parallel, dirs, samples):
  105. ## Alignment and preparation requiring the entire input file (multicore cluster)
  106. # Assign GATK supplied memory if required for post-process recalibration
  107. align_programs = ["aligner", "samtools", "sambamba"]
  108. if any(tz.get_in(["algorithm", "recalibrate"], utils.to_single_data(d)) in [True, "gatk"] for d in samples):
  109. align_programs.append("gatk")
  110. with prun.start(_wres(parallel, align_programs,
  111. (["reference", "fasta"], ["reference", "aligner"], ["files"])),
  112. samples, config, dirs, "multicore",
  113. multiplier=alignprep.parallel_multiplier(samples)) as run_parallel:
  114. with profile.report("organize samples", dirs):
  115. samples = run_parallel("organize_samples", [[dirs, config, run_info_yaml,
  116. [x[0]["description"] for x in samples]]])
  117. with profile.report("alignment preparation", dirs):
  118. samples = run_parallel("prep_align_inputs", samples)
  119. samples = run_parallel("disambiguate_split", [samples])
  120. with profile.report("alignment", dirs):
  121. samples = run_parallel("process_alignment", samples)
  122. samples = disambiguate.resolve(samples, run_parallel)
  123. samples = alignprep.merge_split_alignments(samples, run_parallel)
  124. with profile.report("callable regions", dirs):
  125. samples = run_parallel("prep_samples", [samples])
  126. samples = run_parallel("postprocess_alignment", samples)
  127. samples = run_parallel("combine_sample_regions", [samples])
  128. samples = run_parallel("calculate_sv_bins", [samples])
  129. samples = run_parallel("calculate_sv_coverage", samples)
  130. samples = run_parallel("normalize_sv_coverage", [samples])
  131. samples = region.clean_sample_data(samples)
  132. with profile.report("hla typing", dirs):
  133. samples = hla.run(samples, run_parallel)
  134. ## Variant calling on sub-regions of the input file (full cluster)
  135. with prun.start(_wres(parallel, ["gatk", "picard", "variantcaller"]),
  136. samples, config, dirs, "full",
  137. multiplier=region.get_max_counts(samples), max_multicore=1) as run_parallel:
  138. with profile.report("alignment post-processing", dirs):
  139. samples = region.parallel_prep_region(samples, run_parallel)
  140. with profile.report("variant calling", dirs):
  141. samples = genotype.parallel_variantcall_region(samples, run_parallel)
  142. with profile.report("joint squaring off/backfilling", dirs):
  143. samples = joint.square_off(samples, run_parallel)
  144. ## Finalize variants, BAMs and population databases (per-sample multicore cluster)
  145. with prun.start(_wres(parallel, ["gatk", "gatk-vqsr", "snpeff", "bcbio_variation",
  146. "gemini", "samtools", "fastqc", "sambamba",
  147. "bcbio-variation-recall", "qsignature",
  148. "svcaller", "kraken", "preseq"]),
  149. samples, config, dirs, "multicore2",
  150. multiplier=structural.parallel_multiplier(samples)) as run_parallel:
  151. with profile.report("variant post-processing", dirs):
  152. samples = run_parallel("postprocess_variants", samples)
  153. samples = run_parallel("split_variants_by_sample", samples)
  154. with profile.report("prepped BAM merging", dirs):
  155. samples = region.delayed_bamprep_merge(samples, run_parallel)
  156. with profile.report("validation", dirs):
  157. samples = run_parallel("compare_to_rm", samples)
  158. samples = genotype.combine_multiple_callers(samples)
  159. with profile.report("ensemble calling", dirs):
  160. samples = ensemble.combine_calls_parallel(samples, run_parallel)
  161. with profile.report("validation summary", dirs):
  162. samples = validate.summarize_grading(samples)
  163. with profile.report("structural variation", dirs):
  164. samples = structural.run(samples, run_parallel, "initial")
  165. with profile.report("structural variation", dirs):
  166. samples = structural.run(samples, run_parallel, "standard")
  167. with profile.report("structural variation ensemble", dirs):
  168. samples = structural.run(samples, run_parallel, "ensemble")
  169. with profile.report("structural variation validation", dirs):
  170. samples = run_parallel("validate_sv", samples)
  171. with profile.report("heterogeneity", dirs):
  172. samples = heterogeneity.run(samples, run_parallel)
  173. with profile.report("population database", dirs):
  174. samples = population.prep_db_parallel(samples, run_parallel)
  175. # after SV calling and SNV merging
  176. with profile.report("create CNV PON", dirs):
  177. samples = structural.create_cnv_pon(samples)
  178. with profile.report("peddy check", dirs):
  179. samples = peddy.run_peddy_parallel(samples, run_parallel)
  180. with profile.report("quality control", dirs):
  181. samples = qcsummary.generate_parallel(samples, run_parallel)
  182. with profile.report("archive", dirs):
  183. samples = archive.compress(samples, run_parallel)
  184. with profile.report("upload", dirs):
  185. samples = run_parallel("upload_samples", samples)
  186. for sample in samples:
  187. run_parallel("upload_samples_project", [sample])
  188. logger.info("Timing: finished")
  189. return samples
  190. def _debug_samples(i, samples):
  191. print("---", i, len(samples))
  192. for sample in (utils.to_single_data(x) for x in samples):
  193. print(" ", sample["description"], sample.get("region"), \
  194. utils.get_in(sample, ("config", "algorithm", "variantcaller")), \
  195. utils.get_in(sample, ("config", "algorithm", "jointcaller")), \
  196. utils.get_in(sample, ("metadata", "batch")), \
  197. [x.get("variantcaller") for x in sample.get("variants", [])], \
  198. sample.get("work_bam"), \
  199. sample.get("vrn_file"))
  200. def standardpipeline(config, run_info_yaml, parallel, dirs, samples):
  201. ## Alignment and preparation requiring the entire input file (multicore cluster)
  202. with prun.start(_wres(parallel, ["aligner", "samtools", "sambamba"]),
  203. samples, config, dirs, "multicore") as run_parallel:
  204. with profile.report("organize samples", dirs):
  205. samples = run_parallel("organize_samples", [[dirs, config, run_info_yaml,
  206. [x[0]["description"] for x in samples]]])
  207. with profile.report("alignment", dirs):
  208. samples = run_parallel("process_alignment", samples)
  209. with profile.report("callable regions", dirs):
  210. samples = run_parallel("prep_samples", [samples])
  211. samples = run_parallel("postprocess_alignment", samples)
  212. samples = run_parallel("combine_sample_regions", [samples])
  213. samples = region.clean_sample_data(samples)
  214. ## Quality control
  215. with prun.start(_wres(parallel, ["fastqc", "qsignature", "kraken", "gatk", "samtools", "preseq"]),
  216. samples, config, dirs, "multicore2") as run_parallel:
  217. with profile.report("quality control", dirs):
  218. samples = qcsummary.generate_parallel(samples, run_parallel)
  219. with profile.report("upload", dirs):
  220. samples = run_parallel("upload_samples", samples)
  221. for sample in samples:
  222. run_parallel("upload_samples_project", [sample])
  223. logger.info("Timing: finished")
  224. return samples
  225. def rnaseqpipeline(config, run_info_yaml, parallel, dirs, samples):
  226. samples = rnaseq_prep_samples(config, run_info_yaml, parallel, dirs, samples)
  227. with prun.start(_wres(parallel, ["aligner", "picard", "samtools"],
  228. ensure_mem={"tophat": 10, "tophat2": 10, "star": 2, "hisat2": 8}),
  229. samples, config, dirs, "alignment",
  230. multiplier=alignprep.parallel_multiplier(samples)) as run_parallel:
  231. with profile.report("alignment", dirs):
  232. samples = run_parallel("disambiguate_split", [samples])
  233. samples = run_parallel("process_alignment", samples)
  234. with prun.start(_wres(parallel, ["samtools", "cufflinks"]),
  235. samples, config, dirs, "rnaseqcount") as run_parallel:
  236. with profile.report("disambiguation", dirs):
  237. samples = disambiguate.resolve(samples, run_parallel)
  238. with profile.report("transcript assembly", dirs):
  239. samples = rnaseq.assemble_transcripts(run_parallel, samples)
  240. with profile.report("estimate expression (threaded)", dirs):
  241. samples = rnaseq.quantitate_expression_parallel(samples, run_parallel)
  242. with prun.start(_wres(parallel, ["dexseq", "express"]), samples, config,
  243. dirs, "rnaseqcount-singlethread", max_multicore=1) as run_parallel:
  244. with profile.report("estimate expression (single threaded)", dirs):
  245. samples = rnaseq.quantitate_expression_noparallel(samples, run_parallel)
  246. samples = rnaseq.combine_files(samples)
  247. with prun.start(_wres(parallel, ["gatk", "vardict"]), samples, config,
  248. dirs, "rnaseq-variation") as run_parallel:
  249. with profile.report("RNA-seq variant calling", dirs):
  250. samples = rnaseq.rnaseq_variant_calling(samples, run_parallel)
  251. with prun.start(_wres(parallel, ["samtools", "fastqc", "qualimap",
  252. "kraken", "gatk", "preseq"], ensure_mem={"qualimap": 4}),
  253. samples, config, dirs, "qc") as run_parallel:
  254. with profile.report("quality control", dirs):
  255. samples = qcsummary.generate_parallel(samples, run_parallel)
  256. with profile.report("create SummarizedExperiment object", dirs):
  257. samples = rnaseq.load_summarizedexperiment(samples)
  258. with profile.report("upload", dirs):
  259. samples = run_parallel("upload_samples", samples)
  260. for sample in samples:
  261. run_parallel("upload_samples_project", [sample])
  262. with profile.report("bcbioRNAseq loading", dirs):
  263. tools_on = dd.get_in_samples(samples, dd.get_tools_on)
  264. bcbiornaseq_on = tools_on and "bcbiornaseq" in tools_on
  265. if bcbiornaseq_on:
  266. if len(samples) < 3:
  267. logger.warn("bcbioRNASeq needs at least three samples total, skipping.")
  268. elif len(samples) > 100:
  269. logger.warn("Over 100 samples, skipping bcbioRNASeq.")
  270. else:
  271. run_parallel("run_bcbiornaseqload", [sample])
  272. logger.info("Timing: finished")
  273. return samples
  274. def fastrnaseqpipeline(config, run_info_yaml, parallel, dirs, samples):
  275. samples = rnaseq_prep_samples(config, run_info_yaml, parallel, dirs, samples)
  276. ww = initialize_watcher(samples)
  277. with prun.start(_wres(parallel, ["samtools"]), samples, config,
  278. dirs, "fastrnaseq") as run_parallel:
  279. with profile.report("fastrnaseq", dirs):
  280. samples = rnaseq.fast_rnaseq(samples, run_parallel)
  281. ww.report("fastrnaseq", samples)
  282. samples = rnaseq.combine_files(samples)
  283. with profile.report("quality control", dirs):
  284. samples = qcsummary.generate_parallel(samples, run_parallel)
  285. ww.report("qcsummary", samples)
  286. with profile.report("upload", dirs):
  287. samples = run_parallel("upload_samples", samples)
  288. for samples in samples:
  289. run_parallel("upload_samples_project", [samples])
  290. logger.info("Timing: finished")
  291. return samples
  292. def singlecellrnaseqpipeline(config, run_info_yaml, parallel, dirs, samples):
  293. samples = rnaseq_prep_samples(config, run_info_yaml, parallel, dirs, samples)
  294. with prun.start(_wres(parallel, ["samtools", "rapmap"]), samples, config,
  295. dirs, "singlecell-rnaseq") as run_parallel:
  296. with profile.report("singlecell-rnaseq", dirs):
  297. samples = rnaseq.singlecell_rnaseq(samples, run_parallel)
  298. with profile.report("quality control", dirs):
  299. samples = qcsummary.generate_parallel(samples, run_parallel)
  300. with profile.report("upload", dirs):
  301. samples = run_parallel("upload_samples", samples)
  302. for samples in samples:
  303. run_parallel("upload_samples_project", [samples])
  304. logger.info("Timing: finished")
  305. return samples
  306. def smallrnaseqpipeline(config, run_info_yaml, parallel, dirs, samples):
  307. # causes a circular import at the top level
  308. from bcbio.srna.group import report as srna_report
  309. samples = rnaseq_prep_samples(config, run_info_yaml, parallel, dirs, samples)
  310. with prun.start(_wres(parallel, ["aligner", "picard", "samtools"],
  311. ensure_mem={"bowtie": 8, "bowtie2": 8, "star": 2}),
  312. [samples[0]], config, dirs, "alignment") as run_parallel:
  313. with profile.report("prepare", dirs):
  314. samples = run_parallel("seqcluster_prepare", [samples])
  315. with profile.report("seqcluster alignment", dirs):
  316. samples = run_parallel("srna_alignment", [samples])
  317. with prun.start(_wres(parallel, ["aligner", "picard", "samtools"],
  318. ensure_mem={"tophat": 10, "tophat2": 10, "star": 2, "hisat2": 8}),
  319. samples, config, dirs, "alignment_samples",
  320. multiplier=alignprep.parallel_multiplier(samples)) as run_parallel:
  321. with profile.report("alignment", dirs):
  322. samples = run_parallel("process_alignment", samples)
  323. with prun.start(_wres(parallel, ["picard", "miraligner"]),
  324. samples, config, dirs, "annotation") as run_parallel:
  325. with profile.report("small RNA annotation", dirs):
  326. samples = run_parallel("srna_annotation", samples)
  327. with prun.start(_wres(parallel, ["seqcluster", "mirge"],
  328. ensure_mem={"seqcluster": 8}),
  329. [samples[0]], config, dirs, "cluster") as run_parallel:
  330. with profile.report("cluster", dirs):
  331. samples = run_parallel("seqcluster_cluster", [samples])
  332. with prun.start(_wres(parallel, ["picard", "fastqc"]),
  333. samples, config, dirs, "qc") as run_parallel:
  334. with profile.report("quality control", dirs):
  335. samples = qcsummary.generate_parallel(samples, run_parallel)
  336. with profile.report("report", dirs):
  337. srna_report(samples)
  338. with profile.report("upload", dirs):
  339. samples = run_parallel("upload_samples", samples)
  340. for sample in samples:
  341. run_parallel("upload_samples_project", [sample])
  342. return samples
  343. def chipseqpipeline(config, run_info_yaml, parallel, dirs, samples):
  344. with prun.start(_wres(parallel, ["aligner", "picard"]),
  345. samples, config, dirs, "multicore",
  346. multiplier=alignprep.parallel_multiplier(samples)) as run_parallel:
  347. with profile.report("organize samples", dirs):
  348. samples = run_parallel("organize_samples", [[dirs, config, run_info_yaml,
  349. [x[0]["description"] for x in samples]]])
  350. with profile.report("alignment", dirs):
  351. samples = run_parallel("prepare_sample", samples)
  352. samples = run_parallel("trim_sample", samples)
  353. samples = run_parallel("disambiguate_split", [samples])
  354. samples = run_parallel("process_alignment", samples)
  355. with profile.report("disambiguation", dirs):
  356. samples = disambiguate.resolve(samples, run_parallel)
  357. samples = run_parallel("clean_chipseq_alignment", samples)
  358. with prun.start(_wres(parallel, ["peakcaller"]),
  359. samples, config, dirs, "peakcalling",
  360. multiplier = peaks._get_multiplier(samples)) as run_parallel:
  361. with profile.report("peakcalling", dirs):
  362. samples = peaks.peakcall_prepare(samples, run_parallel)
  363. samples = peaks.call_consensus(samples)
  364. samples = run_parallel("run_chipseq_count", samples)
  365. samples = peaks.create_peaktable(samples)
  366. with prun.start(_wres(parallel, ["picard", "fastqc"]),
  367. samples, config, dirs, "qc") as run_parallel:
  368. with profile.report("quality control", dirs):
  369. samples = qcsummary.generate_parallel(samples, run_parallel)
  370. samples = atac.create_ataqv_report(samples)
  371. with profile.report("upload", dirs):
  372. samples = run_parallel("upload_samples", samples)
  373. for sample in samples:
  374. run_parallel("upload_samples_project", [sample])
  375. logger.info("Timing: finished")
  376. return samples
  377. def wgbsseqpipeline(config, run_info_yaml, parallel, dirs, samples):
  378. with prun.start(_wres(parallel, ["fastqc", "picard"], ensure_mem={"fastqc" : 4}),
  379. samples, config, dirs, "trimming") as run_parallel:
  380. with profile.report("organize samples", dirs):
  381. samples = run_parallel("organize_samples", [[dirs, config, run_info_yaml,
  382. [x[0]["description"] for x in samples]]])
  383. samples = run_parallel("prepare_sample", samples)
  384. samples = run_parallel("trim_bs_sample", samples)
  385. with prun.start(_wres(parallel, ["aligner", "bismark", "picard", "samtools"]),
  386. samples, config, dirs, "multicore",
  387. multiplier=alignprep.parallel_multiplier(samples)) as run_parallel:
  388. with profile.report("alignment", dirs):
  389. samples = run_parallel("process_alignment", samples)
  390. with prun.start(_wres(parallel, ['samtools']), samples, config, dirs,
  391. 'deduplication') as run_parallel:
  392. with profile.report('deduplicate', dirs):
  393. samples = run_parallel('deduplicate_bismark', samples)
  394. with prun.start(_wres(parallel, ["caller"], ensure_mem={"caller": 5}),
  395. samples, config, dirs, "multicore2",
  396. multiplier=24) as run_parallel:
  397. with profile.report("cpg calling", dirs):
  398. samples = run_parallel("cpg_calling", samples)
  399. with prun.start(_wres(parallel, ["picard", "fastqc", "samtools"]),
  400. samples, config, dirs, "qc") as run_parallel:
  401. with profile.report("quality control", dirs):
  402. samples = qcsummary.generate_parallel(samples, run_parallel)
  403. with profile.report("upload", dirs):
  404. samples = run_parallel("upload_samples", samples)
  405. for sample in samples:
  406. run_parallel("upload_samples_project", [sample])
  407. logger.info("Timing: finished")
  408. return samples
  409. def rnaseq_prep_samples(config, run_info_yaml, parallel, dirs, samples):
  410. """
  411. organizes RNA-seq and small-RNAseq samples, converting from BAM if
  412. necessary and trimming if necessary
  413. """
  414. pipeline = dd.get_in_samples(samples, dd.get_analysis)
  415. trim_reads_set = any([tz.get_in(["algorithm", "trim_reads"], d) for d in dd.sample_data_iterator(samples)])
  416. resources = ["picard"]
  417. needs_trimming = (_is_smallrnaseq(pipeline) or trim_reads_set)
  418. if needs_trimming:
  419. resources.append("atropos")
  420. with prun.start(_wres(parallel, resources),
  421. samples, config, dirs, "trimming",
  422. max_multicore=1 if not needs_trimming else None) as run_parallel:
  423. with profile.report("organize samples", dirs):
  424. samples = run_parallel("organize_samples", [[dirs, config, run_info_yaml,
  425. [x[0]["description"] for x in samples]]])
  426. samples = run_parallel("prepare_sample", samples)
  427. if needs_trimming:
  428. with profile.report("adapter trimming", dirs):
  429. if _is_smallrnaseq(pipeline):
  430. samples = run_parallel("trim_srna_sample", samples)
  431. else:
  432. samples = run_parallel("trim_sample", samples)
  433. return samples
  434. def _get_pipeline(item):
  435. from bcbio.log import logger
  436. analysis_type = item.get("analysis", "").lower()
  437. if analysis_type not in SUPPORTED_PIPELINES:
  438. logger.error("Cannot determine which type of analysis to run, "
  439. "set in the run_info under details.")
  440. sys.exit(1)
  441. else:
  442. return SUPPORTED_PIPELINES[analysis_type]
  443. def _pair_samples_with_pipelines(run_info_yaml, config):
  444. """Map samples defined in input file to pipelines to run.
  445. """
  446. samples = config_utils.load_config(run_info_yaml)
  447. if isinstance(samples, dict):
  448. resources = samples.pop("resources")
  449. samples = samples["details"]
  450. else:
  451. resources = {}
  452. ready_samples = []
  453. for sample in samples:
  454. if "files" in sample:
  455. del sample["files"]
  456. # add any resources to this item to recalculate global configuration
  457. usample = copy.deepcopy(sample)
  458. usample.pop("algorithm", None)
  459. if "resources" not in usample:
  460. usample["resources"] = {}
  461. for prog, pkvs in resources.items():
  462. if prog not in usample["resources"]:
  463. usample["resources"][prog] = {}
  464. if pkvs is not None:
  465. for key, val in pkvs.items():
  466. usample["resources"][prog][key] = val
  467. config = config_utils.update_w_custom(config, usample)
  468. sample["resources"] = {}
  469. ready_samples.append(sample)
  470. paired = [(x, _get_pipeline(x)) for x in ready_samples]
  471. d = defaultdict(list)
  472. for x in paired:
  473. d[x[1]].append([x[0]])
  474. return d, config
  475. SUPPORTED_PIPELINES = {"variant2": variant2pipeline,
  476. "snp calling": variant2pipeline,
  477. "variant": variant2pipeline,
  478. "standard": standardpipeline,
  479. "minimal": standardpipeline,
  480. "rna-seq": rnaseqpipeline,
  481. "smallrna-seq": smallrnaseqpipeline,
  482. "chip-seq": chipseqpipeline,
  483. "wgbs-seq": wgbsseqpipeline,
  484. "fastrna-seq": fastrnaseqpipeline,
  485. "scrna-seq": singlecellrnaseqpipeline}
  486. def _is_smallrnaseq(pipeline):
  487. return pipeline.lower() == "smallrna-seq"

main.py at commit 594339f, under MIT · at the source

Overview

Authors: Sean J. Miller1,2, Dmitry Prokopenko1, Ping Bai3, Prasenjit Mondal1, Abigael Scott1, Wei Zhang1, Ashley Gomm1, Siyi Zhang1, Daniel D. Child1, Nolan Shen1, Joseph Ward1, Scott Schulte1, Dan Lei1, Brian P. Hafler2,4,5, Changning Wang3, Rudolph E. Tanzi1, Can Zhang1
  1. Genetics and Aging Research Unit McCance Center for Brain Health MassGeneral Institute for Neurodegenerative Disease Department of Neurology Massachusetts General Hospital and Harvard Medical School Charlestown Massachusetts USA
  2. Department of Ophthalmology and Visual Science Yale School of Medicine New Haven Connecticut USA
  3. Athinoula A. Martinos Center for Biomedical Imaging Department of Radiology Massachusetts General Hospital and Harvard Medical School Charlestown Massachusetts USA
  4. Department of Pathology Yale School of Medicine New Haven Connecticut USA
  5. The Broad Institute MIT and Harvard Cambridge Massachusetts USA
Institutions: Harvard University (United States); Yale University (United States); Massachusetts General Hospital (United States); Athinoula A. Martinos Center for Biomedical Imaging (United States); Broad Institute (United States); Massachusetts Institute of Technology (United States)
Journal: Alzheimer's & dementia : the journal of the Alzheimer's Association, volume 22, issue 4, article e71271
Dates: received 25 August 2025; accepted 30 December 2025; published online 16 April 2026; in print April 2026
Type: Research article · Language: English
License: CC BY-NC-ND
Identifiers: DOI 10.1002/alz.71271 · PMID 41988936 · PMCID PMC13084705 · OpenAlex W7154624661
Open access: hybrid, a free copy (OpenAlex)
Status: code verified
Categories: histology / microscopy (modality), human (organism), mouse (organism), Alzheimer's / dementia (population), cellular / molecular (subfield)
Methods: Statistics, Smoothing, state filtering, decompositions, Machine learning, Evoked potentials
Keywords: Alzheimer's disease, amyloid beta, amyloid precursor protein, Golgi brefeldin A‐resistant guanine nucleotide exchange factor‐1, golgicide A, neuropathology
MeSH: Alzheimer Disease*, Brain*, Guanine Nucleotide Exchange Factors*, Plaque, Amyloid*, Amyloid beta-Peptides, Amyloid beta-Protein Precursor, Animals, Disease Models, Animal, Female, Humans, Male, Mice, Mice, Transgenic (* major topic)
Topic: Alzheimer's disease research and treatments (Physiology, Medicine), according to OpenAlex
Funding: Cure Alzheimer’s Fund; MADRC (1P30AG062421‐01); U.S. Department of Health & Human Services | NIH | National Eye Institute (NEI) (R01‐EY034234); H. Eric Cushing Foundation; Nancy Lurie Marks Family Foundation; C.J.L. charitable Foundation; NIH (R01AG077016)
Citations: not cited yet (Europe PMC); 80 references in the paper

Abstract

The abstract is not reproduced here: the paper's license (CC BY-NC-ND) does not allow it. Read it in the paper, at the publisher or on Europe PMC.

Repository

Its files are read in the Code ↔ Paper reader above, with 1 match between paragraphs and lines of code.

chapmanb/bcbio-nextgen

License: MIT
State: the link answers, verified on 28 September 2026
Evidence: files inventoried
Commit: 594339fd1a9694c8f8ed61084e2bd5740cfd59b0, 24 August 2024
Languages: Python (297), Shell (17), R (2)
Size: 581 files, 316 scripts
Software Heritage: archived
Found in: the text, “Genetic discovery”
Holds: README, license file, environment (requirements-conda.txt, requirements-dev.txt, requirements.txt, setup.cfg, setup.py, docs/requirements-local.txt, docs/requirements.txt), tests, continuous integration, documentation, 1 notebook
Not found: CITATION.cff
Tools: pysam (31 files), BEDTools (27 files), pandas (27 files), NumPy (19 files), Biopython (5 files), SciPy (3 files), DESeq2 (2 files), SAMtools (2 files), statsmodels (2 files), tidyverse (2 files), BCFtools (1 file), ggplot2 (1 file), h5py (1 file), Matplotlib (1 file)
Availability: 1 check, the latest on 28 September 2026: the link answers
  • 28 September 2026: the link answers
318 files

Tracing map

Proposed by the machine: these links were found in the paper and verified at the source, without human review. The map will receive a Zenodo DOI once one of the paper's authors has validated it with their ORCID.

What the map holds:

  • 1 repository of the authors' code, each at its verified commit, with its license and how the link was found in the paper;
  • 316 scripts, each with its path and the digest of its content;
  • 1 match between paragraphs of the paper and lines of the code (method lexical-v1);
  • neither the text of the paper nor the code itself.

Its JSON (tracing-map.json) is deposited on Zenodo with its DOI once the map is validated.

Data

No dataset and no data link were found in the paper.

Versions

The history of this record: each version stored by the harvester or made by a correction of its authors or of the maintainers of its code, and what changed in its facts. The texts of the paper (its abstract, its availability statements) are not part of it; versions that changed only those are not listed.

Version 1, 28 September 2026: the first record

Recorded: type, language, journal, volume, issue, pages, dates, 17 authors, 6 keywords, 13 MeSH terms, 7 funders, 80 references.

Cite

This paper

Miller, S. J., Prokopenko, D., Bai, P., Mondal, P., Scott, A., Zhang, W., Gomm, A., Zhang, S., Child, D. D., Shen, N., Ward, J., Schulte, S., Lei, D., Hafler, B. P., Wang, C., Tanzi, R. E., & Zhang, C. (2026). Intracellular protein GBF1 displays significant associations with amyloid pathology in Alzheimer's disease. Alzheimer's & dementia : the journal of the Alzheimer's Association, 22(4), e71271. https://doi.org/10.1002/alz.71271

BibTeX

@article{miller2026intracellular,
author = {Miller, Sean J. and Prokopenko, Dmitry and Bai, Ping and Mondal, Prasenjit and Scott, Abigael and Zhang, Wei and Gomm, Ashley and Zhang, Siyi and Child, Daniel D. and Shen, Nolan and Ward, Joseph and Schulte, Scott and Lei, Dan and Hafler, Brian P. and Wang, Changning and Tanzi, Rudolph E. and Zhang, Can},
title = {{Intracellular protein GBF1 displays significant associations with amyloid pathology in Alzheimer's disease}},
journal = {Alzheimer's \& dementia : the journal of the Alzheimer's Association},
year = {2026},
month = apr,
volume = {22},
number = {4},
pages = {e71271},
publisher = {Wiley},
issn = {1552-5260},
doi = {10.1002/alz.71271},
url = {https://doi.org/10.1002/alz.71271},
pmid = {41988936},
pmcid = {PMC13084705}
}

RIS

TY - JOUR
AU - Miller, Sean J.
AU - Prokopenko, Dmitry
AU - Bai, Ping
AU - Mondal, Prasenjit
AU - Scott, Abigael
AU - Zhang, Wei
AU - Gomm, Ashley
AU - Zhang, Siyi
AU - Child, Daniel D.
AU - Shen, Nolan
AU - Ward, Joseph
AU - Schulte, Scott
AU - Lei, Dan
AU - Hafler, Brian P.
AU - Wang, Changning
AU - Tanzi, Rudolph E.
AU - Zhang, Can
TI - Intracellular protein GBF1 displays significant associations with amyloid pathology in Alzheimer's disease
T2 - Alzheimer's & dementia : the journal of the Alzheimer's Association
J2 - Alzheimers Dement
PY - 2026
DA - 2026/04/01
VL - 22
IS - 4
SP - e71271
SN - 1552-5260
PB - Wiley
DO - 10.1002/alz.71271
UR - https://doi.org/10.1002/alz.71271
LA - en
ER -

CSL-JSON

{
"id": "10.1002/alz.71271",
"type": "article-journal",
"title": "Intracellular protein GBF1 displays significant associations with amyloid pathology in Alzheimer's disease",
"container-title": "Alzheimer's & dementia : the journal of the Alzheimer's Association",
"author": [
{
"family": "Miller",
"given": "Sean J."
},
{
"family": "Prokopenko",
"given": "Dmitry"
},
{
"family": "Bai",
"given": "Ping"
},
{
"family": "Mondal",
"given": "Prasenjit"
},
{
"family": "Scott",
"given": "Abigael"
},
{
"family": "Zhang",
"given": "Wei"
},
{
"family": "Gomm",
"given": "Ashley"
},
{
"family": "Zhang",
"given": "Siyi"
},
{
"family": "Child",
"given": "Daniel D."
},
{
"family": "Shen",
"given": "Nolan"
},
{
"family": "Ward",
"given": "Joseph"
},
{
"family": "Schulte",
"given": "Scott"
},
{
"family": "Lei",
"given": "Dan"
},
{
"family": "Hafler",
"given": "Brian P."
},
{
"family": "Wang",
"given": "Changning"
},
{
"family": "Tanzi",
"given": "Rudolph E."
},
{
"family": "Zhang",
"given": "Can"
}
],
"container-title-short": "Alzheimers Dement",
"volume": "22",
"issue": "4",
"page": "e71271",
"DOI": "10.1002/alz.71271",
"PMID": "41988936",
"PMCID": "PMC13084705",
"ISSN": "1552-5260",
"publisher": "Wiley",
"URL": "https://doi.org/10.1002/alz.71271",
"language": "en",
"issued": {
"date-parts": [
[
2026,
4,
1
]
]
}
}

The tracing map gets a citation of its own once an author has validated it and it has a DOI.

Similar papers

The papers with a page that share the most with this one: the tools found in their code, their categories, datasets, cited references and authors, the rarest counting most.

[1] doi:10.1038/s41467-026-76675-1 [code]
Long-read proteogenomic atlas of human neuronal differentiation reveals isoform diversity informing neurodevelopmental risk mechanisms.
Journal: Nature communications
In common: pysam, Biopython, BEDTools, 9 other tools
[2] doi:10.1038/s41592-026-03211-w [code]
Spatial isoform sequencing at single-cell resolution reveals cell-type-specific spatial isoform variability in multiple brain cell types.
Journal: Nature methods
In common: pysam, Biopython, BEDTools, 9 other tools, mouse
[3] doi:10.1038/s41467-026-71790-5 [code]
Recurrent DNA break clusters drive replication-stress-induced copy number variants and genome diversification.
Journal: Nature communications
In common: BCFtools, pysam, BEDTools, 8 other tools, mouse, cellular / molecular
[4] doi:10.1038/s41586-026-10512-9 [code]
Astrocyte glucocorticoid receptor signalling restricts neuronal plasticity.
Journal: Nature
In common: pysam, Biopython, BEDTools, 8 other tools, mouse, cellular / molecular
[5] doi:10.21203/rs.3.rs-9927928/v1 [code]
Genome-wide and allele-resolved maps of the radial architecture of the mouse genome
Journal: Research Square (preprint)
In common: BCFtools, pysam, BEDTools, 8 other tools, mouse
[6] doi:10.1126/sciadv.aed2952 [code]
Activation of transposable elements is linked to a region- and cell type-specific interferon response in Parkinson's disease.
Journal: Science advances
In common: pysam, Biopython, BEDTools, 8 other tools, cellular / molecular
[7] doi:10.1007/s00262-026-04390-3 [code]
Identification and prioritisation of tumour antigen candidates from 79 glioblastoma transcriptomes.
Journal: Cancer immunology, immunotherapy : CII
In common: BCFtools, pysam, BEDTools, 8 other tools, cellular / molecular
[8] doi:10.1038/s42003-026-10957-8 [code]
Brain defence by the extracellular matrix protein Cochlin.
Journal: Communications biology
In common: BCFtools, Biopython, SAMtools, 8 other tools, mouse, cellular / molecular
[9] doi:10.1038/s44318-026-00818-9 [code]
FAM134B-mediated ER-phagy degrades APP and suppresses Alzheimer's disease pathology.
Journal: The EMBO journal
In common: Biopython, BEDTools, DESeq2, 7 other tools, Alzheimer's / dementia, mouse, cellular / molecular, 1 reference
[10] doi:10.1093/molbev/msag035 [code]
Mammalian mitochondrial DNA accumulates insertions and deletions with age in energetically demanding tissues.
Journal: Molecular biology and evolution
In common: BCFtools, Biopython, BEDTools, 7 other tools, mouse, cellular / molecular

Contribute

The authors of this paper can claim it, correct its record and validate its tracing map, and the maintainers of its code (its owner, or a public member of its organization) correct what it says of their repository; anyone signed in can ask for its removal. Every request goes to OSCR's own machine, which answers it; your account page follows them.

Sign in with ORCID to claim this paper as one of its authors, correct its record or validate its tracing map: when the paper's metadata lists your ORCID iD, you are recognized at once. Maintainers of its code: sign in with GitHub, then claim the repository on your account page.

Request its removal

To ask OSCR to remove this record, the copies of its authors' scripts or its tracing map, use the removal request page: signed in, you say who you are, what to remove and why, then review and confirm the request. Published rules decide every request (how).

Discussion, reproductions, activity

Discussion: questions and error reports about this paper and its code, from signed-in readers and its authors. It opens with sign-in.

Reproductions: reports from readers who ran the authors' code: what they reproduced, with which environment, commit and data. It opens with sign-in.

Activity: what happens around this paper: new versions of its record, its map's validation, discussions and reproductions. It opens with sign-in.