Recurrent DNA break clusters drive replication-stress-induced copy number variants and genome diversification.
The 3 matches
- [1] § Methods › Alignment, QC, and CNV calling ↔ src/coral.h, lines 500–584 · score 0.60 · CNV calling, mapping quality, ploidy, Somatic, depth, genotyped
- [2] § Methods › Alignment, QC, and CNV calling ↔ workflow/scripts/haplotagging_scripts/haplotagTable.R, lines 16–63 · score 0.57 · findOverlaps, GRanges, exported, BED, Overlapping, filtered
- [3] § Methods › Demultiplexing and alignment ↔ transloc_pipeline_v3/bin/TranslocPipeline.pl, lines 1078–1147 · score 0.52 · TranslocWrapper, Bowtie2, pipeline, adapters, prey, bait
Paper
Loaded from Europe PMC by your browser, not stored by OSCR: doi.org · Europe PMC
The paper is loaded when this pane is shown.
The authors' code
C/C++ header · 933 lines · 35 KB · BSD-3-Clause · 1 match
- #ifndef CORAL_H
- #define CORAL_H
- #include <limits>
- #include <iomanip>
- #include <boost/icl/split_interval_map.hpp>
- #include <boost/dynamic_bitset.hpp>
- #include <boost/unordered_map.hpp>
- #include <boost/date_time/posix_time/posix_time.hpp>
- #include <boost/date_time/gregorian/gregorian.hpp>
- #include <boost/math/special_functions/round.hpp>
- #include <htslib/sam.h>
- #include <htslib/faidx.h>
- #include "bed.h"
- #include "scan.h"
- #include "gcbias.h"
- #include "cnv.h"
- #include "version.h"
- namespace torali
- {
- struct CountDNAConfig {
- bool basecov;
- bool somatic;
- bool adaptive;
- bool hasStatsFile;
- bool hasScanFile;
- bool noScanWindowSelection;
- bool regionalGc;
- bool hasSegFile;
- bool hasGenoFile;
- bool hasExcludeFile;
- uint32_t nchr;
- uint32_t minClip;
- uint32_t minRefSep;
- uint32_t minBpSupport;
- float cnMergeTol;
- float penalty;
- uint32_t meanisize;
- uint32_t window_size;
- uint32_t window_offset;
- uint32_t scanWindow;
- uint32_t minChrLen;
- uint32_t minCnvSize;
- uint32_t targetReads;
- uint32_t minSegWin;
- double targetExpCov;
- uint16_t minQual;
- uint16_t mapqUniq;
- uint16_t mad;
- float ploidy;
- float ctrlPloidy;
- float expectedCN;
- float purity;
- float exclgc;
- float uniqueToTotalCovRatio;
- float fracWindow;
- float fragmentUnique;
- int32_t cnvMinQual;
- float cnvDelConfirm;
- float cnvDelRatio;
- std::string sampleName;
- boost::filesystem::path segfile;
- boost::filesystem::path genofile;
- boost::filesystem::path outfile;
- boost::filesystem::path covfile;
- boost::filesystem::path genome;
- boost::filesystem::path statsFile;
- boost::filesystem::path bamFile;
- boost::filesystem::path scanFile;
- boost::filesystem::path exclude;
- std::set<int32_t> refIdx;
- };
- struct CovWin {
- uint32_t start;
- uint32_t end;
- uint32_t winlen;
- double covsum;
- double expcov;
- double ucov;
- double tcov;
- double aall;
- double eall;
- bool valid;
- CovWin(uint32_t const s, uint32_t const e, uint32_t const w, double const cs, double const ec, double const uc, double const tc, double const aa, double const ea, bool vld) : start(s), end(e), winlen(w), covsum(cs), expcov(ec), ucov(uc), tcov(tc), aall(aa), eall(ea), valid(vld) {}
- };
- struct CountDNAConfigLib {
- uint16_t madCutoff;
- uint16_t madNormalCutoff;
- boost::filesystem::path genome;
- std::vector<boost::filesystem::path> files;
- };
- template<typename TConfig>
- inline int32_t
- bamCount(TConfig const& c, LibraryInfo const& li, std::vector<GcBias> const& gcbias, std::pair<uint32_t, uint32_t> const& gcbound, std::vector<double> const& regcorr, uint32_t const regWin) {
- // Load bam file
- samFile* samfile = sam_open(c.bamFile.string().c_str(), "r");
- hts_set_fai_filename(samfile, c.genome.string().c_str());
- hts_idx_t* idx = sam_index_load(samfile, c.bamFile.string().c_str());
- bam_hdr_t* hdr = sam_hdr_read(samfile);
- // Parse BAM file
- boost::posix_time::ptime now = boost::posix_time::second_clock::local_time();
- std::cerr << '[' << boost::posix_time::to_simple_string(now) << "] " << "Count fragments" << std::endl;
- // Open output files
- boost::iostreams::filtering_ostream dataOut;
- if (!c.covfile.empty()) {
- dataOut.push(boost::iostreams::gzip_compressor());
- dataOut.push(boost::iostreams::file_sink(c.covfile.string(), std::ios_base::out | std::ios_base::binary));
- dataOut << "chr\tstart\tend\t" << c.sampleName << "_uniqfrac\t" << c.sampleName << "_logR\t" << c.sampleName << "_CN" << std::endl;
- }
- // CNVs
- std::vector<CNV> cnvs;
- if (c.hasGenoFile) parseVcfCNV(c, hdr, cnvs);
- // Iterate chromosomes
- faidx_t* faiRef = fai_load(c.genome.string().c_str());
- for(int32_t refIndex=0; refIndex < (int32_t) hdr->n_targets; ++refIndex) {
- if ((!c.hasGenoFile) && (chrNoData(c, refIndex, idx))) continue;
- // Haploid chromosome?
- float chrCtrlPloidy = c.ctrlPloidy;
- float chrPloidy = c.ploidy;
- if (c.refIdx.find(refIndex) != c.refIdx.end()) {
- chrCtrlPloidy = c.ctrlPloidy - 1;
- chrPloidy = c.ploidy - 1;
- }
- // Reference sequence
- std::string tname(hdr->target_name[refIndex]);
- int32_t seqlen = faidx_seq_len(faiRef, tname.c_str());
- if (seqlen == - 1) continue;
- else seqlen = -1;
- char* ref = faidx_fetch_seq(faiRef, tname.c_str(), 0, faidx_seq_len(faiRef, tname.c_str()), &seqlen);
- if (ref == NULL) continue;
- // Get GC content
- std::vector<uint16_t> uniqContent(hdr->target_len[refIndex], 0);
- std::vector<uint16_t> gcContent(hdr->target_len[refIndex], 0);
- {
- // GC map
- typedef boost::dynamic_bitset<> TBitSet;
- TBitSet gcref(hdr->target_len[refIndex], false);
- for(uint32_t i = 0; i < hdr->target_len[refIndex]; ++i) {
- if ((ref[i] == 'c') || (ref[i] == 'C') || (ref[i] == 'g') || (ref[i] == 'G')) gcref[i] = 1;
- }
- // Sum across fragment
- int32_t halfwin = (int32_t) (c.meanisize / 2);
- int32_t gcsum = 0;
- for(int32_t pos = halfwin; pos < (int32_t) hdr->target_len[refIndex] - halfwin; ++pos) {
- if (pos == halfwin) {
- for(int32_t i = pos - halfwin; i<=pos+halfwin; ++i) gcsum += gcref[i];
- } else {
- gcsum -= gcref[pos - halfwin - 1];
- gcsum += gcref[pos + halfwin];
- }
- gcContent[pos] = gcsum;
- }
- }
- // Broad window for replication wave
- std::vector<float> tileFac;
- if ((!regcorr.empty()) && (regWin > 0)) {
- uint32_t ntile = hdr->target_len[refIndex] / regWin + 1;
- tileFac.resize(ntile, 1.0);
- for(uint32_t t = 0; t < ntile; ++t) {
- uint32_t s = t * regWin;
- uint32_t e = std::min((uint32_t) hdr->target_len[refIndex], s + regWin);
- double gcnum = 0;
- uint32_t winlen = 0;
- for(uint32_t pos = s; pos < e; ++pos) {
- if ((gcContent[pos] > gcbound.first) && (gcContent[pos] < gcbound.second)) { gcnum += gcContent[pos]; ++winlen; }
- }
- if (winlen > 0) tileFac[t] = (float) regCorrFactor(regcorr, (gcnum / (double) winlen) / (double) c.meanisize);
- }
- }
- // Coverage track
- typedef uint16_t TCount;
- uint32_t maxCoverage = std::numeric_limits<TCount>::max();
- typedef std::vector<TCount> TCoverage;
- TCoverage cov(hdr->target_len[refIndex], 0);
- TCoverage covUniq(hdr->target_len[refIndex], 0);
- TCoverage covAll(hdr->target_len[refIndex], 0);
- TCoverage covTot;
- if (!c.basecov) covTot.resize(hdr->target_len[refIndex], 0);
- TCoverage& covMap = (!c.basecov) ? covTot : cov;
- // Split-read breakpoints
- std::vector<int32_t> clips;
- {
- // Mate map
- typedef boost::unordered_map<std::size_t, bool> TMateMap;
- TMateMap mateMap;
- // Count reads
- hts_itr_t* iter = sam_itr_queryi(idx, refIndex, 0, hdr->target_len[refIndex]);
- bam1_t* rec = bam_init1();
- int32_t lastAlignedPos = 0;
- std::set<std::size_t> lastAlignedPosReads;
- while (sam_itr_next(samfile, iter, rec) >= 0) {
- if (rec->core.flag & (BAM_FQCFAIL | BAM_FDUP | BAM_FUNMAP | BAM_FSECONDARY | BAM_FSUPPLEMENTARY)) continue;
- if ((rec->core.flag & BAM_FPAIRED) && ((rec->core.flag & BAM_FMUNMAP) || (rec->core.tid != rec->core.mtid))) continue;
- // Base coverage
- addBaseCoverage3(rec, covAll, covMap, covUniq, c.minQual, c.mapqUniq, hdr->target_len[refIndex], maxCoverage);
- // Filter by quality
- if (rec->core.qual < c.minQual) continue;
- // Collect split-read breakpoints
- if (rec->core.qual >= c.mapqUniq) addSplitReadBreakpoints(rec, c.minClip, c.minRefSep, hdr->target_len[refIndex], clips);
- // Base-level counting done
- if (c.basecov) continue;
- // Fragment coverage
- int32_t midPoint = rec->core.pos + halfAlignmentLength(rec);
- if (rec->core.flag & BAM_FPAIRED) {
- std::size_t seed = hash_sr(rec);
- // Clean-up the read store for identical alignment positions
- if (rec->core.pos > lastAlignedPos) {
- lastAlignedPosReads.clear();
- lastAlignedPos = rec->core.pos;
- }
- if (_firstPairObs(rec, lastAlignedPosReads)) {
- // First read
- lastAlignedPosReads.insert(seed);
- std::size_t hv = hash_pair(rec);
- mateMap[hv] = true;
- continue;
- } else {
- // Second read
- std::size_t hv = hash_pair_mate(rec);
- auto itMM = mateMap.find(hv);
- if ((itMM == mateMap.end()) || (!itMM->second)) continue; // Mate discarded
- mateMap.erase(itMM);
- }
- // update midpoint
- int32_t isize = (rec->core.pos + alignmentLength(rec)) - rec->core.mpos;
- if ((li.minNormalISize < isize) && (isize < li.maxNormalISize)) midPoint = rec->core.mpos + (int32_t) (isize/2);
- }
- // Count fragment
- if ((midPoint >= 0) && (midPoint < (int32_t) hdr->target_len[refIndex]) && (cov[midPoint] < maxCoverage - 1)) ++cov[midPoint];
- }
- // Clean-up
- bam_destroy1(rec);
- hts_itr_destroy(iter);
- }
- // Callable positions: unique if mostly high-MAPQ reads
- for(uint32_t pos = 0; pos < hdr->target_len[refIndex]; ++pos) {
- bool u;
- if (covMap[pos] == 0) u = ((ref[pos] != 'N') && (ref[pos] != 'n'));
- else u = (2 * (uint32_t) covUniq[pos] >= (uint32_t) covMap[pos]);
- uniqContent[pos] = (u ? (uint16_t) c.meanisize : 0);
- }
- // hom-del or unmappable?
- uint32_t maxHomDel = 1000000;
- uint32_t rstart = 0;
- while (rstart < hdr->target_len[refIndex]) {
- if (covMap[rstart] == 0) {
- uint32_t rend = rstart;
- while ((rend < hdr->target_len[refIndex]) && (covMap[rend] == 0)) ++rend;
- bool leftOK = (rstart > 0) && (uniqContent[rstart - 1] > 0);
- bool rightOK = (rend < hdr->target_len[refIndex]) && (uniqContent[rend] > 0);
- if ((!leftOK) || (!rightOK) || (rend - rstart > maxHomDel)) {
- for(uint32_t k = rstart; k < rend; ++k) uniqContent[k] = 0;
- }
- rstart = rend;
- } else ++rstart;
- }
- // Genome-wide read-depth windows
- DepthTrack const dt = uniqueTrack();
- std::vector<CovWin> wins;
- if (c.adaptive) {
- double covsum = 0;
- double expraw = 0;
- double expcor = 0;
- double ucov = 0;
- double tcov = 0;
- double aall = 0;
- double eall = 0;
- uint32_t winlen = 0;
- uint32_t start = 0;
- for(uint32_t pos = 0; pos < hdr->target_len[refIndex]; ++pos) {
- ucov += covUniq[pos];
- tcov += covMap[pos];
- if ((gcContent[pos] > gcbound.first) && (gcContent[pos] < gcbound.second)) {
- aall += covAll[pos];
- eall += gcbias[gcContent[pos]].coverageTotal;
- }
- if (_posUsed(dt, c, gcContent, uniqContent, gcbound, pos)) {
- double e1 = _expCov(dt, gcbias, gcContent[pos]);
- covsum += cov[pos];
- expraw += e1;
- expcor += e1 * (tileFac.empty() ? 1.0 : (double) tileFac[pos / regWin]);
- ++winlen;
- if (expraw >= c.targetExpCov) {
- wins.push_back(CovWin(start, pos + 1, winlen, covsum, expcor, ucov, tcov, aall, eall, true));
- covsum = 0;
- expraw = 0;
- expcor = 0;
- ucov = 0;
- tcov = 0;
- aall = 0;
- eall = 0;
- winlen = 0;
- start = pos + 1;
- }
- }
- }
- } else {
- // Fixed windows (non-overlapping)
- for(uint32_t start = 0; start < hdr->target_len[refIndex]; start = start + c.window_offset) {
- if (start + c.window_size < hdr->target_len[refIndex]) {
- double covsum = 0;
- double expcov = 0;
- double ucov = 0;
- double tcov = 0;
- double aall = 0;
- double eall = 0;
- uint32_t winlen = 0;
- for(uint32_t pos = start; pos < start + c.window_size; ++pos) {
- ucov += covUniq[pos];
- tcov += covMap[pos];
- if ((gcContent[pos] > gcbound.first) && (gcContent[pos] < gcbound.second)) {
- aall += covAll[pos];
- eall += gcbias[gcContent[pos]].coverageTotal;
- }
- if (_posUsed(dt, c, gcContent, uniqContent, gcbound, pos)) {
- covsum += cov[pos];
- expcov += _expCov(dt, gcbias, gcContent[pos]) * (tileFac.empty() ? 1.0 : (double) tileFac[pos / regWin]);
- ++winlen;
- }
- }
- bool valid = (winlen >= c.fracWindow * c.window_size);
- wins.push_back(CovWin(start, start + c.window_size, winlen, covsum, expcov, ucov, tcov, aall, eall, valid));
- }
- }
- }
- // Separate true hom. dels from unmappable
- uint32_t nw = wins.size();
- std::vector<bool> naFlag(nw, false);
- std::vector<bool> suspect(nw, false);
- std::vector<bool> strong(nw, false);
- double lowFrac = 0.1; // close to CN0
- double flankFrac = 0.5; // clear no CN0
- uint32_t maxHomDelWin = 1000000; // max size for hom. DEL
- for(uint32_t i = 0; i < nw; ++i) {
- if ((!wins[i].valid) || (wins[i].expcov <= 0)) {
- naFlag[i] = true;
- continue;
- }
- double r = wins[i].covsum / wins[i].expcov;
- suspect[i] = (r < lowFrac);
- strong[i] = (r >= flankFrac);
- }
- for(uint32_t i = 0; i < nw; ) {
- if (naFlag[i] || (!suspect[i])) {
- ++i;
- continue;
- }
- uint32_t a = i;
- uint32_t b = i;
- while ((b + 1 < nw) && (!naFlag[b+1]) && suspect[b+1]) ++b;
- uint32_t runBp = wins[b].end - wins[a].start;
- bool leftStrong = (a > 0) && (!naFlag[a-1]) && strong[a-1];
- bool rightStrong = (b + 1 < nw) && (!naFlag[b+1]) && strong[b+1];
- bool keepDel = leftStrong && rightStrong && (runBp <= maxHomDelWin);
- if (!keepDel) {
- for(uint32_t k = a; k <= b; ++k) naFlag[k] = true;
- }
- i = b + 1;
- }
- // Total-depth deletion signal
- double delRatio = c.cnvDelRatio;
- // Flag non-unique windows
- bool uniqGate = c.basecov;
- if (uniqGate) {
- for(uint32_t i = 0; i < nw; ++i) {
- if (naFlag[i]) continue;
- double rTot = (wins[i].eall > 0) ? (wins[i].aall / wins[i].eall) : 1.0;
- if (rTot < delRatio) continue;
- if ((wins[i].tcov > 0) && (wins[i].ucov <= c.uniqueToTotalCovRatio * wins[i].tcov)) naFlag[i] = true;
- }
- }
- // Flag low callable windows
- if ((c.adaptive) && (nw > 4)) {
- std::vector<int32_t> spans(nw);
- for(uint32_t i = 0; i < nw; ++i) spans[i] = (int32_t) (wins[i].end - wins[i].start);
- std::vector<int32_t> tmp(spans);
- std::sort(tmp.begin(), tmp.end());
- int32_t medSpan = tmp[tmp.size() / 2];
- for(uint32_t i = 0; i < nw; ++i) tmp[i] = std::abs(spans[i] - medSpan);
- std::sort(tmp.begin(), tmp.end());
- int32_t madSpan = tmp[tmp.size() / 2];
- int32_t maxSpan = std::max(medSpan + (int32_t) c.mad * madSpan, 2 * medSpan);
- for(uint32_t i = 0; i < nw; ++i) {
- if (spans[i] > maxSpan) {
- double rTot = (wins[i].eall > 0) ? (wins[i].aall / wins[i].eall) : 1.0;
- if (rTot >= delRatio) naFlag[i] = true;
- }
- }
- }
- // Exclude NA windows
- std::vector<std::pair<int32_t, int32_t> > naiv;
- for(uint32_t i = 0; i < nw; ++i) {
- if (naFlag[i]) {
- for(uint32_t k = wins[i].start; k < wins[i].end; ++k) uniqContent[k] = 0;
- int32_t nas = (int32_t) wins[i].start;
- int32_t nae = (int32_t) wins[i].end;
- if ((!naiv.empty()) && (naiv.back().second >= nas)) naiv.back().second = std::max(naiv.back().second, nae);
- else naiv.push_back(std::make_pair(nas, nae));
- }
- }
- // Split-read breakpoints
- std::vector<SVBreakpoint> chrbp;
- collectBreakpoints(c, gcbound, gcContent, uniqContent, gcbias, cov, hdr, refIndex, clips, chrbp);
- // CNV discovery
- if (!c.hasGenoFile) segmentRD(c, gcbound, gcContent, uniqContent, gcbias, tileFac, regWin, cov, hdr, refIndex, chrbp, uniqueTrack(), naiv, cnvs);
- // CNV genotyping
- genotypeCNVs(c, gcbound, gcContent, uniqContent, gcbias, tileFac, regWin, cov, covUniq, covMap, covAll, ref, hdr, refIndex, cnvs);
- if (ref != NULL) free(ref);
- // Write windows
- if (!c.covfile.empty()) {
- std::string chrn(hdr->target_name[refIndex]);
- for(uint32_t i = 0; i < nw; ++i) {
- double uniqFrac;
- if (uniqGate) uniqFrac = (wins[i].tcov > 0) ? (wins[i].ucov / wins[i].tcov) : -1.0;
- else uniqFrac = (wins[i].end > wins[i].start) ? ((double) wins[i].winlen / (double) (wins[i].end - wins[i].start)) : -1.0;
- if (naFlag[i]) {
- dataOut << chrn << "\t" << wins[i].start << "\t" << wins[i].end << "\t" << uniqFrac << "\tNA\tNA" << std::endl;
- } else {
- double cn = chrPloidy;
- double logR = 0;
- if (wins[i].expcov > 0) {
- cn = (c.expectedCN * wins[i].covsum / wins[i].expcov - chrCtrlPloidy * (1 - c.purity)) / c.purity;
- logR = std::log2((wins[i].covsum + 1.0) / (wins[i].expcov + 1.0));
- }
- dataOut << chrn << "\t" << wins[i].start << "\t" << wins[i].end << "\t" << uniqFrac << "\t" << logR << "\t" << cn << std::endl;
- }
- }
- }
- }
- // Sort CNVs
- sort(cnvs.begin(), cnvs.end());
- // Merge CNVs
- if (!c.hasGenoFile) mergeAdjacentSameCN(cnvs, c.cnMergeTol);
- // Exclude regions
- if (c.hasExcludeFile) {
- typedef boost::icl::interval_set<uint32_t> TChrIntervals;
- typedef TChrIntervals::interval_type TIVal;
- std::vector<TChrIntervals> validRegions;
- if (_parseExcludeIntervals(c, hdr, validRegions)) {
- std::vector<CNV> keep;
- keep.reserve(cnvs.size());
- for(uint32_t i = 0; i < cnvs.size(); ++i) {
- int32_t s = cnvs[i].start;
- int32_t e = cnvs[i].end;
- if ((e <= s) || (cnvs[i].chr < 0) || (cnvs[i].chr >= (int32_t) validRegions.size())) { keep.push_back(cnvs[i]); continue; }
- TChrIntervals ci;
- ci.insert(TIVal::right_open((uint32_t) s, (uint32_t) e));
- TChrIntervals inter = ci & validRegions[cnvs[i].chr];
- uint32_t validbp = 0;
- for(TChrIntervals::const_iterator it = inter.begin(); it != inter.end(); ++it) validbp += (it->upper() - it->lower());
- if ((double) validbp >= 0.5 * (double) (e - s)) keep.push_back(cnvs[i]);
- }
- cnvs.swap(keep);
- }
- }
- // Genotype CNVs
- cnvVCF(c, cnvs);
- // clean-up
- fai_destroy(faiRef);
- bam_hdr_destroy(hdr);
- hts_idx_destroy(idx);
- sam_close(samfile);
- if (!c.covfile.empty()) {
- dataOut.pop();
- dataOut.pop();
- }
- return 0;
- }
- int coral(int argc, char **argv) {
- CountDNAConfig c;
- std::string haploidChr;
- std::string mode;
- // Parameter
- boost::program_options::options_description generic("Generic options");
- generic.add_options()
- ("help,?", "show help message")
- ("genome,g", boost::program_options::value<boost::filesystem::path>(&c.genome), "genome file")
- ("exclude,x", boost::program_options::value<boost::filesystem::path>(&c.exclude), "file with regions to exclude")
- ("quality,q", boost::program_options::value<uint16_t>(&c.minQual)->default_value(10), "min. mapping quality")
- ("outfile,o", boost::program_options::value<boost::filesystem::path>(&c.outfile), "BCF output file")
- ("covfile,c", boost::program_options::value<boost::filesystem::path>(&c.covfile), "gzipped coverage file")
- ("segmentation,u", boost::program_options::value<boost::filesystem::path>(&c.segfile), "segmentation BED output file")
- ;
- boost::program_options::options_description cnv("CNV calling");
- cnv.add_options()
- ("cnv-size,z", boost::program_options::value<uint32_t>(&c.minCnvSize)->default_value(1000), "min. CNV size")
- ("vcffile,v", boost::program_options::value<boost::filesystem::path>(&c.genofile), "input CNV BCF file for re-genotyping")
- ("minclip", boost::program_options::value<uint32_t>(&c.minClip)->default_value(25), "min. clipping length")
- ("minrefsep", boost::program_options::value<uint32_t>(&c.minRefSep)->default_value(30), "min. reference separation")
- ("min-bp-support", boost::program_options::value<uint32_t>(&c.minBpSupport)->default_value(3), "min. split-read support")
- ("penalty", boost::program_options::value<float>(&c.penalty)->default_value(3), "segmentation penalty")
- ("cnv-merge", boost::program_options::value<float>(&c.cnMergeTol)->default_value(0.25), "min. log2 ratio to separate CNVs")
- ("cnv-qual", boost::program_options::value<int32_t>(&c.cnvMinQual)->default_value(5), "min. quality for PASS")
- ;
- boost::program_options::options_description cancer("Ploidy/purity correction");
- cancer.add_options()
- ("mode,m", boost::program_options::value<std::string>(&mode)->default_value("germline"), "CNV calling mode [germline|somatic]")
- ("ploidy,y", boost::program_options::value<float>(&c.ploidy)->default_value(2), "sample ploidy")
- ("purity,p", boost::program_options::value<float>(&c.purity)->default_value(1), "sample purity [0.1, 1]")
- ("ctrl-ploidy", boost::program_options::value<float>(&c.ctrlPloidy)->default_value(2), "control ploidy")
- ("haploid-chr", boost::program_options::value<std::string>(&haploidChr), "haploid chromosomes, e.g. chrX,chrY")
- ;
- boost::program_options::options_description window("Read-depth windows");
- window.add_options()
- ("window,w", boost::program_options::value<uint32_t>(&c.window_size)->default_value(0), "window size in bp (0: automatic)")
- ("fraction-unique", boost::program_options::value<float>(&c.uniqueToTotalCovRatio)->default_value(0.8), "uniqueness filter [0,1]")
- ("cnv-del-confirm", boost::program_options::value<float>(&c.cnvDelConfirm)->default_value(1.5), "max. total-depth CN for DEL")
- ("cnv-del-ratio", boost::program_options::value<float>(&c.cnvDelRatio)->default_value(0.75), "total-depth ratio")
- ("basecov", "force base-level counting")
- ("fragmentcov", "force fragment-level counting")
- ("no-regional-gc", "disable broad GC correction")
- ;
- boost::program_options::options_description hidden("Hidden options");
- hidden.add_options()
- ("input-file", boost::program_options::value<boost::filesystem::path>(&c.bamFile), "input BAM/CRAM file")
- ("fragment", boost::program_options::value<float>(&c.fragmentUnique)->default_value(0.97), "min. fragment uniqueness [0,1]")
- ("statsfile", boost::program_options::value<boost::filesystem::path>(&c.statsFile), "gzipped stats output file (optional)")
- ("window-offset", boost::program_options::value<uint32_t>(&c.window_offset)->default_value(0), "window offset (0: window size)")
- ("fraction-window", boost::program_options::value<float>(&c.fracWindow)->default_value(0.25), "min. callable window fraction [0,1]")
- ("mapq-uniq", boost::program_options::value<uint16_t>(&c.mapqUniq)->default_value(20), "min. MAPQ for a uniquely-placed read")
- ("target-reads", boost::program_options::value<uint32_t>(&c.targetReads)->default_value(150), "target reads/window")
- ("min-windows", boost::program_options::value<uint32_t>(&c.minSegWin)->default_value(2), "min. windows per segment")
- ("scan-window", boost::program_options::value<uint32_t>(&c.scanWindow)->default_value(10000), "GC scanning window size")
- ("scan-regions", boost::program_options::value<boost::filesystem::path>(&c.scanFile), "GC scanning regions in BED format")
- ("mad-cutoff", boost::program_options::value<uint16_t>(&c.mad)->default_value(3), "median + 3 * mad count cutoff")
- ("percentile", boost::program_options::value<float>(&c.exclgc)->default_value(0.0005), "excl. extreme GC fraction")
- ("no-window-selection", "no scan window selection")
- ;
- boost::program_options::positional_options_description pos_args;
- pos_args.add("input-file", -1);
- // Set the visibility
- boost::program_options::options_description cmdline_options;
- cmdline_options.add(generic).add(cnv).add(cancer).add(window).add(hidden);
- boost::program_options::options_description visible_options;
- visible_options.add(generic).add(cnv).add(cancer).add(window);
- // Parse command-line
- boost::program_options::variables_map vm;
- boost::program_options::store(boost::program_options::command_line_parser(argc, argv).options(cmdline_options).positional(pos_args).run(), vm);
- boost::program_options::notify(vm);
- // Check command line arguments
- if ((vm.count("help")) || (!vm.count("input-file")) || (!vm.count("genome"))) {
- std::cerr << std::endl;
- std::cerr << "Usage: delly " << argv[0] << " [OPTIONS] -g <genome.fa> <aligned.bam>" << std::endl;
- std::cerr << visible_options << "\n";
- return 1;
- }
- // Show cmd
- boost::posix_time::ptime now = boost::posix_time::second_clock::local_time();
- std::cerr << '[' << boost::posix_time::to_simple_string(now) << "] ";
- std::cerr << "delly ";
- for(int i=0; i<argc; ++i) { std::cerr << argv[i] << ' '; }
- std::cerr << std::endl;
- // Stats file
- if (vm.count("statsfile")) c.hasStatsFile = true;
- else c.hasStatsFile = false;
- // Scan regions
- if (vm.count("scan-regions")) c.hasScanFile = true;
- else c.hasScanFile = false;
- // Exclude regions
- if (vm.count("exclude")) c.hasExcludeFile = true;
- else c.hasExcludeFile = false;
- // Scan window selection
- if (vm.count("no-window-selection")) c.noScanWindowSelection = true;
- else c.noScanWindowSelection = false;
- c.regionalGc = (!vm.count("no-regional-gc"));
- // Adaptive windows
- c.adaptive = false;
- c.targetExpCov = 0;
- if (c.window_size == 0) c.adaptive = true;
- if (c.targetReads == 0) c.targetReads = 150;
- // Calling mode
- c.somatic = (mode == "somatic");
- if (c.somatic) {
- // More sensitive segmentation
- if (vm["penalty"].defaulted()) c.penalty = 1.25;
- if (vm["cnv-merge"].defaulted()) c.cnMergeTol = 0;
- }
- // Purity
- if (c.purity > 1) c.purity = 1;
- if (c.purity < 0.1) c.purity = 0.1;
- // Expected ploidy
- c.expectedCN = c.purity * c.ploidy + (1.0 - c.purity) * c.ctrlPloidy;
- // Segmentation
- if (vm.count("segmentation")) c.hasSegFile = true;
- else c.hasSegFile = false;
- // Window offset
- if ((c.window_offset == 0) || (c.window_offset > c.window_size)) c.window_offset = c.window_size;
- // Check input VCF file (CNV genotyping)
- if (vm.count("vcffile")) {
- if (!(boost::filesystem::exists(c.genofile) && boost::filesystem::is_regular_file(c.genofile) && boost::filesystem::file_size(c.genofile))) {
- std::cerr << "Input VCF/BCF file is missing: " << c.genofile.string() << std::endl;
- return 1;
- }
- htsFile* ifile = bcf_open(c.genofile.string().c_str(), "r");
- if (ifile == NULL) {
- std::cerr << "Fail to open file " << c.genofile.string() << std::endl;
- return 1;
- }
- bcf_hdr_t* hdr = bcf_hdr_read(ifile);
- if (hdr == NULL) {
- std::cerr << "Fail to open index file " << c.genofile.string() << std::endl;
- return 1;
- }
- bcf_hdr_destroy(hdr);
- bcf_close(ifile);
- c.hasGenoFile = true;
- } else c.hasGenoFile = false;
- // Check outfile
- if (!vm.count("outfile")) c.outfile = "-";
- else {
- if (c.outfile.string() != "-") {
- if (!_outfileValid(c.outfile)) return 1;
- }
- }
- // Check bam file
- LibraryInfo li;
- bool pairedLib = false;
- if (!(boost::filesystem::exists(c.bamFile) && boost::filesystem::is_regular_file(c.bamFile) && boost::filesystem::file_size(c.bamFile))) {
- std::cerr << "Alignment file is missing: " << c.bamFile.string() << std::endl;
- return 1;
- } else {
- // Get scan regions
- typedef boost::icl::interval_set<uint32_t> TChrIntervals;
- typedef typename TChrIntervals::interval_type TIVal;
- typedef std::vector<TChrIntervals> TRegionsGenome;
- TRegionsGenome scanRegions;
- // Open BAM file
- samFile* samfile = sam_open(c.bamFile.string().c_str(), "r");
- if (samfile == NULL) {
- std::cerr << "Fail to open file " << c.bamFile.string() << std::endl;
- return 1;
- }
- hts_idx_t* idx = sam_index_load(samfile, c.bamFile.string().c_str());
- if (idx == NULL) {
- if (bam_index_build(c.bamFile.string().c_str(), 0) != 0) {
- std::cerr << "Fail to open index for " << c.bamFile.string() << std::endl;
- return 1;
- }
- }
- bam_hdr_t* hdr = sam_hdr_read(samfile);
- if (hdr == NULL) {
- std::cerr << "Fail to open header for " << c.bamFile.string() << std::endl;
- return 1;
- }
- c.nchr = hdr->n_targets;
- c.minChrLen = setMinChrLen(hdr, 0.95);
- std::string sampleName = "unknown";
- getSMTag(std::string(hdr->text), c.bamFile.stem().string(), sampleName);
- c.sampleName = sampleName;
- // Check special chromosomes
- if (haploidChr.size()) _selectedRefIndices(c, hdr, haploidChr);
- // Check matching chromosome names
- faidx_t* faiRef = fai_load(c.genome.string().c_str());
- uint32_t refFound = 0;
- for(int32_t refIndex=0; refIndex < hdr->n_targets; ++refIndex) {
- std::string tname(hdr->target_name[refIndex]);
- if (faidx_has_seq(faiRef, tname.c_str())) ++refFound;
- else {
- std::cerr << "Warning: BAM chromosome " << tname << " not present in reference genome!" << std::endl;
- }
- }
- fai_destroy(faiRef);
- if (!refFound) {
- std::cerr << "Reference genome chromosome naming disagrees with BAM file!" << std::endl;
- return 1;
- }
- // Estimate library params
- if (c.hasScanFile) {
- if (!_parseBedIntervals(c.scanFile.string(), c.hasScanFile, hdr, scanRegions)) {
- std::cerr << "Warning: Couldn't parse BED intervals. Do the chromosome names match?" << std::endl;
- return 1;
- }
- } else {
- scanRegions.resize(hdr->n_targets);
- for (int32_t refIndex = 0; refIndex < hdr->n_targets; ++refIndex) {
- scanRegions[refIndex].insert(TIVal::right_open(0, hdr->target_len[refIndex]));
- }
- }
- typedef std::vector<LibraryInfo> TSampleLibrary;
- TSampleLibrary sampleLib(1, LibraryInfo());
- CountDNAConfigLib dellyConf;
- dellyConf.genome = c.genome;
- dellyConf.files.push_back(c.bamFile);
- dellyConf.madCutoff = 9;
- dellyConf.madNormalCutoff = c.mad;
- getLibraryParams(dellyConf, scanRegions, sampleLib);
- li = sampleLib[0];
- pairedLib = (li.median > 0);
- if (!li.median) {
- li.median = 250;
- li.mad = 15;
- li.minNormalISize = 0;
- li.maxNormalISize = 400;
- }
- c.meanisize = ((int32_t) (li.median / 2)) * 2 + 1;
- // Coverage-aware GC scan window
- bool autoScanWin = (!vm.count("scan-window")) || (vm["scan-window"].defaulted());
- if ((autoScanWin) && (idx != NULL)) {
- uint64_t totalMapped = 0;
- uint64_t genomeLen = 0;
- for(int32_t refIndex = 0; refIndex < hdr->n_targets; ++refIndex) {
- uint64_t mapped = 0;
- uint64_t unmapped = 0;
- hts_idx_get_stat(idx, refIndex, &mapped, &unmapped);
- if (mapped > 0) {
- totalMapped += mapped;
- genomeLen += hdr->target_len[refIndex];
- }
- }
- if (pairedLib) totalMapped /= 2; // fragments
- if ((totalMapped > 0) && (genomeLen > 0)) {
- double fragPerBp = (double) totalMapped / (double) genomeLen;
- uint32_t targetScanReads = 30;
- uint32_t autoScan = (uint32_t) (targetScanReads / fragPerBp);
- if (autoScan < c.scanWindow) autoScan = c.scanWindow;
- uint32_t maxScan = 1000000;
- if (autoScan > maxScan) autoScan = maxScan;
- if (autoScan > c.scanWindow) c.scanWindow = autoScan;
- }
- }
- // Clean-up
- bam_hdr_destroy(hdr);
- hts_idx_destroy(idx);
- sam_close(samfile);
- }
- // Counting model
- if (vm.count("basecov")) c.basecov = true;
- else if (vm.count("fragmentcov")) c.basecov = false;
- else c.basecov = ((!pairedLib) && (li.rs >= 500));
- if ((c.basecov) && (!vm.count("cnv-del-confirm"))) c.cnvDelConfirm = std::numeric_limits<float>::max();
- // GC bias estimation
- typedef std::pair<uint32_t, uint32_t> TGCBound;
- TGCBound gcbound;
- std::vector<GcBias> gcbias(c.meanisize + 1, GcBias());
- typedef std::vector<ScanWindow> TWindowCounts;
- typedef std::vector<TWindowCounts> TGenomicWindowCounts;
- TGenomicWindowCounts scanCounts(c.nchr, TWindowCounts());
- {
- // Scan genomic windows
- scan(c, li, scanCounts);
- // Check coverage
- {
- std::vector<uint32_t> sampleScanVec;
- for(uint32_t i = 0; i < scanCounts.size(); ++i) {
- for(uint32_t j = 0; j < scanCounts[i].size(); ++j) {
- sampleScanVec.push_back(scanCounts[i][j].cov);
- }
- if (sampleScanVec.size() > 1000000) break;
- }
- std::sort(sampleScanVec.begin(), sampleScanVec.end());
- if (sampleScanVec.empty()) {
- std::cerr << "Not enough windows!" << std::endl;
- return 1;
- }
- if (sampleScanVec[sampleScanVec.size()/2] < 5) {
- std::cerr << "Coverage in the GC scan window is too low." << std::endl;
- return 1;
- }
- }
- // Select stable windows
- selectWindows(c, scanCounts);
- // Estimate GC bias
- gcBias(c, scanCounts, li, gcbias, gcbound);
- // Statistics output
- if (c.hasStatsFile) {
- // Open stats file
- boost::iostreams::filtering_ostream statsOut;
- statsOut.push(boost::iostreams::gzip_compressor());
- statsOut.push(boost::iostreams::file_sink(c.statsFile.string(), std::ios_base::out | std::ios_base::binary));
- // Library Info
- statsOut << "LP\t" << li.rs << ',' << li.median << ',' << li.mad << ',' << li.minNormalISize << ',' << li.maxNormalISize << std::endl;
- // Scan window summry
- samFile* samfile = sam_open(c.bamFile.string().c_str(), "r");
- bam_hdr_t* hdr = sam_hdr_read(samfile);
- statsOut << "SW\tchrom\tstart\tend\tselected\tcoverage\tuniqcov" << std::endl;
- for(uint32_t refIndex = 0; refIndex < (uint32_t) hdr->n_targets; ++refIndex) {
- for(uint32_t i = 0; i < scanCounts[refIndex].size(); ++i) {
- statsOut << "SW\t" << hdr->target_name[refIndex] << '\t' << scanCounts[refIndex][i].start << '\t' << scanCounts[refIndex][i].end << '\t' << scanCounts[refIndex][i].select << '\t' << scanCounts[refIndex][i].cov << '\t' << scanCounts[refIndex][i].uniqcov << std::endl;
- }
- }
- bam_hdr_destroy(hdr);
- sam_close(samfile);
- // GC bias summary
- statsOut << "GC\tgcsum\tsample\treference\tpercentileSample\tpercentileReference\tfractionSample\tfractionReference\tobsexp\tmeancoverage\tmeancoverageTotal\treferenceTotal" << std::endl;
- for(uint32_t i = 0; i < gcbias.size(); ++i) statsOut << "GC\t" << i << "\t" << gcbias[i].sample << "\t" << gcbias[i].reference << "\t" << gcbias[i].percentileSample << "\t" << gcbias[i].percentileReference << "\t" << gcbias[i].fractionSample << "\t" << gcbias[i].fractionReference << "\t" << gcbias[i].obsexp << "\t" << gcbias[i].coverage << "\t" << gcbias[i].coverageTotal << "\t" << gcbias[i].referenceTotal << std::endl;
- statsOut << "BoundsGC\t" << gcbound.first << "," << gcbound.second << std::endl;
- statsOut.pop();
- statsOut.pop();
- }
- }
- // Coverage-aware window size
- uint32_t effWin = (c.window_size > 0) ? c.window_size : 50000;
- if (c.adaptive) {
- // Mean expected coverage at CN2 over the callable GC range
- double covMean = 0;
- uint64_t refCnt = 0;
- for(uint32_t i = gcbound.first + 1; i < gcbound.second; ++i) {
- covMean += gcbias[i].coverage * (double) gcbias[i].reference;
- refCnt += gcbias[i].reference;
- }
- if (refCnt) covMean /= (double) refCnt;
- if (covMean <= 0) {
- // Fixed windows
- c.adaptive = false;
- c.window_size = 10000;
- c.window_offset = c.window_size;
- } else {
- double readLen = (li.rs > 0) ? (double) li.rs : (double) c.meanisize;
- double molPerBp = c.basecov ? (covMean / readLen) : covMean;
- if (molPerBp <= 0) molPerBp = 1e-9;
- double winBp = (double) c.targetReads / molPerBp;
- double minWin = std::max(100.0, 4.0 * readLen);
- double maxWin = 2000000.0;
- if (winBp < minWin) winBp = minWin;
- if (winBp > maxWin) winBp = maxWin;
- c.targetExpCov = covMean * winBp;
- effWin = (uint32_t) winBp;
- double effReads = molPerBp * winBp;
- double covDepth = c.basecov ? covMean : (covMean * readLen);
- now = boost::posix_time::second_clock::local_time();
- std::cerr << '[' << boost::posix_time::to_simple_string(now) << "] " << "Auto window size: " << (uint32_t) winBp << " bp, " << (uint32_t) effReads << " reads/window (" << (c.basecov ? "base-level" : "fragment") << ", coverage " << std::fixed << std::setprecision(2) << covDepth << std::defaultfloat << "x)" << std::endl;
- }
- }
- // Regional GC correction
- std::vector<double> regcorr;
- uint32_t regWin = std::max((uint32_t) 50000, effWin);
- if (c.regionalGc) estimateRegionalGc(c, gcbound, gcbias, scanCounts, regWin, regcorr);
- // Count reads
- if (bamCount(c, li, gcbias, gcbound, regcorr, regWin)) {
- std::cerr << "Read counting error!" << std::endl;
- return 1;
- }
- // Done
- now = boost::posix_time::second_clock::local_time();
- std::cerr << '[' << boost::posix_time::to_simple_string(now) << "] " << "Done." << std::endl;
- return 0;
- }
- }
- #endif
coral.h at commit c6474a2, under BSD-3-Clause · at the source
Overview
- Brain Mosaicism and Tumorigenesis Laboratory, German Cancer Research Center, Heidelberg, Germany
- Faculty of Bioscience, Ruprecht-Karl-University of Heidelberg, Heidelberg, Germany
- European Molecular Biology Laboratory (EMBL), Genome Biology Unit, Heidelberg, Germany
- Data Science Centre, European Molecular Biology Laboratory (EMBL), Heidelberg, Germany
- Division of Theoretical Systems Biology, German Cancer Research Center, Heidelberg, Germany
- Faculty of Medicine, Ruprecht-Karl-University of Heidelberg, Heidelberg, Germany
- Bridging Research Division on Mechanisms of Genomic Variation and Data Science, German Cancer Research Center, Heidelberg, Germany
Abstract
Copy number variants (CNVs) are strongly implicated in neurological and psychiatric disorders and brain cancer, yet the process by which replication stress generates CNVs—and why some recur while others remain rare—remains poorly understood. Here, we show that recurrent DNA-break clusters (RDCs) act as common initiating lesions that drive both recurrent and non-recurrent CNVs. In murine neural progenitor cells subjected to chemically induced replication stress, bulk whole-genome sequencing identifies recurrent CNVs enriched at late-replicating RDCs within actively transcribed genes. Single-cell genome sequencing further uncovers frequent, non-recurrent CNVs associated with RDCs that arise during the transition from early to late DNA replication. These CNVs represent stable, heritable structural variants with breakpoints consistently enriched at RDCs. CRISPR/
Reproduced under the paper's license (CC BY), from the paper cited above.
Repositories
Its files are read in the Code ↔ Paper reader above, with 3 matches between paragraphs and lines of code.
DKFZ-ODCF/AlignmentAndQCWorkflows
2701b22ad5391ce53747345fff80b87f521cc60f, 9 May 2023Availability: 1 check, the latest on 29 September 2026: the link answers
- 29 September 2026: the link answers
99 files
- ConfigSummary.py, Python, 369 lines
- resources/
analysisTools/ , Python, 189 linesbisulfiteWorkflow/ WGBS_read_pair_reconstru ction.py - resources/
analysisTools/ , Python, 1 linebisulfiteWorkflow/ __init__.py - resources/
analysisTools/ , Python, 386 linesbisulfiteWorkflow/ bcall_2012Nov27.py - resources/
analysisTools/ , Python, 431 linesbisulfiteWorkflow/ bcall_tagmentation_2012N ov25.py - resources/
analysisTools/ , Python, 304 linesbisulfiteWorkflow/ bconv.py - resources/
analysisTools/ , Python, 104 linesbisulfiteWorkflow/ bfq.py - resources/
analysisTools/ , Python, 157 linesbisulfiteWorkflow/ bsnp.py - resources/
analysisTools/ , Python, 124 linesbisulfiteWorkflow/ bspike.py - resources/
analysisTools/ , Shell, 274 linesbisulfiteWorkflow/ bwaMemSortSlimWithReadCo nversionForBisulfiteData .sh - resources/
analysisTools/ , Shell, 118 linesbisulfiteWorkflow/ convert_meth_calls_moabs .sh - resources/
analysisTools/ , Python, 102 linesbisulfiteWorkflow/ faconv.py - resources/
analysisTools/ , Python, 151 linesbisulfiteWorkflow/ fapos.py - resources/
analysisTools/ , Python, 165 linesbisulfiteWorkflow/ fqconv.py - resources/
analysisTools/ , Python, 106 linesbisulfiteWorkflow/ fqsplit.py - resources/
analysisTools/ , Shell, 112 linesbisulfiteWorkflow/ methylCtools_methylation _calling.sh - resources/
analysisTools/ , Shell, 117 linesbisulfiteWorkflow/ methylCtools_methylation _calling_meta.sh - resources/
analysisTools/ , Shell, 14 linesexomePipeline/ createOnTargetCoveragePl ot.sh - resources/
analysisTools/ , Shell, 18 linesexomePipeline/ samtoolsOnTargetCoverage .sh - resources/
analysisTools/ , R, 25 linesexomePipeline/ targetCoverageDistributi onPlot.r - resources/
analysisTools/ , Shell, 115 linesexomePipeline/ targetExtractCoverageSli m.sh - resources/
analysisTools/ , Perl, 110 linesexomePipeline/ targetcov.pl - resources/
analysisTools/ , Perl, 81 linesqcPipeline/ Illumina2PhredScore.pl - resources/
analysisTools/ , Perl, 37 linesqcPipeline/ PhredOrIllumina.pl - resources/
analysisTools/ , Python, 1 lineqcPipeline/ __init__.py - resources/
analysisTools/ , Python, 83 linesqcPipeline/ addMappability.py - resources/
analysisTools/ , Perl, 601 linesqcPipeline/ annotate_vcf.pl - resources/
analysisTools/ , Shell, 139 linesqcPipeline/ bashLib.sh - resources/
analysisTools/ , Python, 186 linesqcPipeline/ bsnp.py - resources/
analysisTools/ , Shell, 55 linesqcPipeline/ bwaCommonAlignmentSettin gs.sh - resources/
analysisTools/ , Shell, 276 linesqcPipeline/ bwaMemSortSlim.sh - resources/
analysisTools/ , Shell, 37 linesqcPipeline/ checkFastQC.sh - resources/
analysisTools/ , R, 43 linesqcPipeline/ chrom_diff.r - resources/
analysisTools/ , Shell, 5 linesqcPipeline/ cleanupScript.sh - resources/
analysisTools/ , Shell, 23 linesqcPipeline/ cnvMergeFilter.sh - resources/
analysisTools/ , Python, 54 linesqcPipeline/ convertTabToJson.py - resources/
analysisTools/ , R, 337 linesqcPipeline/ correctGCBias.R - resources/
analysisTools/ , R, 251 linesqcPipeline/ correctGCBias_functions. R - resources/
analysisTools/ , Shell, 54 linesqcPipeline/ correct_gc_bias.sh - resources/
analysisTools/ , R, 209 linesqcPipeline/ coveragePlot.R - resources/
analysisTools/ , Shell, 34 linesqcPipeline/ environments/ conda.sh - resources/
analysisTools/ , Shell, 160 linesqcPipeline/ environments/ tbi-lsf-cluster.sh - resources/
analysisTools/ , Shell, 51 linesqcPipeline/ environments/ tbi-pbs-cluster.sh - resources/
analysisTools/ , Perl, 55 linesqcPipeline/ failedReadFilter.pl - resources/
analysisTools/ , Perl, 87 linesqcPipeline/ fakeDupmarkMetrics.pl - resources/
analysisTools/ , Perl, 87 linesqcPipeline/ filter_readbins.pl - resources/
analysisTools/ , Perl, 343 linesqcPipeline/ flags_isizes_PEaberratio ns.pl - resources/
analysisTools/ , Python, 27 linesqcPipeline/ gccorrection_python_modu les/ Options.py - resources/
analysisTools/ , Python, 72 linesqcPipeline/ gccorrection_python_modu les/ Tabfile.py - resources/
analysisTools/ , Python, 1 lineqcPipeline/ gccorrection_python_modu les/ __init__.py - resources/
analysisTools/ , Shell, 19 linesqcPipeline/ genomeCoverage.sh - resources/
analysisTools/ , Shell, 28 linesqcPipeline/ genomeCoveragePlots.sh - resources/
analysisTools/ , R, 88 linesqcPipeline/ getSex.R - resources/
analysisTools/ , R, 74 linesqcPipeline/ getopt.R - resources/
analysisTools/ , Perl, 460 linesqcPipeline/ groupedGenomeCoverages.p l - resources/
analysisTools/ , R, 33 linesqcPipeline/ insertsizePlot.R - resources/
analysisTools/ , Shell, 397 linesqcPipeline/ mergeAndMarkOrRemoveDupl icatesSlim.sh - resources/
analysisTools/ , Python, 79 linesqcPipeline/ merge_and_filter_cnv.py - resources/
analysisTools/ , Perl, 73 linesqcPipeline/ missingReadGroups.pl - resources/
analysisTools/ , Shell, 17 linesqcPipeline/ picardCollectMetrics.sh - resources/
analysisTools/ , Python, 506 linesqcPipeline/ prepare-trimmomatic-adap ters.py - resources/
analysisTools/ , Perl, 640 linesqcPipeline/ qcJson.pl - resources/
analysisTools/ , R, 135 linesqcPipeline/ qq.R - resources/
analysisTools/ , Shell, 67 linesqcPipeline/ vcfAnno.sh - resources/
analysisTools/ , Shell, 339 linesqcPipeline/ workflowLib.sh - resources/
analysisTools/ , Perl, 309 linesqcPipeline/ writeQCsummary.pl - resources/
analysisTools/ , Shell, 89 linesqcPipelineTools/ compile_coverageQcD.sh - resources/
analysisTools/ , Shell, 89 linesqcPipelineTools/ compile_genomeCoverageD. sh - resources/
tests/ , Shell, 30 linesanalysisTools/ qcPipeline/ bashLib.sh - resources/
tests/ , Shell, 181 linesanalysisTools/ qcPipeline/ workflowLib.sh - resources/
tests/ , Shell, 20 linesrunTests.sh - revertAndUnstageCompiled
Files.sh , Shell, 11 lines - src/
de/ , Java, 24 linesdkfz/ b080/ co/ AlignmentAndQCWorkflowsP lugin.java - src/
de/ , Java, 18 linesdkfz/ b080/ co/ files/ AlignedSequenceFile.java - src/
de/ , Java, 51 linesdkfz/ b080/ co/ files/ AlignedSequenceFileGroup .java - src/
de/ , Java, 452 linesdkfz/ b080/ co/ files/ BamFile.java - src/
de/ , Java, 18 linesdkfz/ b080/ co/ files/ BamIndexFile.java - src/
de/ , Java, 15 linesdkfz/ b080/ co/ files/ BamMetricsAlignmentSumma ryFile.java - src/
de/ , Java, 19 linesdkfz/ b080/ co/ files/ BamMetricsCollectionFile Group.java - src/
de/ , Java, 18 linesdkfz/ b080/ co/ files/ BamMetricsFile.java - src/
de/ , Java, 53 linesdkfz/ b080/ co/ files/ ChromosomeDiffFileGroup. java - src/
de/ , Java, 17 linesdkfz/ b080/ co/ files/ ChromosomeDiffPlotFile.j ava - src/
de/ , Java, 20 linesdkfz/ b080/ co/ files/ ChromosomeDiffTextFile.j ava - src/
de/ , Java, 59 linesdkfz/ b080/ co/ files/ CoverageTextFile.java - src/
de/ , Java, 57 linesdkfz/ b080/ co/ files/ CoverageTextFileGroup.ja va - src/
de/ , Java, 21 linesdkfz/ b080/ co/ files/ FastqcFile.java - src/
de/ , Java, 22 linesdkfz/ b080/ co/ files/ FastqcGroup.java - src/
de/ , Java, 17 linesdkfz/ b080/ co/ files/ FlagstatsFile.java - src/
de/ , Java, 27 linesdkfz/ b080/ co/ files/ GenomeCoveragePlotFile.j ava - src/
de/ , Java, 20 linesdkfz/ b080/ co/ files/ GenomeCoveragePlotFileGr oup.java - src/
de/ , Java, 53 linesdkfz/ b080/ co/ files/ InsertSizesFileGroup.jav a - src/
de/ , Java, 18 linesdkfz/ b080/ co/ files/ InsertSizesPlotFile.java - src/
de/ , Java, 18 linesdkfz/ b080/ co/ files/ InsertSizesTextFile.java - src/
de/ , Java, 16 linesdkfz/ b080/ co/ files/ InsertSizesValueFile.jav a - src/
de/ , Java, 138 linesdkfz/ b080/ co/ files/ LaneFile.java - src/
de/ , Java, 122 linesdkfz/ b080/ co/ files/ LaneFileGroup.java - src/
de/ , Java, 38 linesdkfz/ b080/ co/ files/ QCSummaryFile.java - LICENSE, License, 1,054 lines
- README.md, Text, 20 lines
CL-CHEN-Lab/RepliSeq
359c213d846061eba0a688412f203dfb05369bf1, 6 September 2021Availability: 1 check, the latest on 29 September 2026: the link answers
- 29 September 2026: the link answers
15 files
- R/
calculateNoiseRatios.R , R, 26 lines - R/
calculateS50.R , R, 37 lines - R/
calculateURI.R , R, 40 lines - R/
doubleXchr.R , R, 24 lines - R/
normalizeRS.R , R, 25 lines - R/
readRS.R , R, 34 lines - R/
removeNoise.R , R, 24 lines - R/
rescaleRS.R , R, 46 lines - R/
smoothRS.R , R, 43 lines - R/
weight_phases.R , R, 50 lines - R/
writeBedgraph.R , R, 26 lines - R/
writeBigWig.R , R, 37 lines - vignettes/
how-to-use.Rmd , R, 320 lines - LICENSE, License, 674 lines
- README.md, Text, 183 lines
brainbreaks/GROseq
724b09bd564e0cb2ea41a6855ec055c86dd35e11, 19 March 2024Availability: 1 check, the latest on 29 September 2026: the link answers
- 29 September 2026: the link answers
12 files
- preprocess/
download.py , Python, 150 lines - preprocess/
gff_longest_transcript.p , Python, 74 linesy - preprocess/
lsf.py , Python, 104 lines - src/
Add.col.py , Python, 23 lines - src/
COVT.gmean_cal.py , Python, 16 lines - src/
GROSeqPL_v3_updated.py , Python, 192 lines - src/
__init__.py , Python, 1 line - src/
bam2bw.py , Python, 74 lines - src/
bam2wig.py , Python, 104 lines - src/
extract_rpkm.py , Python, 12 lines - src/
misc.py , Python, 59 lines - README.md, Text, 152 lines
dellytools/delly
c6474a25a125982a85829749ab7c14c63eaf17a2, 24 September 2026Availability: 1 check, the latest on 29 September 2026: the link answers
- 29 September 2026: the link answers
43 files
- R/
cnv.R , R, 33 lines - R/
gcbias.R , R, 19 lines - R/
rd.R , R, 84 lines - scripts/
delly2bnd.py , Python, 165 lines - src/
align.h , C/C++, 249 lines - src/
asmode.h , C/C++, 897 lines - src/
assemble.h , C/C++, 969 lines - src/
bed.h , C/C++, 71 lines - src/
bolog.h , C/C++, 156 lines - src/
cluster.h , C/C++, 646 lines - src/
cnv.h , C/C++, 899 lines - src/
coral.h , C/C++, 933 lines, 1 match - src/
coverage.h , C/C++, 785 lines - src/
delly.cpp , C++, 87 lines - src/
delly.h , C/C++, 430 lines - src/
edlib.cpp , C++, 1,480 lines - src/
edlib.h , C/C++, 277 lines - src/
filter.h , C/C++, 1,407 lines - src/
gaf.h , C/C++, 172 lines - src/
gcbias.h , C/C++, 445 lines - src/
genotype.h , C/C++, 401 lines - src/
gfa.h , C/C++, 214 lines - src/
gotoh.h , C/C++, 194 lines - src/
junction.h , C/C++, 752 lines - src/
merge.h , C/C++, 2,166 lines - src/
methyl.h , C/C++, 572 lines - src/
modvcf.h , C/C++, 816 lines - src/
msa.h , C/C++, 250 lines - src/
needle.h , C/C++, 409 lines - src/
pangenome.h , C/C++, 289 lines - src/
ploidy.h , C/C++, 368 lines - src/
popgen.h , C/C++, 305 lines - src/
scan.h , C/C++, 285 lines - src/
shortpe.h , C/C++, 626 lines - src/
split.h , C/C++, 677 lines - src/
svanno.h , C/C++, 241 lines - src/
tags.h , C/C++, 375 lines - src/
tegua.h , C/C++, 450 lines - src/
threadpool.h , C/C++, 81 lines - src/
util.h , C/C++, 941 lines - src/
version.h , C/C++, 60 lines - LICENSE, License, 21 lines
- README.md, Text, 266 lines
friendsofstrandseq/mosaicatcher-pipeline
180b6591078f2a1bd28b2cdd757c6ee31c6299fc, 26 August 2026Availability: 1 check, the latest on 29 September 2026: the link answers
- 29 September 2026: the link answers
275 files
- .github/
scripts/ , Shell, 179 lineschangelog_generator.sh - .github/
scripts/ , Shell, 63 linespull-apptainer-container s.sh - .github/
scripts/ , Shell, 24 linessync_ai_docs.sh - afac/
DL_scNOVA.dev.py , Python, 35 lines - afac/
StrandPhaseR_pipeline.R , R, 23 lines - afac/
Untitled.ipynb , Jupyter, 33 lines - afac/
Untitled1.ipynb , Jupyter, 1 line - afac/
alice_binning.ipynb , Jupyter, 33 lines - afac/
argparse_script.py , Python, 1 line - afac/
bamtofastq.sh , Shell, 13 lines - afac/
bcftools_ploidy.py.ipynb , Jupyter, 22 lines - afac/
change-log-builder.sh , Shell, 12 lines - afac/
chrxy_analysis.ipynb , Jupyter, 108 lines - afac/
chrxy_analysis.py , Python, 69 lines - afac/
compare_tf1_tf2_results. , Python, 49 linespy - afac/
compute_gc_windows.py.ip , Jupyter, 53 linesynb - afac/
convert_strandphaser_inp , R, 17 linesut.R - afac/
convert_tf_model.py , Python, 31 lines - afac/
counts_to_ucsc_bed.py , Python, 45 lines - afac/
dev.py.ipynb , Jupyter, 131 lines - afac/
dev_log_useful.py , Python, 52 lines - afac/
draft.py , Python, 28 lines - afac/
facount_bedtools_gc.sh , Shell, 23 lines - afac/
filter_bad_cells.py , Python, 78 lines - afac/
gitpod_prepa.sh , Shell, 5 lines - afac/
handle_selected.py , Python, 12 lines - afac/
haplotag.Table.snakemake , R, 19 lines.R - afac/
haplotagProbs.snakemake. , R, 22 linesR - afac/
hgsvc_samples_symlink.py , Python, 69 lines - afac/
lite_mosaic_count.py , Python, 15 lines - afac/
load_model_tf2.py , Python, 25 lines - afac/
load_rdata.R , R, 1 line - afac/
mice_preparation/ , Jupyter, 32 linesmerge_predictions.ipynb - afac/
mosaiClassifier.snakemak , R, 34 linese.R - afac/
mosaiClassifier_call.sna , R, 26 lineskemake.R - afac/
mosaicatcher_storage.py. , Jupyter, 46 linesipynb - afac/
mosaiclassifier_arbigent , R, 76 lines_flat.R - afac/
paper_fastq_md5.sh , Shell, 12 lines - afac/
plate_plot.dev.ipynb , Jupyter, 41 lines - afac/
ploidy_plots.py.ipynb , Jupyter, 64 lines - afac/
ploidy_processing.py.ipy , Jupyter, 10 linesnb - afac/
plot-clustering.snakemak , R, 8 linese.R - afac/
plot_plate.R , R, 68 lines - afac/
prepare_lite_example/ , Shell, 11 linesBM510_from_complete_to_c hr21.sh - afac/
prepare_segments.r , R, 13 lines - afac/
qc_plots.py.ipynb , Jupyter, 28 lines - afac/
quota_korbel.py , Python, 164 lines - afac/
scNOVA/ , Jupyter, 21 linesdummy_subclonality.py.ip ynb - afac/
segdups_liftover.py , Python, 28 lines - afac/
sv_df.ipynb , Jupyter, 134 lines - afac/
sv_to_ucsc_bed.py , Python, 48 lines - afac/
test_ggplot.r , R, 23 lines - afac/
test_pypdf.py , Python, 11 lines - afac/
test_pysam.py , Python, 27 lines - afac/
test_readgaligmentspairs , R, 14 lines.R - afac/
tmp_input_smk.py , Python, 78 lines - afac/
touch_stranphaser_files. , Shell, 26 linessh - afac/
trycatch_test.R , R, 84 lines - afac/
ucsc_vizu.py , Python, 99 lines - afac/
update_sm_tag.sh , Shell, 67 lines - afac/
update_tag.sh , Shell, 8 lines - afac/
update_timestamps.py , Python, 25 lines - afac/
zenodo_notebook.ipynb , Jupyter, 153 lines - afac/
zenodo_push.py , Python, 93 lines - github-actions-runner/
add_T2T_part_to_Dockerfi , Shell, 34 linesle.sh - github-actions-runner/
bioconductor_install.R , R, 6 lines - github-actions-runner/
clean.sh , Shell, 46 lines - run_debug.sh, Shell, 7 lines
- workflow/
scripts/ , R, 158 linesGC/ GC_correction.R - workflow/
scripts/ , R, 115 linesGC/ library_size_normalisati on.R - workflow/
scripts/ , R, 189 linesGC/ variance_stabilizing_tra nsformation.R - workflow/
scripts/ , R, 107 linesarbigent/ add_filter.R - workflow/
scripts/ , R, 290 linesarbigent/ bulk_helpers.R - workflow/
scripts/ , R, 222 linesarbigent/ clean_genotype.R - workflow/
scripts/ , R, 202 linesarbigent/ clean_genotype_defs.R - workflow/
scripts/ , R, 399 linesarbigent/ genotypes_qc.R - workflow/
scripts/ , R, 162 linesarbigent/ phasing/ gt_concordance_pics.R - workflow/
scripts/ , R, 79 linesarbigent/ phasing/ gt_concordance_pics_back up.R - workflow/
scripts/ , R, 296 linesarbigent/ phasing/ phayzR.R - workflow/
scripts/ , Shell, 29 linesarbigent/ phasing/ similarity-matrix-wrappe r.sh - workflow/
scripts/ , Perl, 236 linesarbigent/ phasing/ similarity-matrix.pl - workflow/
scripts/ , R, 303 linesarbigent/ postprocess_helpers.R - workflow/
scripts/ , R, 339 linesarbigent/ probability_helpers.R - workflow/
scripts/ , R, 139 linesarbigent/ probability_helpers_2.R - workflow/
scripts/ , R, 205 linesarbigent/ qc_res_verdicted.R - workflow/
scripts/ , R, 410 linesarbigent/ regenotype.R - workflow/
scripts/ , R, 689 linesarbigent/ regenotype_helpers.R - workflow/
scripts/ , R, 419 linesarbigent/ revision_big_table.R - workflow/
scripts/ , R, 344 linesarbigent/ table_to_vcfs.R - workflow/
scripts/ , R, 167 linesarbigent/ vcf_defs.R - workflow/
scripts/ , Python, 131 linesarbigent_utils/ create_hdf.py - workflow/
scripts/ , R, 69 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier.snakemak e.R - workflow/
scripts/ , R, 62 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier/ debugUtils.R - workflow/
scripts/ , R, 132 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier/ generateHaploStates.R - workflow/
scripts/ , R, 187 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier/ getCountsPerSegment.R - workflow/
scripts/ , R, 74 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier/ getDispParAndSegType.R - workflow/
scripts/ , R, 102 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier/ getStrandStates.R - workflow/
scripts/ , R, 107 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier/ haploAndGenoName.R - workflow/
scripts/ , R, 276 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier/ makeSVcalls.R - workflow/
scripts/ , R, 354 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier/ mosaiClassifier.R - workflow/
scripts/ , R, 12 linesarbigent_utils/ mosaiclassifier_scripts/ mosaiClassifier_CN_call. snakemake.R - workflow/
scripts/ , Jupyter, 63 linesarbigent_utils/ posprocess_arbigent_call s.py.ipynb - workflow/
scripts/ , Python, 613 linesarbigent_utils/ watson_crick.py - workflow/
scripts/ , R, 158 linesashleys/ GC/ GC_correction.R - workflow/
scripts/ , Jupyter, 274 linesashleys/ GC/ GC_correction_dev.R.ipyn b - workflow/
scripts/ , R, 174 linesashleys/ GC/ VST_dev.R - workflow/
scripts/ , R, 51 linesashleys/ GC/ gc.R - workflow/
scripts/ , R, 115 linesashleys/ GC/ library_size_normalisati on.R - workflow/
scripts/ , R, 189 linesashleys/ GC/ variance_stabilizing_tra nsformation.R - workflow/
scripts/ , Python, 74 linesashleys/ housekeeping/ scripts/ add_containers_to_rules. py - workflow/
scripts/ , Python, 79 linesashleys/ housekeeping/ scripts/ update_rules_containers. py - workflow/
scripts/ , Python, 105 linesashleys/ housekeeping/ update_rules_for_contain ers.py - workflow/
scripts/ , R, 100 linesashleys/ plotting/ plot_plate.R - workflow/
scripts/ , R, 109 linesashleys/ plotting/ plot_plate_dev.R - workflow/
scripts/ , Python, 231 linesashleys/ utils/ check_all_modes_containe rs.py - workflow/
scripts/ , Python, 68 linesashleys/ utils/ check_container_hashes.p y - workflow/
scripts/ , Python, 52 linesashleys/ utils/ convert_to_container_fun ction.py - workflow/
scripts/ , Python, 38 linesashleys/ utils/ dev_all_cells_correct.py - workflow/
scripts/ , Python, 22 linesashleys/ utils/ dump_config.py - workflow/
scripts/ , Python, 24 linesashleys/ utils/ generate_exclude_file.py - workflow/
scripts/ , Python, 109 linesashleys/ utils/ handle_input.py - workflow/
scripts/ , Python, 67 linesashleys/ utils/ jupyter_utils.py - workflow/
scripts/ , Python, 85 linesashleys/ utils/ make_log_useful_ashleys. py - workflow/
scripts/ , Python, 181 linesashleys/ utils/ pipeline_aesthetic_start _ashleys.py - workflow/
scripts/ , Python, 57 linesashleys/ utils/ populated_counts_for_qc_ plot.py - workflow/
scripts/ , Python, 57 linesashleys/ utils/ positive_control_bypass_ dev.py - workflow/
scripts/ , Python, 82 linesashleys/ utils/ positive_negative_contro l_bypass.py - workflow/
scripts/ , Python, 67 linesashleys/ utils/ publishdir.py - workflow/
scripts/ , Python, 9 linesashleys/ utils/ reformat_ms_norm.py - workflow/
scripts/ , Python, 20 linesashleys/ utils/ rm_unselected_cells.py - workflow/
scripts/ , Python, 8 linesashleys/ utils/ samtools_idxstats_aggr.p y - workflow/
scripts/ , Python, 24 linesashleys/ utils/ symlink_selected_bam.py - workflow/
scripts/ , Python, 6 linesashleys/ utils/ tune_predictions_based_o n_threshold.py - workflow/
scripts/ , Python, 91 linesashleys/ utils/ update_container_referen ces.py - workflow/
scripts/ , Python, 92 linesashleys/ utils/ update_rule_containers.p y - workflow/
scripts/ , R, 13 linesbreakpointr_scripts/ breakpointR_pipeline.R - workflow/
scripts/ , Python, 39 linesbreakpointr_scripts/ prepare_breakpointr.py - workflow/
scripts/ , Shell, 41 linescoverage_analysis_mm_sam ples/ analyze_all_samples.sh - workflow/
scripts/ , Python, 155 linescoverage_analysis_mm_sam ples/ analyze_chrom_coverage.p y - workflow/
scripts/ , Python, 215 linescoverage_analysis_mm_sam ples/ analyze_coverage.py - workflow/
scripts/ , Shell, 55 linescoverage_analysis_mm_sam ples/ collect_coverage.sh - workflow/
scripts/ , Python, 150 linescoverage_analysis_mm_sam ples/ compare_all_mm_samples.p y - workflow/
scripts/ , Python, 106 linescoverage_analysis_mm_sam ples/ compare_info_raw.py - workflow/
scripts/ , Python, 70 linescoverage_analysis_mm_sam ples/ compare_samples.py - workflow/
scripts/ , Python, 113 linesgenome_browsing/ generate_jbrowse_tracks. py - workflow/
scripts/ , R, 240 lines, 1 matchhaplotagging_scripts/ haplotagTable.R - workflow/
scripts/ , R, 13 lineshaplotagging_scripts/ haplotagTable.snakemake. R - workflow/
scripts/ , Python, 93 linesmosaiclassifier_scripts/ call-complex-regions.py - workflow/
scripts/ , R, 87 linesmosaiclassifier_scripts/ haplotagProbs.R - workflow/
scripts/ , R, 21 linesmosaiclassifier_scripts/ haplotagProbs.snakemake. R - workflow/
scripts/ , R, 31 linesmosaiclassifier_scripts/ mosaiClassifier.snakemak e.R - workflow/
scripts/ , R, 62 linesmosaiclassifier_scripts/ mosaiClassifier/ debugUtils.R - workflow/
scripts/ , R, 132 linesmosaiclassifier_scripts/ mosaiClassifier/ generateHaploStates.R - workflow/
scripts/ , R, 206 linesmosaiclassifier_scripts/ mosaiClassifier/ getCountsPerSegment.R - workflow/
scripts/ , R, 74 linesmosaiclassifier_scripts/ mosaiClassifier/ getDispParAndSegType.R - workflow/
scripts/ , R, 77 linesmosaiclassifier_scripts/ mosaiClassifier/ getStrandStates.R - workflow/
scripts/ , R, 58 linesmosaiclassifier_scripts/ mosaiClassifier/ haploAndGenoName.R - workflow/
scripts/ , R, 129 linesmosaiclassifier_scripts/ mosaiClassifier/ makeSVcalls.R - workflow/
scripts/ , R, 405 linesmosaiclassifier_scripts/ mosaiClassifier/ mosaiClassifier.R - workflow/
scripts/ , R, 44 linesmosaiclassifier_scripts/ mosaiClassifier_call.sna kemake.R - workflow/
scripts/ , R, 12 linesmosaiclassifier_scripts/ mosaiClassifier_call_bia llelic.snakemake.R - workflow/
scripts/ , Python, 104 linesnormalization/ merge-blacklist.py - workflow/
scripts/ , R, 160 linesnormalization/ normalize.R - workflow/
scripts/ , Python, 17 linesploidy/ ploidy_bcftools.py - workflow/
scripts/ , Python, 648 linesploidy/ ploidy_estimator.py - workflow/
scripts/ , Python, 6 linesploidy/ summarise_ploidy.py - workflow/
scripts/ , R, 1 lineplotting/ GC_correction.R - workflow/
scripts/ , Python, 35 linesplotting/ dividing_pdf.py - workflow/
scripts/ , Shell, 63 linesplotting/ generate_IGV_session.sh - workflow/
scripts/ , Python, 56 linesplotting/ ploidy_plot.py - workflow/
scripts/ , Python, 426 linesplotting/ plot-clustering-scale.py - workflow/
scripts/ , R, 308 linesplotting/ plot-clustering.R - workflow/
scripts/ , R, 21 linesplotting/ plot-clustering.snakemak e.R - workflow/
scripts/ , R, 784 linesplotting/ plot-sv-calls.R - workflow/
scripts/ , Python, 277 linesplotting/ plot_clustering_scale_fl at.py - workflow/
scripts/ , R, 474 linesplotting/ qc.R - workflow/
scripts/ , R, 21 linesplotting/ scTRIP_multiplot/ scTRIP_multiplot_run.R - workflow/
scripts/ , Shell, 43 linesplotting/ split_ucsc_file.sh - workflow/
scripts/ , R, 151 linesplotting/ sv_consistency_barplot.R - workflow/
scripts/ , R, 7 linesplotting/ sv_consistency_barplot.s nakemake.R - workflow/
scripts/ , Python, 99 linesplotting/ ucsc_vizu.py - workflow/
scripts/ , Python, 32 linespostprocessing/ apply_filter.py - workflow/
scripts/ , Python, 36 linespostprocessing/ create-sv-group-track.py - workflow/
scripts/ , Perl, 237 linespostprocessing/ filter_MosaiCatcher_call s.pl - workflow/
scripts/ , Perl, 213 linespostprocessing/ group_nearby_calls_of_sa me_AF_and_generate_outpu t_table.pl - workflow/
scripts/ , Python, 26 linesscNOVA_scripts/ Deeplearning_Nucleosome_ predict_train_RPE.py - workflow/
scripts/ , R, 188 linesscNOVA_scripts/ NO_chromVAR.R - workflow/
scripts/ , Perl, 9 linesscNOVA_scripts/ Strand_seq_deeptool_Gene s_for_CNN.pl - workflow/
scripts/ , Perl, 18 linesscNOVA_scripts/ Strand_seq_sc_process_ha plo_pipeline.pl - workflow/
scripts/ , R, 62 linesscNOVA_scripts/ annot_expressed.R - workflow/
scripts/ , Python, 61 linesscNOVA_scripts/ assert_list_of_cells.py - workflow/
scripts/ , R, 81 linesscNOVA_scripts/ combine_feature_sc.R - workflow/
scripts/ , R, 77 linesscNOVA_scripts/ combine_features.R - workflow/
scripts/ , R, 77 linesscNOVA_scripts/ combine_features_ploidy. R - workflow/
scripts/ , R, 33 linesscNOVA_scripts/ count_add_peakind.R - workflow/
scripts/ , R, 7 linesscNOVA_scripts/ count_add_peakind.snakem ake.R - workflow/
scripts/ , R, 136 linesscNOVA_scripts/ count_norm.R - workflow/
scripts/ , R, 136 linesscNOVA_scripts/ count_norm_ploidy.R - workflow/
scripts/ , R, 30 linesscNOVA_scripts/ count_sort_annotate_chri d_CREs.R - workflow/
scripts/ , R, 8 linesscNOVA_scripts/ count_sort_annotate_gene id.R - workflow/
scripts/ , R, 14 linesscNOVA_scripts/ count_sort_label.R - workflow/
scripts/ , Python, 40 linesscNOVA_scripts/ dev_aggr.py - workflow/
scripts/ , R, 158 linesscNOVA_scripts/ feature_sc_var.R - workflow/
scripts/ , Python, 6 linesscNOVA_scripts/ filter_input_subclonalit y.py - workflow/
scripts/ , Python, 4 linesscNOVA_scripts/ filter_sv_calls.py - workflow/
scripts/ , Python, 5 linesscNOVA_scripts/ gather_infer_expr_genes_ split.py - workflow/
scripts/ , R, 227 linesscNOVA_scripts/ generate_CN_for_CNN.R - workflow/
scripts/ , R, 319 linesscNOVA_scripts/ generate_CN_for_chromVAR .R - workflow/
scripts/ , R, 185 linesscNOVA_scripts/ generate_CREs_CN.R - workflow/
scripts/ , R, 159 linesscNOVA_scripts/ generate_feature_sc.R - workflow/
scripts/ , R, 157 linesscNOVA_scripts/ generate_feature_sc_CN.R - workflow/
scripts/ , R, 343 linesscNOVA_scripts/ infer_diff_gene_expressi on.R - workflow/
scripts/ , R, 479 linesscNOVA_scripts/ infer_diff_gene_expressi on_alt.R - workflow/
scripts/ , Perl, 61 linesscNOVA_scripts/ merge_bam_clones.pl - workflow/
scripts/ , Perl, 91 linesscNOVA_scripts/ perl_test_all_snake.pl - workflow/
scripts/ , Perl, 9 linesscNOVA_scripts/ rename.pl - workflow/
scripts/ , R, 15 linesscNOVA_scripts/ script_PLSDA/ Pred_PLS_R_scNOVA.R - workflow/
scripts/ , R, 8 linesscNOVA_scripts/ script_PLSDA/ auto_R.R - workflow/
scripts/ , R, 27 linesscNOVA_scripts/ script_PLSDA/ conpred_R.R - workflow/
scripts/ , R, 25 linesscNOVA_scripts/ script_PLSDA/ normc.R - workflow/
scripts/ , R, 88 linesscNOVA_scripts/ script_PLSDA/ pls_R_scNOVA.R - workflow/
scripts/ , R, 43 linesscNOVA_scripts/ script_PLSDA/ plsnipal_R.R - workflow/
scripts/ , R, 8 linesscNOVA_scripts/ script_PLSDA/ vip_R_v2_scNOVA.R - workflow/
scripts/ , R, 56 linesscNOVA_scripts/ script_PLSDA/ vip_null_R.R - workflow/
scripts/ , Python, 472 linessegmentation_scripts/ detect_strand_states.py - workflow/
scripts/ , Python, 201 linesstats/ callset_summary_stats.py - workflow/
scripts/ , Python, 31 linesstats/ summary_stats.py - workflow/
scripts/ , Python, 56 linesstats/ transpose_table.py - workflow/
scripts/ , R, 77 linesstrandphaser_scripts/ StrandPhaseR_pipeline.R - workflow/
scripts/ , Python, 47 linesstrandphaser_scripts/ combine_strandphaser_out put.py - workflow/
scripts/ , R, 10 linesstrandphaser_scripts/ helper.convert_strandpha ser_input.R - workflow/
scripts/ , R, 44 linesstrandphaser_scripts/ helper.convert_strandpha ser_output.R - workflow/
scripts/ , R, 53 linesstrandphaser_scripts/ install_strandphaser.R - workflow/
scripts/ , Python, 36 linesstrandphaser_scripts/ prepare_strandphaser.py - workflow/
scripts/ , Shell, 118 linestmux-pipelines/ run_mosaicatcher.sh - workflow/
scripts/ , Shell, 277 linestoolbox/ pull_containers.sh - workflow/
scripts/ , Python, 247 linestoolbox/ reorganise_data/ reoganise_aviti_data.py - workflow/
scripts/ , Python, 80 linestoolbox/ reorganise_data/ reorganise_dkfz_data.py - workflow/
scripts/ , Python, 76 linestoolbox/ reorganise_data/ reorganise_landsdorp_dat a.py - workflow/
scripts/ , Shell, 23 linestoolbox/ reorganise_data/ rsync_fastq_folders.sh - workflow/
scripts/ , Shell, 31 linestoolbox/ setup_shared_caches.sh - workflow/
scripts/ , Jupyter, 668 linestoolbox/ slurm_efficiency_analysi s.ipynb - workflow/
scripts/ , Python, 1 linetoolbox/ watcher/ __init__.py - workflow/
scripts/ , Python, 237 linestoolbox/ watcher/ config.py - workflow/
scripts/ , Python, 173 linestoolbox/ watcher/ heartbeat_monitor.py - workflow/
scripts/ , Python, 731 linestoolbox/ watcher/ watcher.py - workflow/
scripts/ , Python, 47 linesutils/ check_sm_tag.py - workflow/
scripts/ , Python, 101 linesutils/ chrxy_analysis.py - workflow/
scripts/ , Python, 50 linesutils/ detect_single_paired_end .py - workflow/
scripts/ , Python, 22 linesutils/ dump_config.py - workflow/
scripts/ , Python, 88 linesutils/ filter_bad_cells.py - workflow/
scripts/ , Shell, 40 linesutils/ generate_bin_bed.sh - workflow/
scripts/ , Python, 83 linesutils/ generate_exclude_file.py - workflow/
scripts/ , Python, 113 linesutils/ generate_gc_matrix.py - workflow/
scripts/ , Python, 297 linesutils/ handle_input.py - workflow/
scripts/ , Python, 43 linesutils/ handle_input_old_behavio r.py - workflow/
scripts/ , R, 71 linesutils/ install_R_package.R - workflow/
scripts/ , R, 12 linesutils/ install_R_tarball.R - workflow/
scripts/ , Python, 67 linesutils/ jupyter_utils.py - workflow/
scripts/ , Python, 90 linesutils/ make_log_useful.py - workflow/
scripts/ , Python, 210 linesutils/ pipeline_aesthetic_start .py - workflow/
scripts/ , Python, 58 linesutils/ populated_counts_for_qc_ plot.py - workflow/
scripts/ , Python, 9 linesutils/ reformat_ms_norm.py - workflow/
scripts/ , Python, 60 linesutils/ run_summary.py - workflow/
scripts/ , Python, 12 linesutils/ sort_counts.py - workflow/
scripts/ , Python, 24 linesutils/ symlink_selected_bam.py - workflow/
scripts/ , Python, 9 linesutils/ utils.py - workflow/
scripts/ , Python, 44 linesutils/ version.py - workflow/
scripts/ , Python, 44 linesutils/ zenodo.py - workflow/
snakemake_profiles/ , Shell, 16 linesmosaicatcher-pipeline/ v7/ HPC/ dev/ slurm_EMBL/ status-sacct.sh - workflow/
snakemake_profiles/ , Shell, 16 linesmosaicatcher-pipeline/ v7/ HPC/ dev/ slurm_legacy_conda/ status-sacct.sh - workflow/
snakemake_profiles/ , Shell, 16 linesmosaicatcher-pipeline/ v7/ HPC/ dev/ slurm_update_conda/ status-sacct.sh - workflow/
snakemake_profiles/ , Shell, 22 linesmosaicatcher-pipeline/ v9/ HPC/ slurm_EMBL_apptainer/ setup_cache.sh - LICENSE, License, 21 lines
- README.md, Text, 79 lines
Zenodo 10843397
Availability: 1 check, the latest on 29 September 2026: the link answers (HTTP 200)
- 29 September 2026: the link answers (HTTP 200)
35 files
- QCReport/
app.R , R, 1 line - QCReport/
global.R , R, 20 lines - QCReport/
graphics.R , R, 325 lines - QCReport/
logs.R , R, 12 lines - QCReport/
offtargets.R , R, 88 lines - QCReport/
server.R , R, 718 lines - QCReport/
ui.R , R, 106 lines - transloc_pipeline_v3/
R/ , R, 118 linesRsub.R - transloc_pipeline_v3/
R/ , R, 126 linesTLXBaitMH.R - transloc_pipeline_v3/
R/ , R, 89 linesTLXBedIntersect.R - transloc_pipeline_v3/
R/ , R, 172 linesTLXBrkChr.R - transloc_pipeline_v3/
R/ , R, 130 linesTLXWidespreadLowLevel.R - transloc_pipeline_v3/
R/ , R, 179 linesTranslocDedup.R - transloc_pipeline_v3/
R/ , R, 142 linesTranslocFilter.R - transloc_pipeline_v3/
R/ , R, 478 linesTranslocHelper.R - transloc_pipeline_v3/
R/ , R, 253 linesTranslocHotSpots.R - transloc_pipeline_v3/
R/ , R, 366 linesTranslocPlot.R - transloc_pipeline_v3/
R/ , R, 50 linesTranslocRepeatSeq.R - transloc_pipeline_v3/
R/ , R, 171 linesplotSwitchRegions.R - transloc_pipeline_v3/
bin/ , Perl, 153 linesTranslocDedup.pl - transloc_pipeline_v3/
bin/ , Perl, 148 linesTranslocFilter.pl - transloc_pipeline_v3/
bin/ , Perl, 302 linesTranslocHTMLReads.pl - transloc_pipeline_v3/
bin/ , Perl, 1,147 lines, 1 matchTranslocPipeline.pl - transloc_pipeline_v3/
bin/ , Perl, 459 linesTranslocPreprocess.pl - transloc_pipeline_v3/
bin/ , Perl, 150 linesTranslocRepeatSeq.pl - transloc_pipeline_v3/
bin/ , Perl, 495 linesTranslocWrapper.pl - transloc_pipeline_v3/
lib/ , Perl, 601 linesPipelineHelper.pl - transloc_pipeline_v3/
lib/ , Perl, 317 linesTranslocFilters.pl - transloc_pipeline_v3/
lib/ , Perl, 1,302 linesTranslocHelper.pl - transloc_pipeline_v3/
lib/ , Perl, 417 linesTranslocSub.pl - transloc_pipeline_v3/
tools/ , Perl, 353 linesModifyGenome.pl - transloc_pipeline_v3/
tools/ , Perl, 160 linescheckUnmatchedBarcodes.p l - transloc_pipeline_v3/
tools/ , Perl, 34 linestlx2BED-MACS.pl - transloc_pipeline_v3/
tools/ , Perl, 188 linestlxToBed.pl - README.md, Text, 116 lines
Zenodo 10838367
Availability: 1 check, the latest on 29 September 2026: the link answers (HTTP 200)
- 29 September 2026: the link answers (HTTP 200)
12 files
- preprocess/
download.py , Python, 150 lines - preprocess/
gff_longest_transcript.p , Python, 74 linesy - preprocess/
lsf.py , Python, 104 lines - src/
Add.col.py , Python, 23 lines - src/
COVT.gmean_cal.py , Python, 16 lines - src/
GROSeqPL_v3_updated.py , Python, 192 lines - src/
__init__.py , Python, 1 line - src/
bam2bw.py , Python, 74 lines - src/
bam2wig.py , Python, 104 lines - src/
extract_rpkm.py , Python, 12 lines - src/
misc.py , Python, 59 lines - README.md, Text, 149 lines
Zenodo 10838365
Availability: 1 check, the latest on 29 September 2026: the link answers (HTTP 200)
- 29 September 2026: the link answers (HTTP 200)
5 files
- align.R, R, 228 lines
- checksum.R, R, 115 lines
- coverage.R, R, 283 lines
- utils.R, R, 98 lines
- README.md, Text, 104 lines
Code availability
In this manuscript, three singularity containers were applied for HTGTS (10.5281/
Reproduced under the paper's license (CC BY), from the paper cited above.
Tracing map
Proposed by the machine: these links were found in the paper and verified at the source, without human review. The map will receive a Zenodo DOI once one of the paper's authors has validated it with their ORCID.
What the map holds:
- 8 repositories of the authors' code, each at its verified commit, with its license and how the link was found in the paper;
- 484 scripts, each with its path and the digest of its content;
- 3 matches between paragraphs of the paper and lines of the code (method lexical-v1);
- neither the text of the paper nor the code itself.
Its JSON (tracing-map.json) is deposited on Zenodo with its DOI once the map is validated.
Data
Datasets cited
- bioproject:PRJEB95922, at NCBI BioProject; found in “Data availability”
- geo:GSE305346, at NCBI GEO; found in “Data availability”
Data Availability Statement
The whole genome sequencing data generated in this study have been deposited in the European Genome Archive database under accession code PRJEB95922 (https://
In this manuscript, three singularity containers were applied for HTGTS (10.5281/
Reproduced under the paper's license (CC BY), from the paper cited above.
Versions
The history of this record: each version stored by the harvester or made by a correction of its authors or of the maintainers of its code, and what changed in its facts. The texts of the paper (its abstract, its availability statements) are not part of it; versions that changed only those are not listed.
Version 1, 29 September 2026: the first record
Recorded: type, language, journal, volume, issue, pages, dates, 18 authors, 4 keywords, 12 MeSH terms, 2 funders, 62 references.
Cite
This paper
Corazzi, L., Ing, A., Benito, E., Cosenza, M. R., Hasenfeld, P., Weber, T., Marx, A. J. M., Ionasz, V. S., Trausch, N., Benedetto, S., Di Muzio, G., Ding, B., Berlanda, J., Giaisi, M., Claudino, N., Höfer, T., Korbel, J. O., & Wei, P.-C. (2026). Recurrent DNA break clusters drive replication-stress-induc
BibTeX
@article{corazzi2026recu
author = {Corazzi, Lorenzo and Ing, Alex and Benito, Eva and Cosenza, Marco Raffaele and Hasenfeld, Patrick and Weber, Thomas and Marx, Anna J M and Ionasz, Vivien S and Trausch, Nathan and Benedetto, Sarah and Di Muzio, Giulia and Ding, Boyu and Berlanda, Jana and Giaisi, Marco and Claudino, Nina and Höfer, Thomas and Korbel, Jan O and Wei, Pei-Chi},
title = {{Recurrent DNA break clusters drive replication-stress-induc
journal = {Nature communications},
year = {2026},
month = apr,
volume = {17},
number = {1},
pages = {3627},
publisher = {Nature Publishing Group},
issn = {2041-1723},
doi = {10.1038/
url = {https://
pmid = {42009664},
pmcid = {PMC13096501}
}
RIS
TY - JOUR
AU - Corazzi, Lorenzo
AU - Ing, Alex
AU - Benito, Eva
AU - Cosenza, Marco Raffaele
AU - Hasenfeld, Patrick
AU - Weber, Thomas
AU - Marx, Anna J M
AU - Ionasz, Vivien S
AU - Trausch, Nathan
AU - Benedetto, Sarah
AU - Di Muzio, Giulia
AU - Ding, Boyu
AU - Berlanda, Jana
AU - Giaisi, Marco
AU - Claudino, Nina
AU - Höfer, Thomas
AU - Korbel, Jan O
AU - Wei, Pei-Chi
TI - Recurrent DNA break clusters drive replication-stress-induc
T2 - Nature communications
J2 - Nat Commun
PY - 2026
DA - 2026/
VL - 17
IS - 1
SP - 3627
SN - 2041-1723
PB - Nature Publishing Group
DO - 10.1038/
UR - https://
LA - en
ER -
CSL-JSON
{
"id": "10.1038/
"type": "article-journal",
"title": "Recurrent DNA break clusters drive replication-stress-induc
"container-title": "Nature communications",
"author": [
{
"family": "Corazzi",
"given": "Lorenzo"
},
{
"family": "Ing",
"given": "Alex"
},
{
"family": "Benito",
"given": "Eva"
},
{
"family": "Cosenza",
"given": "Marco Raffaele"
},
{
"family": "Hasenfeld",
"given": "Patrick"
},
{
"family": "Weber",
"given": "Thomas"
},
{
"family": "Marx",
"given": "Anna J M"
},
{
"family": "Ionasz",
"given": "Vivien S"
},
{
"family": "Trausch",
"given": "Nathan"
},
{
"family": "Benedetto",
"given": "Sarah"
},
{
"family": "Di Muzio",
"given": "Giulia"
},
{
"family": "Ding",
"given": "Boyu"
},
{
"family": "Berlanda",
"given": "Jana"
},
{
"family": "Giaisi",
"given": "Marco"
},
{
"family": "Claudino",
"given": "Nina"
},
{
"family": "Höfer",
"given": "Thomas"
},
{
"family": "Korbel",
"given": "Jan O"
},
{
"family": "Wei",
"given": "Pei-Chi"
}
],
"container-title-short":
"volume": "17",
"issue": "1",
"page": "3627",
"DOI": "10.1038/
"PMID": "42009664",
"PMCID": "PMC13096501",
"ISSN": "2041-1723",
"publisher": "Nature Publishing Group",
"URL": "https://
"language": "en",
"issued": {
"date-parts": [
[
2026,
4,
20
]
]
}
}
The tracing map gets a citation of its own once an author has validated it and it has a DOI.
Similar papers
The papers with a page that share the most with this one: the tools found in their code, their categories, datasets, cited references and authors, the rarest counting most.
- [1] doi:10.1038/s42003-026-10957-8 [code]
- Brain defence by the extracellular matrix protein Cochlin.Journal: Communications biologyIn common: BCFtools, SAMtools, edgeR, 18 other tools, mouse, cellular / molecular
- [2] doi:10.1186/s13059-026-04177-w [code]
- Genomic sequence evolution underlying human neocortical interareal diversification.Journal: Genome biologyIn common: Snakemake, pysam, BEDTools, 17 other tools, genetics / omics, mouse, cellular / molecular
- [3] doi:10.1016/j.xcrm.2026.102766 [code]
- A longitudinal single-cell and spatial multiomic atlas of pediatric high-grade glioma.Journal: Cell reports. MedicineIn common: pysam, edgeR, Keras, 18 other tools, genetics / omics, cellular / molecular
- [4] doi:10.1093/bioinformatics/btag592 [code]
- Network-based stratification of allele-specific expression reveals patient subgroups in Huntington's disease.Journal: Bioinformatics (Oxford, England)In common: BCFtools, SAMtools, edgeR, 17 other tools, genetics / omics
- [5] doi:10.1016/j.celrep.2026.117073 [code]
- Single-cell epigenomics uncovers heterochromatin instability and transcription factor dysfunction during mouse brain aging.Journal: Cell reportsIn common: Snakemake, pysam, BEDTools, 15 other tools, genetics / omics, mouse, cellular / molecular
- [6] doi:10.1038/s41586-026-10512-9 [code]
- Astrocyte glucocorticoid receptor signalling restricts neuronal plasticity.Journal: NatureIn common: pysam, BEDTools, SAMtools, 15 other tools, mouse, cellular / molecular
- [7] doi:10.1038/s41586-026-10612-6 [code]
- Acquired genetic and cell-state changes in IDH-mutant glioma progression.Journal: NatureIn common: Snakemake, BCFtools, pysam, 13 other tools, cellular / molecular
- [8] doi:10.1038/s41467-026-76675-1 [code]
- Long-read proteogenomic atlas of human neuronal differentiation reveals isoform diversity informing neurodevelopmental risk mechanisms.Journal: Nature communicationsIn common: pysam, BEDTools, SAMtools, 15 other tools, genetics / omics
- [9] doi:10.3389/fnmol.2026.1844705 [code]
- Risperidone regulates the expression of schizophrenia-related genes in the forebrain of adult male mice.Journal: Frontiers in molecular neuroscienceIn common: pysam, BEDTools, SAMtools, 14 other tools, genetics / omics, mouse, cellular / molecular
- [10] doi:10.1038/s41467-026-72598-z [code]
- Functional impact of genetic background on variable expressivity in neurodevelopmental disorders.Journal: Nature communicationsIn common: BCFtools, BEDTools, SAMtools, 12 other tools, 2 references
Contribute
The authors of this paper can claim it, correct its record and validate its tracing map, and the maintainers of its code (its owner, or a public member of its organization) correct what it says of their repository; anyone signed in can ask for its removal. Every request goes to OSCR's own machine, which answers it; your account page follows them.
Sign in with ORCID to claim this paper as one of its authors, correct its record or validate its tracing map: when the paper's metadata lists your ORCID iD, you are recognized at once. Maintainers of its code: sign in with GitHub, then claim the repository on your account page.
Claim this paper
Correct its record
Say what each link of this record is, remove the ones that are not the paper's, add the ones that are missing. The correction becomes a new version of the record, in its Versions section.
Validate its tracing map
You validate the map as this page shows it: 8 repositories of the authors' code, each at its verified commit and with its license, 484 scripts, and 3 matches between paragraphs and code (see the Code and Map sections). It then receives a DOI on Zenodo, with you (your ORCID iD) and OSCR as its creators; the code itself is not deposited.
The map's fingerprint: sha256:e759b2cb9a87fbab…
Add the badge to its README
The badge links the code to this page. Copy one of these into the README of the paper's code: only you decide where it goes, and nothing is changed for you.
Markdown
[, paste the snippet at the top, then “Commit changes…” and, to review it first, “Create a new branch and start a pull request”. You open the pull request; OSCR asks for no permission.
Request its removal
To ask OSCR to remove this record, the copies of its authors' scripts or its tracing map, use the removal request page: signed in, you say who you are, what to remove and why, then review and confirm the request. Published rules decide every request (how).
Discussion, reproductions, activity
Discussion: questions and error reports about this paper and its code, from signed-in readers and its authors. It opens with sign-in.
Reproductions: reports from readers who ran the authors' code: what they reproduced, with which environment, commit and data. It opens with sign-in.
Activity: what happens around this paper: new versions of its record, its map's validation, discussions and reproductions. It opens with sign-in.
