2024
Fuhrmann, Lara; Jablonski, Kim Philipp; Topolsky, Ivan; Batavia, Aashil A; Borgsmüller, Nico; Baykal, Pelin Icer; Carrara, Matteo; Chen, Chaoran; Dondi, Arthur; Dragan, Monica; Dreifuss, David; John, Anika; Langer, Benjamin; Okoniewski, Michal; du Plessis, Louis; Schmitt, Uwe; Singer, Franziska; Stadler, Tanja; Beerenwinkel, Niko
V-pipe 3.0: a sustainable pipeline for within-sample viral genetic diversity estimation Journal Article
In: GigaScience, vol. 13, pp. giae065, 2024.
Abstract | Links | BibTeX | Tags: Project 03, WP 1.3 Virus-host interactions, WP 2.1 Microevolution: Virus quasispecies
@article{Fuhrmann2024b,
title = {V-pipe 3.0: a sustainable pipeline for within-sample viral genetic diversity estimation},
author = {Lara Fuhrmann and Kim Philipp Jablonski and Ivan Topolsky and Aashil A Batavia and Nico Borgsmüller and Pelin Icer Baykal and Matteo Carrara and Chaoran Chen and Arthur Dondi and Monica Dragan and David Dreifuss and Anika John and Benjamin Langer and Michal Okoniewski and Louis du Plessis and Uwe Schmitt and Franziska Singer and Tanja Stadler and Niko Beerenwinkel},
doi = {10.1093/gigascience/giae065},
year = {2024},
date = {2024-09-30},
urldate = {2024-09-30},
journal = {GigaScience},
volume = {13},
pages = {giae065},
abstract = {The large amount and diversity of viral genomic datasets generated by next-generation sequencing technologies poses a set of challenges for computational data analysis workflows, including rigorous quality control, scaling to large sample sizes, and tailored steps for specific applications. Here, we present V-pipe 3.0, a computational pipeline designed for analyzing next-generation sequencing data of short viral genomes. It is developed to enable reproducible, scalable, adaptable, and transparent inference of genetic diversity of viral samples. By presenting 2 large-scale data analysis projects, we demonstrate the effectiveness of V-pipe 3.0 in supporting sustainable viral genomic data science.},
keywords = {Project 03, WP 1.3 Virus-host interactions, WP 2.1 Microevolution: Virus quasispecies},
pubstate = {published},
tppubtype = {article}
}
Fuhrmann, Lara; Langer, Benjamin; Topolsky, Ivan; Beerenwinkel, Niko
VILOCA: Sequencing quality-aware haplotype reconstruction and mutation calling for short- and long-read data Journal Article
In: bioRxiv, 2024.
Abstract | Links | BibTeX | Tags: Project 03, WP 1.3 Virus-host interactions, WP 2.1 Microevolution: Virus quasispecies
@article{Fuhrmann2024,
title = {VILOCA: Sequencing quality-aware haplotype reconstruction and mutation calling for short- and long-read data},
author = {Lara Fuhrmann and Benjamin Langer and Ivan Topolsky and Niko Beerenwinkel},
doi = {10.1101/2024.06.06.597712},
year = {2024},
date = {2024-06-09},
journal = {bioRxiv},
abstract = {RNA viruses exist in large heterogeneous populations within their host. The structure and diversity of virus populations affects disease progression and treatment outcomes. Next-generation sequencing allows detailed viral population analysis, but inferring diversity from error-prone reads is challenging. Here, we present VILOCA, a method for mutation calling and reconstruction of local haplotypes from short- and long-read viral sequencing data. Local haplotypes refer to genomic regions that have approximately the length of the input reads. VILOCA recovers local haplotypes by using a Dirichlet process mixture model to cluster reads around their unobserved haplotypes and leveraging quality scores of the sequencing reads. We assessed the performance of VILOCA in terms of mutation calling and haplotype reconstruction accuracy on simulated and experimental Illumina, PacBio, and Oxford Nanopore data. On simulated and experimental Illumina data, VILOCA performed better or similar to existing methods. On the simulated long-read data, VILOCA is able to recover on average 82% of the ground truth mutations with perfect precision compared to only 64% recall and 90% precision of the second-best method. In summary, VILOCA provides significantly improved accuracy in mutation and haplotype calling, especially for long-read sequencing data, and therefore facilitates the comprehensive characterization of heterogeneous within-host viral populations.},
keywords = {Project 03, WP 1.3 Virus-host interactions, WP 2.1 Microevolution: Virus quasispecies},
pubstate = {published},
tppubtype = {article}
}
