diff --git a/paper/paper.bib b/paper/paper.bib index 1b5a703b..9c3396ba 100644 --- a/paper/paper.bib +++ b/paper/paper.bib @@ -172,38 +172,6 @@ @article{dong_integrated_2019 keywords = {metaerg}, } -@article{zhou_metabolic_2022, - title = {{METABOLIC}: high-throughput profiling of microbial genomes for functional traits, metabolism, biogeochemistry, and community-scale functional networks}, - volume = {10}, - issn = {2049-2618}, - url = {https://microbiomejournal.biomedcentral.com/articles/10.1186/s40168-021-01213-8}, - doi = {10.1186/s40168-021-01213-8}, - shorttitle = {{METABOLIC}}, - abstract = {Abstract - - Background - Advances in microbiome science are being driven in large part due to our ability to study and infer microbial ecology from genomes reconstructed from mixed microbial communities using metagenomics and single-cell genomics. Such omics-based techniques allow us to read genomic blueprints of microorganisms, decipher their functional capacities and activities, and reconstruct their roles in biogeochemical processes. Currently available tools for analyses of genomic data can annotate and depict metabolic functions to some extent; however, no standardized approaches are currently available for the comprehensive characterization of metabolic predictions, metabolite exchanges, microbial interactions, and microbial contributions to biogeochemical cycling. - - - Results - We present {METABOLIC} ({METabolic} And {BiogeOchemistry} {anaLyses} In {miCrobes}), a scalable software to advance microbial ecology and biogeochemistry studies using genomes at the resolution of individual organisms and/or microbial communities. The genome-scale workflow includes annotation of microbial genomes, motif validation of biochemically validated conserved protein residues, metabolic pathway analyses, and calculation of contributions to individual biogeochemical transformations and cycles. The community-scale workflow supplements genome-scale analyses with determination of genome abundance in the microbiome, potential microbial metabolic handoffs and metabolite exchange, reconstruction of functional networks, and determination of microbial contributions to biogeochemical cycles. {METABOLIC} can take input genomes from isolates, metagenome-assembled genomes, or single-cell genomes. Results are presented in the form of tables for metabolism and a variety of visualizations including biogeochemical cycling potential, representation of sequential metabolic transformations, community-scale microbial functional networks using a newly defined metric “{MW}-score” (metabolic weight score), and metabolic Sankey diagrams. {METABOLIC} takes {\textasciitilde} 3 h with 40 {CPU} threads to process {\textasciitilde} 100 genomes and corresponding metagenomic reads within which the most compute-demanding part of hmmsearch takes {\textasciitilde} 45 min, while it takes {\textasciitilde} 5 h to complete hmmsearch for {\textasciitilde} 3600 genomes. Tests of accuracy, robustness, and consistency suggest {METABOLIC} provides better performance compared to other software and online servers. To highlight the utility and versatility of {METABOLIC}, we demonstrate its capabilities on diverse metagenomic datasets from the marine subsurface, terrestrial subsurface, meadow soil, deep sea, freshwater lakes, wastewater, and the human gut. - - - Conclusion - - {METABOLIC} enables the consistent and reproducible study of microbial community ecology and biogeochemistry using a foundation of genome-informed microbial metabolism, and will advance the integration of uncultivated organisms into metabolic and biogeochemical models. {METABOLIC} is written in Perl and R and is freely available under {GPLv}3 at - https://github.com/{AnantharamanLab}/{METABOLIC} - .}, - pages = {33}, - number = {1}, - journaltitle = {Microbiome}, - shortjournal = {Microbiome}, - author = {Zhou, Zhichao and Tran, Patricia Q. and Breister, Adam M. and Liu, Yang and Kieft, Kristopher and Cowley, Elise S. and Karaoz, Ulas and Anantharaman, Karthik}, - urldate = {2023-07-18}, - date = {2022-12}, - langid = {english}, -} - @article{das_ht-argfinder_2022, title = {{HT}-{ARGfinder}: A Comprehensive Pipeline for Identifying Horizontally Transferred Antibiotic Resistance Genes and Directionality in Metagenomic Sequencing Data}, volume = {10}, @@ -357,7 +325,7 @@ @article{porse_biochemical_2018 @online{torsten_seemann_abricate_2020, title = {{ABRicate}}, url = {https://github.com/tseemann/abricate}, - author = {{Torsten Seemann}}, + author = {Seemann, Torsten}, date = {2020}, note = {Github}, } @@ -774,4 +742,451 @@ @misc{mendes_hamronization_2024 urldate = {2024-10-11}, date = {2024-03-11}, langid = {english}, -} \ No newline at end of file +} + +@ARTICLE{Hyatt2010-yv, + title = "Prodigal: prokaryotic gene recognition and translation initiation + site identification", + author = "Hyatt, Doug and Chen, Gwo-Liang and Locascio, Philip F and Land, + Miriam L and Larimer, Frank W and Hauser, Loren J", + journal = "BMC bioinformatics", + volume = 11, + pages = 119, + abstract = "BACKGROUND: The quality of automated gene prediction in microbial + organisms has improved steadily over the past decade, but there is + still room for improvement. Increasing the number of correct + identifications, both of genes and of the translation initiation + sites for each gene, and reducing the overall number of false + positives, are all desirable goals. RESULTS: With our years of + experience in manually curating genomes for the Joint Genome + Institute, we developed a new gene prediction algorithm called + Prodigal (PROkaryotic DYnamic programming Gene-finding ALgorithm). + With Prodigal, we focused specifically on the three goals of + improved gene structure prediction, improved translation + initiation site recognition, and reduced false positives. We + compared the results of Prodigal to existing gene-finding methods + to demonstrate that it met each of these objectives. CONCLUSION: + We built a fast, lightweight, open source gene prediction program + called Prodigal http://compbio.ornl.gov/prodigal/. Prodigal + achieved good results compared to existing methods, and we believe + it will be a valuable asset to automated microbial annotation + pipelines.", + month = mar, + year = 2010, + url = "http://dx.doi.org/10.1186/1471-2105-11-119", + doi = "10.1186/1471-2105-11-119", + pmc = "PMC2848648", + pmid = 20211023, + issn = "1471-2105", + language = "en" +} + +@ARTICLE{Seemann2014-ee, + title = "Prokka: rapid prokaryotic genome annotation", + author = "Seemann, Torsten", + journal = "Bioinformatics", + volume = 30, + number = 14, + pages = "2068--2069", + abstract = "UNLABELLED: The multiplex capability and high yield of current day + DNA-sequencing instruments has made bacterial whole genome + sequencing a routine affair. The subsequent de novo assembly of + reads into contigs has been well addressed. The final step of + annotating all relevant genomic features on those contigs can be + achieved slowly using existing web- and email-based systems, but + these are not applicable for sensitive data or integrating into + computational pipelines. Here we introduce Prokka, a command line + software tool to fully annotate a draft bacterial genome in about + 10 min on a typical desktop computer. It produces + standards-compliant output files for further analysis or viewing + in genome browsers. AVAILABILITY AND IMPLEMENTATION: Prokka is + implemented in Perl and is freely available under an open source + GPLv2 license from http://vicbioinformatics.com/.", + month = jul, + year = 2014, + url = "http://dx.doi.org/10.1093/bioinformatics/btu153", + doi = "10.1093/bioinformatics/btu153", + pmid = 24642063, + issn = "1367-4803,1367-4811", + language = "en" +} + +@ARTICLE{Larralde2022-uu, + title = "Pyrodigal: Python bindings and interface to Prodigal, an + efficient method for gene prediction in prokaryotes", + author = "Larralde, Martin", + journal = "Journal of Open Source Software", + publisher = "The Open Journal", + volume = 7, + number = 72, + pages = 4296, + abstract = "Larralde, M., (2022). Pyrodigal: Python bindings and interface to + Prodigal, an efficient method for gene prediction in prokaryotes. + Journal of Open Source Software, 7(72), 4296, + https://doi.org/10.21105/joss.04296", + month = apr, + year = 2022, + url = "http://dx.doi.org/10.21105/joss.04296", + doi = "10.21105/joss.04296", + issn = "2475-9066" +} + +@ARTICLE{Gruning2018-vr, + title = "Bioconda: sustainable and comprehensive software distribution for + the life sciences", + author = "Grüning, Björn and Dale, Ryan and Sjödin, Andreas and Chapman, + Brad A and Rowe, Jillian and Tomkins-Tinch, Christopher H and + Valieris, Renan and Köster, Johannes and {Bioconda Team}", + journal = "Nature methods", + volume = 15, + number = 7, + pages = "475--476", + month = jul, + year = 2018, + url = "http://dx.doi.org/10.1038/s41592-018-0046-7", + doi = "10.1038/s41592-018-0046-7", + pmid = 29967506, + issn = "1548-7091,1548-7105", + language = "en" +} + +@ARTICLE{Da_Veiga_Leprevost2017-gl, + title = "{BioContainers}: an open-source and community-driven framework + for software standardization", + author = "da Veiga Leprevost, Felipe and Grüning, Björn A and Alves + Aflitos, Saulo and Röst, Hannes L and Uszkoreit, Julian and + Barsnes, Harald and Vaudel, Marc and Moreno, Pablo and Gatto, + Laurent and Weber, Jonas and Bai, Mingze and Jimenez, Rafael C + and Sachsenberg, Timo and Pfeuffer, Julianus and Vera Alvarez, + Roberto and Griss, Johannes and Nesvizhskii, Alexey I and + Perez-Riverol, Yasset", + journal = "Bioinformatics (Oxford, England)", + publisher = "Oxford University Press (OUP)", + volume = 33, + number = 16, + pages = "2580--2582", + abstract = "Abstract Motivation BioContainers (biocontainers.pro) is an + open-source and community-driven framework which provides + platform independent executable environments for bioinformatics + software. BioContainers allows labs of all sizes to easily + install bioinformatics software, maintain multiple versions of + the same software and combine tools into powerful analysis + pipelines. BioContainers is based on popular open-source projects + Docker and rkt frameworks, that allow software to be installed + and executed under an isolated and controlled environment. Also, + it provides infrastructure and basic guidelines to create, manage + and distribute bioinformatics containers with a special focus on + omics technologies. These containers can be integrated into more + comprehensive bioinformatics pipelines and different + architectures (local desktop, cloud environments or HPC + clusters). Availability and Implementation The software is freely + available at github.com/BioContainers/.", + month = aug, + year = 2017, + url = "https://academic.oup.com/bioinformatics/article-pdf/33/16/2580/49041124/bioinformatics_33_16_2580.pdf", + doi = "10.1093/bioinformatics/btx192", + issn = "1367-4803,1367-4811", + language = "en" +} + +@ARTICLE{Janak2026-ek, + title = "When probiotics turn deadly: a case of Lacticaseibacillus + rhamnosus sepsis in a burn patient", + author = "Janák, David and Bakalář, Bohumil and Fridrichová, Marta and + Zwinsova, Barbora and Zajíček, Robert and Lipový, Břetislav and + Borilova Linhartova, Petra", + journal = "Folia microbiologica", + publisher = "Springer Science and Business Media LLC", + pages = "1--6", + abstract = "We report the first comprehensive documented case of sepsis + caused by Lacticaseibacillus rhamnosus infection, likely + resulting from high-dose probiotic supplementation. This sepsis + occured in a 36-year-old woman with deep partial- and + full-thickness burns covering 25\% of total body surface area. + The use of probiotics in patients with organ dysfunction and in + immunocompromised individuals is on the rise. The immune system + and intestinal barriers are often compromised in patients with + burns, which may facilitate the translocation of probiotic + bacteria into bloodstream and lead to bacteremia; however, + isolation of lactobacilli in blood cultures is often disregarded + and considered an artefact caused by contamination. This case + highlights that probiotic-associated L. rhamnosus sepsis, + although rare, can occur in these patients and excessive + probiotic use should therefore be avoided in these individuals.", + month = mar, + year = 2026, + url = "http://dx.doi.org/10.1007/s12223-026-01445-x", + keywords = "Lacticaseibacillus rhamnosus ; Burns; Probiotics; Sepsis", + doi = "10.1007/s12223-026-01445-x", + pmid = 41820736, + issn = "0015-5632,1874-9356", + language = "en" +} + +@ARTICLE{Tighe2024-fq, + title = "Biomolecular analysis of arctic microorganisms capable of + psychrophilic growth on biodegradable and compostable plastic", + author = "Tighe, S W and Curd, E and Tracy, K M and Finstad, K H and + Vellone, D L and Hadley, S R and Dragon, J A", + journal = "Journal of Biomolecular Techniques", + volume = 35, + number = 4, + pages = "3fc1f5fe.601df0cc", + abstract = "As climate change continues to disrupt the polar regions of our + planet, a comprehensive understanding of both phenotypic and + genotypic characteristics of naturally occurring psychrophilic + microorganisms is needed, not only from a microbial profiling and + taxonomic aspect but also from an industrial potential standpoint. + Knowing and understanding the organisms that have the genetic + potential to break down environmental contaminants, such as + microplastics, is of great interest. In this research, the primary + focus was to isolate and characterize the psychrophilic + microorganisms from a snow field near Ilulissat, Greenland and use + a multi-omics approach to identify and characterize the + biodegradation potential against certain biodegradable plastics. + Bacterial stains isolated from Greenland were inoculated into + small individual bioreactor tubes containing a minimal salts media + combined with either polylactic acid or the proprietary Novamont + material used in compostable bags. After 4 weeks of incubations at + 6°C, turbidity (growth) was measured, and DNA and RNA were + extracted and sequenced to identify putative plastic-degrading + genes and biosynthetic gene clusters and determine if they are + actively expressed in culture conditions. Cultured bacteria + comprise 3 genera of bacteria: Pseudomonas, Duganella, and + Massilia. Culture tubes comprised Pseudomonas or Duganella + isolates alone or Pseudomonas in combination with either Duganella + or Massilia isolates. Genomes assembled from cultures contained + genes implicated in plastic degradation, and several contained the + complete pathway for octane oxidation. Cultures contained active + transcripts for most of the identified genes. Several biosynthetic + gene clusters were also identified, which may play a role in + biofilm formation or adaptation to psychrophilic growth. These + data are believed to be the first laboratory culture experiments + of psychrophilic microbial degradation of microplastics by + organisms isolated from polar regions.", + month = dec, + year = 2024, + url = "http://dx.doi.org/10.7171/3fc1f5fe.601df0cc", + doi = "10.7171/3fc1f5fe.601df0cc", + pmc = "PMC12051446", + pmid = 40330174, + issn = "1524-0215,1943-4731", + language = "en" +} + +@ARTICLE{Istanbullugil2026-fg, + title = "Koumiss microbiome: Investigation of the microbial composition + and functional potential of a unique beverage of fermented milk + produced at Kyrgyz mountains", + author = "İstanbullugil, Fatih Ramazan and Sanli, Kemal and Ozturk, Tarık + and Keskin, Birsen Cevher and Düyşöbayeva, Ayturgan and Risvanli, + Ali and Acaröz, Ulas and Acaröz, Damla Arslan and Salykov, Ruslan + and Sahin, Mitat", + journal = "Probiotics and Antimicrobial Proteins", + publisher = "Springer Science and Business Media LLC", + volume = 18, + number = 1, + pages = "18--34", + abstract = "This study aims to investigate the microbial composition of + koumiss made via traditional methods in Kyrgyz mountain pastures. + We collected koumiss samples produced in plastic (P), wood (T), + and leather (D) containers at household settings. These samples + were subjected to shotgun metagenomic sequencing. As a result of + the metagenome analyses, we identified a diversity of bacteria, + yeasts, bacteriophages, and archaea in koumiss produced within + different containers. Koumiss' microbial community was + predominantly composed of lactic acid bacteria (LAB), + particularly Lactobacillus helveticus and Lactococcus lactis. + Additional LAB species such as Lactobacillus kefiranofaciens, + Lactococcus raffinolactis, Lactiplantibacillus plantarum, and + Lactococcus cremoris, as well as non-LAB taxa such as Kluyvera + intermedia, Raoultella planticola, and Hafnia alvei were also + identified as part of the koumiss microbiota. Nonetheless, the + opportunistic pathogen, Enterobacter hormaechei, was among the + detected species. The most abundant yeast species was identified + as Brettanomyces bruxellensis. Other yeast species involving + Monosporozyma unispora, Monosporozyma servazzii, and Yarrowia + lipolytica were also detected within the metagenome. Despite the + type of container material not significantly affecting the + microbial diversity, Bifidobacterium spp. and bacteriophages were + identified at higher levels in plastic containers. We detected + various antimicrobial resistance genes and gene clusters that + produce bioactive compounds within koumiss samples. This study + highlights koumiss' rich microbial composition and its potential + health impacts. It underscores the importance of effectively + utilizing metagenomic and bioinformatics methods for better + comprehension of the microbiota of koumiss.", + month = jan, + year = 2026, + url = "http://dx.doi.org/10.1007/s12602-025-10718-9", + keywords = "Fermented foods; Koumiss; Mare’s milk; Metagenome-assembled + genomes; Shotgun metagenomics", + doi = "10.1007/s12602-025-10718-9", + pmid = 40824425, + issn = "1867-1306,1867-1314", + language = "en" +} + +@ARTICLE{Liepa2026-sw, + title = "Urban wastewater metagenomics reveals the antibiotic resistance + gene distribution across Latvian municipalities", + author = "Liepa, Edgars and Ustinova, Maija and Gudra, Dita and Roga, Ance + and Kalnina, Ineta and Dejus, Brigita and Dejus, Sandis and + Strods, Martins and Tomsone, Laura Elīna and Kibilds, Juris and + Bartkevics, Vadims and Berzins, Aivars and Dumpis, Uga and Juhna, + Talis and Fridmanis, Davids", + journal = "Microorganisms", + publisher = "MDPI AG", + volume = 14, + number = 1, + pages = 145, + abstract = "Antimicrobial resistance (AMR) poses a global health threat, with + urban wastewater systems serving as key reservoirs for resistance + dissemination. This study aimed to investigate the relationships + among urban environments, bacterial communities, and AMR + patterns, and evaluate the specific municipal-scale drivers of + resistance gene distribution. Shotgun metagenomic analysis was + conducted on 45 wastewater samples collected from 15 + municipalities across Latvia to determine the composition of the + resistome and its correlation with local factors. The analysis + identified 417 distinct antibiotic resistance genes (ARGs) + belonging to 108 families, with geographic location serving as + the primary driver of ARG distribution, which explained 65.87\% + of community variation (p = 0.001). Local industrial factors + demonstrated significant effects, with food industry wastewater + significantly influencing both bacterial taxonomy and ARG + profiles (p < 0.05). While the presence of a regional hospital + did not shape the overall municipal resistome, + hospital-associated wastewater showed 19 overlapping ARGs, + including clinically critical carbapenemases. Municipal + wastewater systems function as geographically structured + reservoirs of AMR that are shaped by localized industrial and + healthcare outputs. These findings support wastewater-based AMR + surveillance as a valuable tool for tracking specific resistance + sources.", + month = jan, + year = 2026, + url = "http://dx.doi.org/10.3390/microorganisms14010145", + keywords = "Latvia; antibiotic resistance; metagenomics; resistome; + wastewater-based epidemiology (WBE)", + doi = "10.3390/microorganisms14010145", + pmc = "PMC12843809", + pmid = 41597664, + issn = "2076-2607,2076-2607", + language = "en" +} + +@ARTICLE{Shen2024-mg, + title = "{SeqKit2}: A Swiss army knife for sequence and alignment + processing", + author = "Shen, Wei and Sipos, Botond and Zhao, Liuyang", + journal = "iMeta", + publisher = "Wiley", + pages = "e191", + abstract = "AbstractIn the era of ubiquitous high‐throughput sequencing + studies, there is a growing need for analysis tools that are not + just performant but also comprehensive and user‐friendly enough + to cater to both novice and advanced users. This article + introduces SeqKit2, the next iteration of the widely used + sequence analysis tool SeqKit, featuring expanded functionality, + performance optimizations, and support for additional compression + methods. Retaining a pragmatic subcommand architecture, SeqKit2 + represents substantial enhancement through the inclusion of 19 + additional subcommands, expanding its overall repertoire to a + total of 38 in eight categories. The new subcommands add + functionality such as amplicon processing and robust, + error‐tolerant parsing of sequence records. In addition, three + subcommands designed for real‐time analysis are added for + periodic monitoring of properties of FASTQ and Binary + Alignment/Map alignment records and real‐time streaming from + multiple sequence files. The performance of SeqKit2 is + benchmarked against the old version of SeqKit, Bioawk, Seqtk, and + SeqFu tools. SeqKit2 consistently outperforms its predecessor, + albeit with marginally higher memory usage, while maintaining + competitive runtimes against other tools. With its broad + functionality, proven usability, and ongoing development driven + by user feedback, we hope that bioinformaticians will find + SeqKit2 useful as a “Swiss army knife” of sequence and alignment + processing—equally adept at facilitating ad hoc analyses and + seamlessly integrating into larger pipelines.", + month = apr, + year = 2024, + url = "https://onlinelibrary.wiley.com/doi/abs/10.1002/imt2.191", + keywords = "performance optimization; real-time analysis; sequence + processing; usability; user-friendly", + doi = "10.1002/imt2.191", + issn = "2770-596X,2770-5986", + language = "en" +} + +@ARTICLE{Langer2025-th, + title = "Empowering bioinformatics communities with Nextflow and nf-core", + author = "Langer, Björn E and Amaral, Andreia and Baudement, Marie-Odile + and Bonath, Franziska and Charles, Mathieu and Chitneedi, Praveen + Krishna and Clark, Emily L and Di Tommaso, Paolo and Djebali, + Sarah and Ewels, Philip A and Eynard, Sonia and Fellows Yates, + James A and Fischer, Daniel and Floden, Evan W and Foissac, + Sylvain and Gabernet, Gisela and Garcia, Maxime U and Gillard, + Gareth and Gundappa, Manu Kumar and Guyomar, Cervin and Hakkaart, + Christopher and Hanssen, Friederike and Harrison, Peter W and + Hörtenhuber, Matthias and Kurylo, Cyril and Kühn, Christa and + Lagarrigue, Sandrine and Lallias, Delphine and Macqueen, Daniel J + and Miller, Edmund and Mir-Pedrol, Júlia and Moreira, Gabriel + Costa Monteiro and Nahnsen, Sven and Patel, Harshil and Peltzer, + Alexander and Pitel, Frederique and Ramayo-Caldas, Yuliaxis and + Ribeiro-Dantas, Marcel da Câmara and Rocha, Dominique and + Salavati, Mazdak and Sokolov, Alexey and Espinosa-Carrasco, Jose + and Notredame, Cedric and Community, The Nf-Core", + journal = "Genome Biology", + publisher = "Springer Science and Business Media LLC", + volume = 26, + number = 1, + pages = 228, + abstract = "Standardized analysis pipelines contribute to making data + bioinformatics research compliant with the paradigm of + Findability, Accessibility, Interoperability, and Reusability + (FAIR), and facilitate collaboration. Nextflow and Snakemake, two + popular command-line solutions, are increasingly adopted by + users, complementing GUI-based platforms such as Galaxy. We + report recent developments of the nf-core framework with the new + Nextflow Domain-Specific Language (DSL2). An extensive library of + modules and subworkflows enables research communities to adopt + common standards progressively, as resources and needs allow. We + present an overview of some of the research communities built + around nf-core and showcase its adoption by six EuroFAANG farmed + animal research consortia.", + month = jul, + year = 2025, + url = "http://dx.doi.org/10.1186/s13059-025-03673-9", + doi = "10.1186/s13059-025-03673-9", + pmc = "PMC12309086", + pmid = 40731283, + issn = "1474-7596,1474-760X", + language = "en" +} + +@article{ugarcina_perovic_argnorm_2025, + title = {{argNorm}: normalization of antibiotic resistance gene annotations to the Antibiotic Resistance Ontology ({ARO})}, + volume = {41}, + rights = {https://creativecommons.org/licenses/by/4.0/}, + issn = {1367-4811}, + url = {https://academic.oup.com/bioinformatics/article/doi/10.1093/bioinformatics/btaf173/8114632}, + doi = {10.1093/bioinformatics/btaf173}, + shorttitle = {{argNorm}}, + abstract = {Abstract + + Summary + Currently available and frequently used tools for annotating antibiotic resistance genes ({ARGs}) in genomes and metagenomes provide results using inconsistent nomenclature. This makes the comparison of different {ARG} annotation outputs challenging. The comparability of {ARG} annotation outputs can be improved by mapping gene names and their categories to a common controlled vocabulary such as the Antibiotic Resistance Ontology ({ARO}). We developed {argNorm}, a command line tool and Python library, to normalize all detected genes across six {ARG} annotation tools (eight databases) to the {ARO}. {argNorm} also adds information to the outputs using the same {ARG} categorization so that they are comparable across tools. + + + Availability and implementation + {argNorm} is available as an open-source tool at: https://github.com/{BigDataBiology}/{argNorm}. It can also be downloaded as a {PyPI} package and is available on Bioconda and as an nf-core module.}, + pages = {btaf173}, + number = {5}, + journaltitle = {Bioinformatics}, + author = {Ugarcina Perovic, Svetlana and Ramji, Vedanth and Chong, Hui and Duan, Yiqian and Maguire, Finlay and Coelho, Luis Pedro}, + editor = {Robinson, Peter}, + urldate = {2026-08-20}, + date = {2025-05-06}, + langid = {english}, +} diff --git a/paper/paper.md b/paper/paper.md index 280ddad4..d3711005 100644 --- a/paper/paper.md +++ b/paper/paper.md @@ -90,139 +90,133 @@ affiliations: index: 13 date: 14 April 2026 bibliography: paper.bib +header-includes: + - \usepackage{rotating} + - \usepackage{booktabs} --- # Summary -Genome-mining of bacterial DNA fosters the discovery of antimicrobial resistance-related genes as well as genes required for the biosynthesis of low molecular weight natural products or specialised metabolites. -Despite the availability of many bioinformatic tools to identify such functional genes, screening of genomic features remains inefficient due to heterogeneous computational platforms, accessibility, scalability, and inconsistent reporting and formatting of the results. -Here, we present nf-core/funcscan, an open source bioinformatics pipeline for the screening of microbial functional features from assembled contigs or genomes. -The pipeline currently integrates 13 tools to simultaneously predict antimicrobial peptides, antibiotic resistance genes, biosynthetic gene clusters, and taxonomic classification from partial or full genomes. -It also introduces standardised and aggregated output file reports across all tools, enabling the rapid evaluation, visualisation, and interpretation of results. -Written in the Nextflow workflow language, it is straightforward to install, portable across platforms ranging from personal laptops to high-performance computing clusters, and fully reproducible via the use of software containers. +Genome-mining of bacterial DNA enables the discovery of antimicrobial resistance-related genes, genes required for the biosynthesis of low molecular weight natural products, and other specialised metabolites. +However, execution of the multiple bioinformatic tools used in screening analyses remains inefficient due to heterogenous software interfaces, reporting, and formatting of the output files of similar tools, which limits scalability of such analyses. + +nf-core/funcscan is a portable and reproducible open source Nextflow bioinformatics pipeline for the screening of microbial functional features from assembled contigs or genomes. +The pipeline executes up to 13 tools to simultaneously identify antimicrobial peptides, antibiotic resistance genes, biosynthetic gene clusters, carbohydrate-activate enzymes, and performs taxonomic classification of partial or full genomes. +To facilitate efficient results comparison and evaluation, it supports cross-tool output file standardisation and aggregation. # Statement of need -The emergence and spread of multidrug resistant microbial pathogens poses a serious threat to global health [@murray_global_2022; @world_health_organization_global_2022]. -Traditionally, most anti-infective drugs have been derived from bacterially produced low molecular weight natural products. -To ensure self-resistance against antimicrobial agents, the producing bacteria typically exhibit resistance mechanisms. -As a consequence, the evolution of antimicrobials and the corresponding resistance mechanisms are strongly correlated. -Although antibiotic resistance is tightly linked to self-protection of the producing organisms, the recent excessive use of antibiotics and lack of global surveillance both in healthcare and agriculture has led to an explosion of multidrug resistant bacteria [@ventola_antibiotic_2015; @perry_prehistory_2016; @rascovan_exploring_2016]. -Over the past few decades, the spread of antibiotic resistance genes (ARGs) and pathogenic bacteria carrying them has grown to a major threat to human health. -Identifying new antibiotic agents from novel sources in combination with antibiotic resistance mechanisms and ARG evolution has the potential to aid in the development of new antibiotics. - -Due to this pressing problem, a large suite of different tools has been developed for the rapid identification of different functional gene types from sequencing data. -These tools use different search algorithms and databases (e.g. deepBGC: machine-learning [@hannigan_deep_2019], antiSMASH: rule-based [@blin_antismash_2025]) for the prediction of different types of microbial metabolites. -To maximise the potential of detecting important functional genes, researchers often need to use multiple approaches to ensure maximum detection sensitivity during screening. -Since these tools are often developed as stand-alone tools with specific databases they have to be executed separately. -This impedes scalability due to inefficiency and additionally poses an increased risk of lowering reproducibility when executed manually. -While some tools are available as software containers (e.g. via Docker, Singularity), thus helping reproducibility of results, they often require a series of steps to prepare input data and manually store and filter results. -Additionally, stand-alone tools have their own unique output formats, making cross-comparison of the results between different tools nontrivial, and often results in manual processing and inspection - again further restricting scalability. - -Previous efforts to scale up the predictive power of different tools for functional gene prediction include pipelines such as mettannotator [@gurbich_mettannotator_2025], bacannot [@almeida_scalable_2023], SqueezeMeta [@tamames_squeezemeta_2019], MetaErg [@dong_integrated_2019], METABOLIC [@zhou_metabolic_2022], HT-ARGfinder[@das_ht-argfinder_2022], ARGs-OAP [@yin_args-oap_2022], PathoFact [@de_nies_pathofact_2021], and antiSMASH. -However, to our knowledge, no pipeline has been created that allows for the identification and prediction of antimicrobial peptide (AMP) genes, ARGs, biosynthetic gene clusters (BGCs), carbohydrate-active enzymes (CAZymes), and CAZyme gene clusters (CGCs) simultaneously from multiple samples in a harmonised manner. -Additionally, extensive command-line knowledge and manual installation of software dependencies are required to run many of these existing pipelines. -This effectively precludes their use by biochemists, biomolecular scientists, and biologists who typically have limited computational training. - -Here, we present nf-core/funcscan, a Nextflow [@di_tommaso_nextflow_2017] pipeline following nf-core [@ewels_nf-core_2020] best practices for the simultaneous screening of multiple functional and biosynthetic components from assembled microbial contiguous sequences (contigs). -The pipeline predicts ARGs, BGCs, AMP-encoding genes, CAZymes, CGCs, and provides taxonomic information of the producing organisms from (meta)genomic sequences parallel in a portable, reproducible, and scalable manner. -This allows researchers to obtain a holistic view on the genomic context of identified genes for downstream analyses in the context of antimicrobial resistance. +Researchers often use multiple tools to ensure maximum detection sensitivity during genomic screening for potential gene candidates, as each tool uses different search algorithms and microbial metabolite databases. +However, heterogenous installation, inputs, and execution interfaces of these stand-alone tools impedes scalability, and decreases reproducibility due to user-error when executed manually. +Additionally, each tool often has its own unique output formats, making cross-comparison of results between tools and databases non-trivial, and again requiring inefficient manual postprocessing and inspection. + +This necessity for manual execution and postprocessing of heterogenous outputs impacts the discovery of new drugs. +For example, antibiotics are typically derived from naturally evolved, bacterially-produced, low molecular weight natural products, and the discovery rate of novel molecules has seen recent plateauing. +In combination with an explosion in the evolution of multidrug resistant bacteria [@ventola_antibiotic_2015; @perry_prehistory_2016; @rascovan_exploring_2016], and a lack of global surveillance both in healthcare and agriculture, this is contributing to a major threat to global health [@murray_global_2022; @world_health_organization_global_2022]. +Therefore high-throughput and scalable approaches are needed to allow the rapid identification of metabolites from novel sources, as well as live monitoring of the spread of antibiotic resistance within microbial populations. + +Here, we present nf-core/funcscan, a Nextflow [@di_tommaso_nextflow_2017] pipeline following nf-core best practices [@ewels_nf-core_2020;@Langer2025-th] for the automated and in-parallel screening of different functional gene groups with multiple tools and databases. +The pipeline currently supports detection of antimicrobial peptide (AMPs) genes, antimicrobial resistance genes (ARGs), biosynthetic gene clusters (BGCs), and carbohydrate-active enzyme gene clusters (CGCs). # State of the field -The continuing decrease in sequencing costs and the subsequent increase in available sequenced prokaryotic genomes and metagenomes has gone hand-in-hand with the development of numerous bioinformatics tools to predict gene functions. -Several pipelines have been developed to chain single-purpose tools together to provide a more comprehensive context. - -Pipelines with similar functionality to nf-core/funcscan include the pipeline mettannotator. -This pipeline meets the criteria of scalability and reproducibility on the same level as nf-core/funcscan, due to its similar implementation in Nextflow and in most parts also based on the nf-core pipeline template. -While focused on somewhat different gene types (e.g. snRNA, mobilome), shared features include ARG and BGC prediction as well as aggregation of results. -In contrast, nf-core/funcscan provides additional AMP screening and the integration of taxonomic classifications for all genes to provide additional ecological context around predicted genes. -Regarding pipeline stability and reliability, nf-core/funcscan is the only pipeline to implement comprehensive unit tests on both module and pipeline level, using the nf-test [@forer_improving_2024] framework (Table \ref{tab:pipelines}). - -| Feature | funcscan | mettannotator | bacannot | HT-ARGfinder | PathoFact | SqueezeMeta | MetaERG | ARGs-OAP | -| --------------------------------------- | -------- | ------------- | -------- | ------------ | --------- | ----------- | ------- | -------- | -| ARG screening | + | + | + | + | + | (+) | (+) | + | -| AMP screening | + | − | − | − | − | (+) | (+) | − | -| BGC screening | + | + | − | − | − | (−) | (−) | − | -| CAZyme screening | + | + | − | − | − | − | − | − | -| Taxonomic assignment of contigs | + | − | − | (−) | (−) | + | + | − | -| Results summary | + | + | + | (+) | (+) | + | + | − | -| Container support (Docker, Singularity) | + | + | + | − | − | − | + | (−) | -| Modularity | + | + | + | − | + | (+) | − | − | -| One-click installation | + | + | + | − | − | (−) | − | − | -| Local installation possible | + | + | + | + | + | + | + | − | -| Web-based execution possible | (+) | (+) | (+) | − | − | − | − | − | -| Software reviewing | + | + | − | − | − | − | − | − | -| Automated unit tests | + | + | (−) | (−) | (−) | − | − | − | -| License | MIT | Apache-2.0 | GPL-3.0 | None | GPL-3.0 | GPL-3.0 | AFL | AFL | - -: Comparison of nf-core/funcscan with other related pipelines for ARG, AMP, and BGC discovery. Parentheses indicate either unspecific gene screening or partly fulfilled criteria. \label{tab:pipelines} +Previous efforts to scale up the predictive power of different tools for functional gene prediction include mettannotator [@gurbich_mettannotator_2025], bacannot [@almeida_scalable_2023], PathoFact [@de_nies_pathofact_2021] SqueezeMeta [@tamames_squeezemeta_2019], MetaErg [@dong_integrated_2019], and ARGs-OAP [@yin_args-oap_2022] (Table 1). +However, to our knowledge, these are typically focused on singular gene categories or groups (e.g. antimicrobial resistance), aim to be 'end-to-end' pipelines including read preprocessing and assembly, or do not provide important contextual information about the potential hits (such as taxonomic information). + +\begin{sidewaystable} +\centering +\caption{Comparison of nf-core/funcscan with other related pipelines for ARG, AMP, and BGC discovery. Parentheses indicate either unspecific gene screening or partly fulfilled criteria.} +\label{tab:pipelines} +\begin{tabular}{l|l|l|l|l|l|l|l} +\toprule +Feature & funcscan & mettannotator & bacannot & PathoFact & SqueezeMeta & MetaERG & ARGs-OAP \\ +\hline +ARG screening & + & + & + & + & (+) & (+) & + \\ +AMP screening & + & − & − & − & (+) & (+) & − \\ +BGC screening & + & + & − & − & (−) & (−) & − \\ +CAZyme screening & + & + & − & − & − & − & − \\ +Taxonomic assignment of contigs & + & − & − & (−) & + & + & − \\ +Results summary & + & + & + & (+) & + & + & − \\ +Container support (Docker, Singularity) & + & + & + & − & − & + & (−) \\ +Modularity & + & + & + & + & (+) & − & − \\ +One-click installation & + & + & + & − & (−) & − & − \\ +Local installation possible & + & + & + & + & + & + & − \\ +Web-based execution possible & (+) & (+) & (+) & − & − & − & − \\ +Software reviewing & + & + & − & − & − & − & − \\ +Automated unit tests & + & + & (−) & (−) & − & − & − \\ +License & MIT & Apache-2.0 & GPL-3.0 & GPL-3.0 & GPL-3.0 & AFL & AFL \\ +\bottomrule +\end{tabular} +\end{sidewaystable} + +Extensive command-line knowledge and manual installation of software dependencies are also often required to run many of these existing pipelines. +This can preclude use by biochemists, biologists, etc. who typically have limited computational training. +In contrast, nf-core/funcscan aims to reduce complexity by screening from already assembled sequences, and end on aggregation of the screening results, through multiple methods for execution. + +The main factors that distinguish nf-core/funcscan from the most similar pipeline, metannotator, are: support for metagenomic assembly input (rather than just genomes); automated taxonomic classification of contigs; more ARG tools; standardised prediction output; and confirmed executable on other infrastructure than HPCs. + + + # Workflow overview -nf-core/funcscan simultaneously predicts AMPs, ARGs, BGCs as well as CGCs from partial or full (meta)genomic sequences. -In addition, the bacterial taxonomy of input sequences is determined and standardised summaries of all tool outputs are provided (Fig. \ref{fig:workflow}). +nf-core/funcscan simultaneously predicts AMPs, ARGs, BGCs as well as CGCs from input partial or full (meta)genomic sequences. +Output files from the tools of each of the categories are aggregated and standardised for easy cross-comparison (Fig. \ref{fig:workflow}). ![Workflow overview of nf-core/funcscan. (1), genomic sequences are prepared and annotated with one of four open reading frame annotation tools. Two additional classification workflows can be used to classify contigs taxonomically (light gray) or obtain additional protein domain information (dark gray). -(2), depending on which workflows are selected by the user, the biosynthetic gene cluster (BGC, purple), antimicrobial peptide genes (AMP, orange), antibiotic resistance genes (ARG, yellow), or carbohydrate-active enzymes (CAZyme, blue) workflows with their customisable parameters are executed. +(2), depending on user-choice, the biosynthetic gene cluster (BGC, purple), antimicrobial peptide genes (AMP, orange), antibiotic resistance genes (ARG, yellow), or carbohydrate-active enzymes (CAZyme, blue) workflows with their customisable parameters are executed. (3), the results of all tools for each gene category are aggregated and saved in a human- and machine-readable tabular format.\label{fig:workflow}](figure1.png) ## Input preprocessing and open reading frame annotation -The pipeline processes a two-, four-, or five-column tabular sample-sheet as input (comma-separated, CSV format). -Sample names and paths to the respective nucleotide FASTA files containing (meta)genomic contigs or genomes to be screened are required (two-column sample-sheet). -Optionally, pre-annotated sequence files can be supplied to the pipeline in the four-column sample-sheet variant with open reading frame amino acid sequences in FASTA format, and their respective annotations in GBK format. -Additionally, GFF annotation files can be provided in a fifth column. -During preprocessing, all gzipped sequence files are decompressed, and, for running the BGC subworkflow, short contigs are removed by SeqKit (default: contigs shorter than 3,000 bp) to reduce runtime by removing too-short sequences that produce no biologically meaningful results. -Open reading frames are predicted from the preprocessed sequences by one of four prokaryotic annotation tools (Bakta, Prodigal, Prokka, and Pyrodigal). -If annotated sequence files as described above are provided in the sample-sheet, this step is skipped. +The pipeline takes a two- to five-column tabular sample-sheet as input (comma-separated, CSV format). +This sample-sheet contains sample names, paths to (meta)genomic FASTA files and optionally pre-generated amino-acid FASTA, GFF, or GBK format annotation files. -Various tools of nf-core/funcscan rely on databases and reference files to operate. -The pipeline offers the functionality to download these databases automatically for the user, which can then be stored and reused in future pipeline runs to minimise pipeline runtime, network traffic, and possible download limits. -The database download is applicable for AMPcombi [@herbst_actifensin_2025], AMRFinderPlus [@feldgarden_amrfinderplus_2021; @feldgarden_validating_2019], antiSMASH, Bakta [@schwengers_bakta_2021], BiG-SLiCE [@kautsar_big-slice_2021], DeepARG [@arango-argoty_deeparg_2018], DeepBGC, InterProScan [@jones_interproscan_2014], MMSeqs2 [@mirdita_fast_2021], and RGI [@alcock_card_2023]. +Preprocessing steps reduce runtime by removing too-short sequences with Seqkit [@Shen2024-mg], when they may produce no biologically meaningful results. +Open reading frames are optionally predicted from the preprocessed sequences by one of four prokaryotic annotation tools: Bakta [@schwengers_bakta_2021], Prodigal [@Hyatt2010-yv], Prokka [@Seemann2014-ee], and Pyrodigal [@Larralde2022-uu]. -## Gene prediction and taxonomic classification +When required, the pipeline downloads required screening-tool databases automatically for the user, and makes them available for future pipeline runs to minimise runtime and network traffic. -In a second step, users can choose to scan genomic sequences in parallel with four dedicated workflows for AMPs, ARGs, BGCs, and CAZymes, applying up to currently a total of 13 gene identification tools: +## Gene screening and taxonomic classification -- **ARG subworkflow**: ABRicate [@torsten_seemann_abricate_2020], AMRFinderPlus, DeepARG, fARGene [@berglund_identification_2019], RGI -- **BGC subworkflow**: antiSMASH, DeepBGC, GECCO [@carroll_accurate_2021], hmmsearch [@eddy_accelerated_2011] -- **AMP subworkflow**: ampir [@fingerhut_ampir_2021], AMPlify [@li_models_2023, @li_amplify_2022], hmmsearch, Macrel [@santos-junior_macrel_2020] -- **CAZyme subworkflow**: run_dbCAN [@zheng_dbcan3_2023] +Users choose to scan genomic sequences in parallel with up-to four dedicated subworkflows for AMPs, ARGs, BGCs, and CAZymes. +Up to a total of 13 gene identification tools can be applied: -In an additional optional parallel screening step, all input sequences can be taxonomically classified by MMSeqs2 to determine likely source hosts of each functional hit. -Characterising the taxonomic origin of metagenomic contigs can provide users information about potentially suitable hosts for downstream experiments, e.g. heterologous expression systems [@porse_biochemical_2018]. -The taxonomic classification supports a variety of reference databases (e.g. GTDB, UniProt, UniRef, NR, Kalamari) to suit different user requirements. -Optionally, protein domains and families can be further annotated by InterProScan. +- **ARGs**: ABRicate [@torsten_seemann_abricate_2020], AMRFinderPlus [@feldgarden_amrfinderplus_2021;@feldgarden_validating_2019], DeepARG [@arango-argoty_deeparg_2018], fARGene [@berglund_identification_2019], RGI [@alcock_card_2023] +- **BGCs**: antiSMASH [@blin_antismash_2025], DeepBGC [@hannigan_deep_2019], GECCO [@carroll_accurate_2021], hmmsearch [@eddy_accelerated_2011] +- **AMPs**: ampir [@fingerhut_ampir_2021], AMPlify [@li_models_2023, @li_amplify_2022], hmmsearch, Macrel [@santos-junior_macrel_2020] +- **CAZymes**: run_dbCAN [@zheng_dbcan3_2023] -Reasonable default parameters for commonly tuned parameters of the screening tools are set by the pipeline, and can be adjusted by the user by dedicated command-line arguments or via a Nextflow parameter file. +To provide users information about potentially suitable hosts for downstream experiments, e.g. heterologous expression systems [@porse_biochemical_2018], an additional optional parallel workflow can taxonomically classify input contigs with MMSeqs2 [@mirdita_fast_2021]. +Optionally, generic protein domains and families can be further annotated with InterProScan [@jones_interproscan_2014]. + +Pipeline parameters can be adjusted by userwritten- or nf-core GUI ([https://nf-co.re/launch](https://nf-co.re/launch))-generated Nextflow parameter files, or command-line arguments. ## Aggregation of screening results -All screening tools of nf-core/funcscan have heterogeneous output formats and label their respective gene predictions differently. -nf-core/funcscan aggregates the output of all gene and taxonomic screening tools in each executed subworkflow into single human- and machine-readable tables in CSV format per gene type using dedicated tools. -For the summary of ARGs, we have used the existing hAMRonization [@mendes_hamronization_2024] software. -For AMPs, AMPcombi parses and filters the results of AMP prediction tools, summarises them into single tables, and aligns the AMP hits against a reference AMP database for deeper functional classification. -We wrote a custom script 'comBGC' for aggregating and standardising the output of the BGC tools. -These summaries are finally complemented with results from the optional taxonomic classification workflow. +nf-core/funcscan integrates dedicated tools to aggregate and standardise heterogenous output formats of multiple screening tools into a single human- and machine-readable tables in CSV format per gene type. +nf-core uses hAMRonization [@mendes_hamronization_2024] for ARGs, AMPcombi [@herbst_actifensin_2025] for AMPs, and a custom script 'comBGC' for BGC tool output aggregation. +These summaries are finally optionally complemented with results from the taxonomic classification workflow. + +Building on the aggragation of screening results, two optional downstream analyses can be executed for the ARG and BGC workflows. First, the ARG summary provided by hAMRonization can be further normalised and mapped to the antibiotic resistance ontology (ARO) by argNorm [@ugarcina_perovic_argnorm_2025]. This enhances ARG annotation by categorising drugs that ARGs confer resistance to. Secondly, BGCs predicted by antiSMASH and GECCO can be clustered into Gene Cluster Families (GCFs) by BiG-SLiCE [@kautsar_big-slice_2021] to enable comparative analysis of biosynthetic diversity across samples. ## Reproducibility and scalability -All nf-core pipelines utilise software environments (Conda) or containers (Docker, Singularity) for each integrated tool. -This provides the advantage of isolating the dependencies of all workflows from each other and rendering pipeline execution highly reproducible, portable, and platform-independent. -Thus, the pipeline itself is easy to install as it has only few minimum dependencies (Nextflow itself, and one of Docker, Singularity, Podman, Shifter, Charliecloud, and Conda). -The configuration of the pipeline to the underlying computing system requires knowledge of its software environment and hardware resources. -To facilitate configuration and further portability, nf-core provides already centralised configurations for more than 150 institutional computational infrastructures (e.g. HPCs) via the central nf-core/configs repository ([https://nf-co.re/configs](https://nf-co.re/configs)). -The performance of each pipeline run (including software versions of all applied tools, memory, and CPU usage) is summarised in HTML reports for all steps of all subworkflows for users to estimate future runtime and/or computational resources. +All nf-core pipelines utilise software environments [from the Bioconda project, @Gruning2018-vr] or containers [e.g. Docker, Singularity, primarily from the Biocontainers project, @Da_Veiga_Leprevost2017-gl] for each integrated tool. +This provides the advantage of isolating the dependencies of all workflows from each other, thereby reducing installation problems. +The pipeline is thus easy to install with few minimum dependencies - Nextflow itself, and one of Nextflow-supported container/software environment management systems. +For further portability, nf-core provides integrated configurations for more than 150 institutional computational infrastructure (e.g. HPCs) via nf-core/configs ([https://nf-co.re/configs](https://nf-co.re/configs)). +Users on these infrastructure thus can run the pipelines with no-set up via a single parameter. # Research impact statement -nf-core/funcscan has developed an active user community of scientific users and developers who continuously contribute ideas, bug reports and code via issues and pull requests on GitHub. -The pipeline is actively being used in research (https://www.mdpi.com/2076-2607/14/1/145, https://link.springer.com/article/10.1007/s12602-025-10718-9, https://pmc.ncbi.nlm.nih.gov/articles/PMC12051446/, https://link.springer.com/article/10.1007/s12223-026-01445-x). -Additionally, the pipeline received a contribution of a whole new workflow (CAZyme screening) by new community members outside of the original developers. -Discussions of pipeline as well as research domain related topics happen on the open-to-join nf-core workspace on the Slack platform. This illustrates the public interest and proactive efforts from scientific users to use, maintain, and improve the pipeline functionalities. +nf-core/funcscan has an active user community of scientific users and developers on the nf-core Slack and GitHub ([https://nf-co.re/join](https://nf-co.re/join)). +For example, the pipeline received the contribution of the CAZyme screening from community members outside of the original developers. +User discussions and support on the pipeline and on related research topics occur on the nf-core Slack workspace. +This illustrates the public interest and proactive efforts from scientific users to use, maintain, and improve the pipeline. +The pipeline is also already actively being used in research [e.g., @Tighe2024-fq, @Janak2026-ek, @Istanbullugil2026-fg, @Liepa2026-sw]. # AI usage disclosure @@ -237,6 +231,7 @@ J.F. received a fellowship from the International Leibniz Research School (under This project was funded by grants from the Werner Siemens Foundation (Paleobiotechnology to C.W. and P.S.) and the Deutsche Forschungsgemeinschaft (DFG, German Research Foundation, under Germany’s Excellence Strategy – EXC 2051 – Project-ID 390713860 to C.W. and P.S.). J.A.F.Y and C.W. were funded by the Deutsche Forschungsgemeinschaft (DFG, German Research Foundation) – project number 460129525 (NFDI4Microbiota, FlexFund project EnterArchaeo). +J.A.F.Y and C.W. were supported by the Max Planck Society. This work was supported by the de.NBI Cloud within the German Network for Bioinformatics Infrastructure (de.NBI) and ELIXIR-DE (Forschungszentrum Jülich and W-de.NBI-001, W-de.NBI-004, W-de.NBI-008, W-de.NBI-010, W-de.NBI-013, W-de.NBI-014, W-de.NBI-016, W-de.NBI-022). # References