
<!DOCTYPE article
  PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.4 20241031//EN" "JATS-archivearticle1-4-mathml3.dtd">
<article article-type="research-article" xml:lang="en" dtd-version="1.4"><front><journal-meta><journal-id journal-id-type="nlm-ta">Gigascience</journal-id><journal-id journal-id-type="iso-abbrev">Gigascience</journal-id><journal-id journal-id-type="pmc-domain-id">2056</journal-id><journal-id journal-id-type="pmc-domain">gigasci</journal-id><journal-id journal-id-type="nlm-id">101596872</journal-id><journal-id journal-id-type="publisher-id">gigascience</journal-id><journal-title-group><journal-title>GigaScience</journal-title></journal-title-group><issn pub-type="epub">2047-217X</issn><?publisher_abbrev oup?><publisher><publisher-name>Oxford University Press</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="pmcid">PMC6022608</article-id><article-id pub-id-type="pmcid-ver">PMC6022608.1</article-id><article-id pub-id-type="pmcaid">6022608</article-id><article-id pub-id-type="pmcaiid">6022608</article-id><article-id pub-id-type="pmid">29893851</article-id><article-id pub-id-type="doi">10.1093/gigascience/giy069</article-id><article-id pub-id-type="publisher-id">giy069</article-id><article-version article-version-type="pmc-version">1</article-version><article-categories><subj-group subj-group-type="heading"><subject>Technical Note</subject></subj-group></article-categories><title-group><article-title>AMBER: Assessment of Metagenome BinnERs</article-title></title-group><contrib-group><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-0901-3815</contrib-id><name name-style="western"><surname>Meyer</surname><given-names initials="F">Fernando</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Hofmann</surname><given-names initials="P">Peter</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Belmann</surname><given-names initials="P">Peter</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Garrido-Oter</surname><given-names initials="R">Ruben</given-names></name><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="aff" rid="aff6">6</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Fritz</surname><given-names initials="A">Adrian</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0002-4405-3847</contrib-id><name name-style="western"><surname>Sczyrba</surname><given-names initials="A">Alexander</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref></contrib><contrib contrib-type="author"><contrib-id contrib-id-type="orcid" authenticated="false">http://orcid.org/0000-0003-2370-3430</contrib-id><name name-style="western"><surname>McHardy</surname><given-names initials="AC">Alice C</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="corresp" rid="cor1"/></contrib></contrib-group><aff id="aff1"><label>1</label>Department of Computational Biology of Infection Research, Helmholtz Centre for Infection Research, Braunschweig, Germany</aff><aff id="aff2"><label>2</label>Braunschweig Integrated Centre of Systems Biology, Braunschweig, Germany</aff><aff id="aff3"><label>3</label>Faculty of Technology, Bielefeld University, Bielefeld, Germany</aff><aff id="aff4"><label>4</label>Center for Biotechnology, Bielefeld University, Bielefeld, Germany</aff><aff id="aff5"><label>5</label>Department of Plant Microbe Interactions, Max Planck Institute for Plant Breeding Research, Cologne, Germany</aff><aff id="aff6"><label>6</label>Cluster of Excellence on Plant Sciences</aff><author-notes><corresp id="cor1"><bold>Correspondence address</bold>. Alice C. McHardy, Braunschweig Integrated Centre of Systems Biology (BRICS), Department of Computational Biology of Infection Research, Rebenring 56, 38106 Braunschweig, Germany. E-mail: <email>Alice.McHardy@helmholtz-hzi.de</email></corresp></author-notes><pub-date pub-type="collection"><month>6</month><year>2018</year></pub-date><pub-date pub-type="epub" iso-8601-date="2018-06-08"><day>08</day><month>6</month><year>2018</year></pub-date><volume>7</volume><issue>6</issue><issue-id pub-id-type="pmc-issue-id">315502</issue-id><elocation-id>giy069</elocation-id><history><date date-type="received"><day>12</day><month>1</month><year>2018</year></date><date date-type="rev-recd"><day>27</day><month>4</month><year>2018</year></date><date date-type="accepted"><day>01</day><month>6</month><year>2018</year></date></history><pub-history><event event-type="pmc-release"><date><day>08</day><month>06</month><year>2018</year></date></event><event event-type="pmc-live"><date><day>10</day><month>07</month><year>2018</year></date></event><event event-type="pmc-last-change"><date iso-8601-date="2023-09-26 00:25:09.917"><day>26</day><month>09</month><year>2023</year></date></event></pub-history><permissions><copyright-statement>© The Author(s) 2018. Published by Oxford University Press.</copyright-statement><copyright-year>2018</copyright-year><license xmlns:xlink="http://www.w3.org/1999/xlink" license-type="cc-by" xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref xmlns:ali="http://www.niso.org/schemas/ali/1.0/" specific-use="textmining" content-type="ccbylicense">https://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This is an Open Access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">http://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted reuse, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p></license></permissions><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" content-type="pmc-pdf" xlink:href="giy069.pdf"><?pdf-name giy069.pdf?><?pdf-size 680904?><?pdf-md5 1a7c9c6d6bfd52d9948bbd504ade3b43?><?pdf-image-server-status NEVER_LOAD?><?pdf-cloudpmc-urn urn:app:994d/6022608/1a7c9c6d6bfd/giy069.pdf?></self-uri><self-uri xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="giy069.pdf"/><abstract><title>Abstract</title><p>Reconstructing the genomes of microbial community members is key to the interpretation of shotgun metagenome samples. Genome binning programs deconvolute reads or assembled contigs of such samples into individual bins. However, assessing their quality is difficult due to the lack of evaluation software and standardized metrics. Here, we present Assessment of Metagenome BinnERs (AMBER), an evaluation package for the comparative assessment of genome reconstructions from metagenome benchmark datasets. It calculates the performance metrics and comparative visualizations used in the first benchmarking challenge of the initiative for the Critical Assessment of Metagenome Interpretation (CAMI). As an application, we show the outputs of AMBER for 11 binning programs on two CAMI benchmark datasets. AMBER is implemented in Python and available under the Apache 2.0 license on GitHub.</p></abstract><kwd-group kwd-group-type="keywords"><kwd>binning</kwd><kwd>metagenomics</kwd><kwd>benchmarking</kwd><kwd>performance metrics</kwd><kwd>bioboxes</kwd></kwd-group><counts><page-count count="8"/></counts><custom-meta-group><custom-meta><meta-name>pmc-status-qastatus</meta-name><meta-value>0</meta-value></custom-meta><custom-meta><meta-name>pmc-status-live</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-status-embargo</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-status-released</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-open-access</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-olf</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-manuscript</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-legally-suppressed</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-pdf</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-has-supplement</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-pdf-only</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-suppress-copyright</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-real-version</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-is-scanned-article</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-preprint</meta-name><meta-value>no</meta-value></custom-meta><custom-meta><meta-name>pmc-prop-in-epmc</meta-name><meta-value>yes</meta-value></custom-meta><custom-meta><meta-name>pmc-license-ref</meta-name><meta-value>CC BY</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec sec-type="intro" id="sec1"><title>Introduction</title><p>Metagenomics allows studying microbial communities and their members by shotgun sequencing. Evolutionary divergence and the abundance of these members can vary widely, with genomes occasionally being very closely related to one another, representing strain-level diversity, or evolutionary far apart, whereas abundance can differ by several orders of magnitude. Genome binning software deconvolutes metagenomic reads or assembled sequences into bins representing genomes of the community members. A popular and performant approach in genome binning uses the covariation of read coverage and short <italic toggle="yes">k</italic>-mer composition of contigs with the same origin across co-assemblies of one or more related samples, though the presence of strain-level diversity substantially reduces bin quality [<xref rid="bib1" ref-type="bibr">1</xref>].</p><p>Benchmarking methods for binning and other tasks in metagenomics, such as assembly and profiling, are crucial for both users and method developers. The former need to determine the most suitable programs and parameterizations for particular applications and datasets, and the latter need to compare their novel or improved method with existing ones. When lacking evaluation software or standardized metrics, both need to individually invest considerable effort in assessing methods. The Critical Assessment of Metagenome Interpretation (CAMI) is a community-driven initiative aiming to tackle this problem by establishing evaluation standards and best practices, including the design of benchmark datasets and performance metrics [<xref rid="bib1" ref-type="bibr">1</xref>, <xref rid="bib2" ref-type="bibr">2</xref>]. Following community requirements and suggestions, the first CAMI challenge provided metagenome datasets of microbial communities with different organismal complexities, for which participants could submit their assembly, taxonomic and genomic binning, and taxonomic profiling results. These were subsequently evaluated, using metrics selected by the community [<xref rid="bib1" ref-type="bibr">1</xref>]. Here, we describe the software package Assessment of Metagenome BinnERs (AMBER) for the comparative assessment of genome binning reconstructions from metagenome benchmark datasets. It implements all metrics decided by the community to be most relevant for assessing the quality of genome reconstructions in the first CAMI challenge and is applicable to arbitrary benchmark datasets. AMBER automatically generates binning quality assessment outputs in flat files, as summary tables, rankings, and as visualizations in images and an interactive HTML page. It complements the popular CheckM software that assesses genome bin quality on real metagenome samples based on sets of single-copy marker genes [<xref rid="bib3" ref-type="bibr">3</xref>].</p></sec><sec sec-type="methods" id="sec2"><title>Methods</title><sec id="sec2-1"><title>Input</title><p>AMBER uses three types of files as input to assess binning quality for benchmark datasets: (1) a gold standard mapping of contigs or read IDs to underlying genomes of community members, (2) one or more files with predicted bin assignments for the sequences, and (3) a FASTA or FASTQ file with sequences. Benchmark metagenome sequence samples with a gold standard mapping can, e.g., be created with the CAMISIM metagenome simulator [<xref rid="bib4" ref-type="bibr">4</xref>, <xref rid="bib5" ref-type="bibr">5</xref>]. A gold standard mapping can also be obtained for sequences (reads or contigs), provided that reference genomes are available, by aligning the sequences to these genomes. Popular read aligners include Bowtie [<xref rid="bib6" ref-type="bibr">6</xref>] and BWA [<xref rid="bib7" ref-type="bibr">7</xref>]. MetaQUAST [<xref rid="bib8" ref-type="bibr">8</xref>] can also be used for contig alignment while it evaluates metagenome assemblies. High-confidence alignments can then be used as mappings of the sequences to the genomes. The input files (1) and (2) use the Bioboxes binning format [<xref rid="bib9" ref-type="bibr">9</xref>, <xref rid="bib10" ref-type="bibr">10</xref>]. AMBER also accepts individual FASTA files as bin assignments for each bin, as provided by MaxBin [<xref rid="bib11" ref-type="bibr">11</xref>]. These can be converted to the Bioboxes format. Example files are provided in the AMBER GitHub repository [<xref rid="bib12" ref-type="bibr">12</xref>].</p></sec><sec id="sec2-2"><title>Metrics and accompanying visualizations</title><p>AMBER uses the gold standard mapping to calculate a range of relevant metrics [<xref rid="bib1" ref-type="bibr">1</xref>] for one or more genome binnings of a given dataset. Below, we provide a more formal definition of all metrics than provided in [<xref rid="bib1" ref-type="bibr">1</xref>], together with an explanation of their biological meaning.</p><sec id="sec2-2-1"><title>Assessing the quality of bins</title><p>The purity and completeness, both ranging from 0 to 1, are commonly used measures for quantifying bin assignment quality, usually in combination [<xref rid="bib13" ref-type="bibr">13</xref>]. We provide formal definitions below. Since predicted genome bins have no label, e.g., a taxonomic one, the first step in calculating genome purity and completeness is to map each predicted genome bin to an underlying genome. For this, AMBER uses one of the following choices:
<list list-type="order"><list-item><p>A predicted genome bin is mapped to the most abundant genome in that bin in number of base pairs. More precisely, let <inline-formula><tex-math id="M1"><?equation-image-name M1.gif?><?equation-image-status READY?><?equation-image-md5 a6c1be96c1a315a0cfeea56cc209d7ce?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/a6c1be96c1a3/M1.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$X$\end{document}</tex-math></inline-formula> be the set of predicted genome bins and <inline-formula><tex-math id="M2"><?equation-image-name M2.gif?><?equation-image-status READY?><?equation-image-md5 dc5ac8241f183c98094cddec7e0999ea?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/dc5ac8241f18/M2.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$Y$\end{document}</tex-math></inline-formula> be the set of underlying genomes. We define a mapping of the predicted genome bin <inline-formula><tex-math id="M3"><?equation-image-name M3.gif?><?equation-image-status READY?><?equation-image-md5 23dfe7d23cf8cadd84873ce1cdb1b6bc?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/23dfe7d23cf8/M3.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$x \in X$\end{document}</tex-math></inline-formula> as<inline-formula><tex-math id="M4"><?equation-image-name M4.gif?><?equation-image-status READY?><?equation-image-md5 27ea51ef219fce59d365da9186a8cd0a?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/27ea51ef219f/M4.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ g( x )\ = \ y$\end{document}</tex-math></inline-formula>, such that genome <inline-formula><tex-math id="M5"><?equation-image-name M5.gif?><?equation-image-status READY?><?equation-image-md5 e5395764aae3a8a3b23943e6b2699b03?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e5395764aae3/M5.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$y$\end{document}</tex-math></inline-formula> maps to<inline-formula><tex-math id="M6"><?equation-image-name M6.gif?><?equation-image-status READY?><?equation-image-md5 d52520c2d23b37d5864a7d95ea8e86c8?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d52520c2d23b/M6.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ x$\end{document}</tex-math></inline-formula> and the overlap between <inline-formula><tex-math id="M7"><?equation-image-name M7.gif?><?equation-image-status READY?><?equation-image-md5 d52520c2d23b37d5864a7d95ea8e86c8?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d52520c2d23b/M7.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$x$\end{document}</tex-math></inline-formula> and <inline-formula><tex-math id="M8"><?equation-image-name M8.gif?><?equation-image-status READY?><?equation-image-md5 e5395764aae3a8a3b23943e6b2699b03?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e5395764aae3/M8.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$y$\end{document}</tex-math></inline-formula>, in base pairs, is maximal among all <inline-formula><tex-math id="M9"><?equation-image-name M9.gif?><?equation-image-status READY?><?equation-image-md5 28bb86d08344003d468ce6e802c83b66?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/28bb86d08344/M9.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$y \in Y$\end{document}</tex-math></inline-formula>, i.e.,
<disp-formula id="equ1"><label>(1)</label><tex-math id="M10"><?equation-image-name M10.gif?><?equation-image-status READY?><?equation-image-md5 69406c42b1f681fb6f4b39d25a36a233?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/69406c42b1f6/M10.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
g\left( x \right)\ = \begin{array}{@{}*{1}{c}@{}} {\arg {\rm{max}}}\\ {y \in Y} \end{array}{\rm{\ }}\left| {x\ \cap y} \right|.
\end{eqnarray*}
\end{document}</tex-math></disp-formula></p></list-item><list-item><p>A predicted genome bin is mapped to the genome whose largest fraction of base pairs has been assigned to the bin. In this case, we define a mapping <inline-formula><tex-math id="M11"><?equation-image-name M11.gif?><?equation-image-status READY?><?equation-image-md5 7a4b103db3600f7b2dcc53bded0630c2?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/7a4b103db360/M11.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$g'( x )\ = \ y$\end{document}</tex-math></inline-formula> as
<disp-formula id="equ2"><label>(2)</label><tex-math id="M12"><?equation-image-name M12.gif?><?equation-image-status READY?><?equation-image-md5 940d2876d96b43c2afe7c1b65d78022c?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/940d2876d96b/M12.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
g'\left( x \right)\ = \begin{array}{@{}*{1}{c}@{}} {\arg {\rm{max}}}\\ {y \in Y} \end{array}\ \frac{{\left| {x\ \cap \ y} \right|}}{{\left| y \right|}}.
\end{eqnarray*}
\end{document}</tex-math></disp-formula></p></list-item></list></p><p>If more than a genome is completely included in the bin, i.e., <inline-formula><tex-math id="M13"><?equation-image-name M13.gif?><?equation-image-status READY?><?equation-image-md5 80b83ad2d2371119f4f790fb223a350f?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/80b83ad2d237/M13.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$| {x\ \cap \ y} |/| y | = \ 1.0$\end{document}</tex-math></inline-formula> for more than a<inline-formula><tex-math id="M14"><?equation-image-name M14.gif?><?equation-image-status READY?><?equation-image-md5 9c91b722ea2905c74da4e81df663d44b?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/9c91b722ea29/M14.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ y \in Y$\end{document}</tex-math></inline-formula>, then the largest genome is mapped.</p><p>Using either option, each predicted genome bin is mapped to a single genome, but a genome can map to multiple bins or remain unmapped. Option 1 maps to each bin the genome that best represents the bin, since the majority of the base pairs in the bin belong to that genome. Option 2 maps to each bin the genome that best represents that genome, since most of the genome is contained in that specific bin. AMBER uses per default option 1. In the following, we use <inline-formula><tex-math id="M15"><?equation-image-name M15.gif?><?equation-image-status READY?><?equation-image-md5 b3a9f9178b4bec58fdbaf638ee5f1f02?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/b3a9f9178b4b/M15.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$g^*$\end{document}</tex-math></inline-formula> to denote one of these mappings for simplicity whenever possible.</p><p>The <bold>purity <inline-formula><tex-math id="M16"><?equation-image-name M16.gif?><?equation-image-status READY?><?equation-image-md5 904292cc846e1fa3857a5357ab0232b0?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/904292cc846e/M16.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\boldsymbol{p}}$\end{document}</tex-math></inline-formula></bold>, also known as precision or specificity, quantifies the quality of genome bin predictions in terms of how trustworthy those assignments are. Specifically, the purity represents the ratio of base pairs originating from the mapped genome to all bin base pairs. For every predicted genome bin<inline-formula><tex-math id="M17"><?equation-image-name M17.gif?><?equation-image-status READY?><?equation-image-md5 d52520c2d23b37d5864a7d95ea8e86c8?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d52520c2d23b/M17.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ x$\end{document}</tex-math></inline-formula>,
<disp-formula id="equ3"><label>(3)</label><tex-math id="M18"><?equation-image-name M18.gif?><?equation-image-status READY?><?equation-image-md5 a298a619069d1f5d37afb58c7d8ae459?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/a298a619069d/M18.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
{p_x} = \frac{{T{P_x}}}{{T{P_x} + F{P_x}}}\
\end{eqnarray*}
\end{document}</tex-math></disp-formula>is determined, where the true positives <inline-formula><tex-math id="M19"><?equation-image-name M19.gif?><?equation-image-status READY?><?equation-image-md5 0646fa46a2bf7e9e467e2c8c5e6fc9ce?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/0646fa46a2bf/M19.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$T{P_x}$\end{document}</tex-math></inline-formula> are the number of base pairs that overlap with the mapped genome <inline-formula><tex-math id="M20"><?equation-image-name M20.gif?><?equation-image-status READY?><?equation-image-md5 e51a963812dd8514ab1e5e0e242ce2d6?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e51a963812dd/M20.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$g^*(x)$\end{document}</tex-math></inline-formula>, i.e., <inline-formula><tex-math id="M21"><?equation-image-name M21.gif?><?equation-image-status READY?><?equation-image-md5 23b244afef1fcb206ac0908c1691f572?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/23b244afef1f/M21.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$T{P_x} = |x \cap g^*(x)|$\end{document}</tex-math></inline-formula>, and the false positives <inline-formula><tex-math id="M22"><?equation-image-name M22.gif?><?equation-image-status READY?><?equation-image-md5 1e08de7e0b20f594427211dca5926d38?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/1e08de7e0b20/M22.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$F{P_x}\ $\end{document}</tex-math></inline-formula>are the number of base pairs belonging to other genomes and incorrectly assigned to the bin. The sum <inline-formula><tex-math id="M23"><?equation-image-name M23.gif?><?equation-image-status READY?><?equation-image-md5 e237af6fa521ba57fcb88a0cba1aef04?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e237af6fa521/M23.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$T{P_x} + F{P_x}$\end{document}</tex-math></inline-formula> corresponds to the size of bin <inline-formula><tex-math id="M24"><?equation-image-name M24.gif?><?equation-image-status READY?><?equation-image-md5 d52520c2d23b37d5864a7d95ea8e86c8?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d52520c2d23b/M24.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$x$\end{document}</tex-math></inline-formula> in base pairs. See Fig. <xref ref-type="fig" rid="fig1">1</xref> for an example of predicted genome bins and respective true and false positives.</p><fig id="fig1" orientation="portrait" position="float"><label>Figure 1:</label><caption><p>Schematic representation of establishing a bin-to-genome mapping for calculation of bin quality metrics. Reads and contigs of individual genomes are represented by different symbols and grouped by genome (left) or predicted genome bins (right). A bin-to-genome mapping is established using one of the criteria outlined in the text, with the upper bin mapping to genome C and the lower bin mapping to genome D. The mapping implies <italic toggle="yes">TP</italic>s, <italic toggle="yes">FP</italic>s, and <italic toggle="yes">FN</italic>s for calculation of genome bin purity, completeness, contamination, and overall sample assignment accuracy.</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" position="float" orientation="portrait" xlink:href="giy069fig1.jpg"><?image-name giy069fig1.jpg?><?image-size 114398?><?image-md5 10a9caf693c2e5d3e0c453785235fc65?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 643?><?image-original-width 2128?><?image-scaled-height 214?><?image-scaled-width 709?><?image-cloudpmc-urn urn:cdn:blobs/994d/6022608/10a9caf693c2/giy069fig1.jpg?><?thumb-name giy069fig1.gif?><?thumb-size 15439?><?thumb-md5 9121ce60bb227e63ec7680897da77cfb?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 60?><?thumb-scaled-width 200?><?thumb-cloudpmc-urn urn:cdn:blobs/994d/6022608/9121ce60bb22/giy069fig1.gif?></graphic></fig><p>A related metric, the<bold>contamination <inline-formula><tex-math id="M25"><?equation-image-name M25.gif?><?equation-image-status READY?><?equation-image-md5 23cda586986b00451e4cbd16328deef5?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/23cda586986b/M25.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\boldsymbol{c}}$\end{document}</tex-math></inline-formula></bold>, can be regarded as the opposite of purity and reflects the fraction of incorrect sequence data assigned to a bin (given a mapping to a certain genome). Usually, it suffices to consider either purity or contamination. It is defined for every predicted genome bin<inline-formula><tex-math id="M26"><?equation-image-name M26.gif?><?equation-image-status READY?><?equation-image-md5 d52520c2d23b37d5864a7d95ea8e86c8?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d52520c2d23b/M26.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ x$\end{document}</tex-math></inline-formula> as
<disp-formula id="equ4"><label>(4)</label><tex-math id="M27"><?equation-image-name M27.gif?><?equation-image-status READY?><?equation-image-md5 6b42d869a47fc492444e2ea1876abe74?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/6b42d869a47f/M27.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
{c_x} = 1 - {p_x}\ .
\end{eqnarray*}
\end{document}</tex-math></disp-formula></p><p>The <bold>completeness <inline-formula><tex-math id="M28"><?equation-image-name M28.gif?><?equation-image-status READY?><?equation-image-md5 fa6776ad3cb98fb2474ee85ea9fe2a4b?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/fa6776ad3cb9/M28.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\boldsymbol{r}}$\end{document}</tex-math></inline-formula></bold>, also known as recall or sensitivity, reflects how complete a predicted genome bin is with regard to the sequences of the mapped underlying genome. For every predicted genome bin<inline-formula><tex-math id="M29"><?equation-image-name M29.gif?><?equation-image-status READY?><?equation-image-md5 d52520c2d23b37d5864a7d95ea8e86c8?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d52520c2d23b/M29.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ x$\end{document}</tex-math></inline-formula>,
<disp-formula id="equ5"><label>(5)</label><tex-math id="M30"><?equation-image-name M30.gif?><?equation-image-status READY?><?equation-image-md5 6c67898a741b0213ec5999cd18f0f8d5?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/6c67898a741b/M30.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
{r_x} = \frac{{T{P_x}}}{{T{P_x} + F{N_x}}}{\rm{\ }}
\end{eqnarray*}
\end{document}</tex-math></disp-formula>is calculated, where the false negatives <inline-formula><tex-math id="M31"><?equation-image-name M31.gif?><?equation-image-status READY?><?equation-image-md5 81eb8ecc9f7faa5a73a91f934faeb98e?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/81eb8ecc9f7f/M31.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$F{N_x}$\end{document}</tex-math></inline-formula> are the number of base pairs of the mapped genome <inline-formula><tex-math id="M32"><?equation-image-name M32.gif?><?equation-image-status READY?><?equation-image-md5 e51a963812dd8514ab1e5e0e242ce2d6?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e51a963812dd/M32.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$g^*(x)$\end{document}</tex-math></inline-formula> that were classified to another bin or left unassigned. The sum <inline-formula><tex-math id="M33"><?equation-image-name M33.gif?><?equation-image-status READY?><?equation-image-md5 911ed62f333b890b005e99acf940217d?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/911ed62f333b/M33.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$T{P_x} + F{N_x}\ $\end{document}</tex-math></inline-formula>corresponds to the size of the mapped genome in base pairs.</p><p>Because multiple bins can map to the same genome, some bins might have a purity of 1.0 for a genome (if they exclusively contain its sequences), but the completeness for those bins sum up to at most 1.0 (if they include together all sequences of that genome). Genomes that remain unmapped are considered to have a completeness of zero and their purity is undefined.</p><p>As summary metrics, the <bold>average purity</bold><inline-formula><tex-math id="M34"><?equation-image-name M34.gif?><?equation-image-status READY?><?equation-image-md5 334746f2cfbc6d92348f487224d74de5?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/334746f2cfbc/M34.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\boldsymbol{\bar{p}}}$\end{document}</tex-math></inline-formula> and <bold>average completeness</bold><inline-formula><tex-math id="M35"><?equation-image-name M35.gif?><?equation-image-status READY?><?equation-image-md5 82ae218897b0dc539979ac1afbf89e0a?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/82ae218897b0/M35.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\boldsymbol{\bar{r}}}\ $\end{document}</tex-math></inline-formula> of all predicted genome bins, which are also known in computer science as the macro-averaged precision and macro-averaged recall, can be calculated [<xref rid="bib14" ref-type="bibr">14</xref>]. To these metrics, small bins contribute in the same way as large bins, differently from the sample-specific metrics discussed below. Specifically, the average purity <inline-formula><tex-math id="M36"><?equation-image-name M36.gif?><?equation-image-status READY?><?equation-image-md5 bc27cd78987e47efc2ee94f14704558a?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/bc27cd78987e/M36.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\bar{p}$\end{document}</tex-math></inline-formula> is the fraction of correctly assigned base pairs for all assignments to a given bin averaged over all predicted genome bins, where unmapped genomes are not considered. This value reflects how trustworthy the bin assignments are on average. Let <inline-formula><tex-math id="M37"><?equation-image-name M37.gif?><?equation-image-status READY?><?equation-image-md5 ba500dd921feea006d1740c1a601d0bf?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/ba500dd921fe/M37.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${n_p} = \ | X |$\end{document}</tex-math></inline-formula> be the number of predicted genome bins. Then <inline-formula><tex-math id="M38"><?equation-image-name M38.gif?><?equation-image-status READY?><?equation-image-md5 bc27cd78987e47efc2ee94f14704558a?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/bc27cd78987e/M38.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\bar{p}$\end{document}</tex-math></inline-formula> is calculated as
<disp-formula id="equ6"><label>(6)</label><tex-math id="M39"><?equation-image-name M39.gif?><?equation-image-status READY?><?equation-image-md5 331dedb4ef4c1b35edf5790d31d5e138?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/331dedb4ef4c/M39.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
\bar{p} = \frac{1}{{{n_p}}}\ \mathop \sum \limits_{x \in X} {p_x}.
\end{eqnarray*}
\end{document}</tex-math></disp-formula></p><p>A related metric, the average <bold>contamination <inline-formula><tex-math id="M40"><?equation-image-name M40.gif?><?equation-image-status READY?><?equation-image-md5 9a33f24265283b2dad9f878fae486396?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/9a33f2426528/M40.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\boldsymbol{\bar{c}}}$\end{document}</tex-math></inline-formula></bold> of a genome bin, is computed as
<disp-formula id="equ7"><label>(7)</label><tex-math id="M41"><?equation-image-name M41.gif?><?equation-image-status READY?><?equation-image-md5 b635588f00ac532955c6a3b87890b8b7?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/b635588f00ac/M41.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
\bar{c} = \ 1 - \bar{p}.
\end{eqnarray*}
\end{document}</tex-math></disp-formula></p><p>If very small bins are of little interest in quality evaluations, the <bold>truncated average purity <inline-formula><tex-math id="M42"><?equation-image-name M42.gif?><?equation-image-status READY?><?equation-image-md5 b11a582eb04033fc7037551683d7d404?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/b11a582eb040/M42.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${{\boldsymbol{\bar{p}}}_{\boldsymbol{\alpha }}}$\end{document}</tex-math></inline-formula></bold> can be calculated, where the smallest predicted genome bins adding up to a specified percentage (the <inline-formula><tex-math id="M43"><?equation-image-name M43.gif?><?equation-image-status READY?><?equation-image-md5 2f7f5bce9d2bab5220135d13c8f2e5ab?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/2f7f5bce9d2b/M43.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\alpha $\end{document}</tex-math></inline-formula> percentile) of the dataset are removed. For instance, the 99% truncated average purity can be calculated by sorting the bins according to their predicted size in base pairs and retaining all larger bins that fall into the 99% quantile, including (equally sized) bins that overlap the threshold. Let<inline-formula><tex-math id="M44"><?equation-image-name M44.gif?><?equation-image-status READY?><?equation-image-md5 31a538892bc9f3666e0af5f885c19e93?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/31a538892bc9/M44.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ S,\ S \subset X$\end{document}</tex-math></inline-formula>, be the subset of predicted genome bins of <inline-formula><tex-math id="M45"><?equation-image-name M45.gif?><?equation-image-status READY?><?equation-image-md5 a6c1be96c1a315a0cfeea56cc209d7ce?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/a6c1be96c1a3/M45.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$X$\end{document}</tex-math></inline-formula> after applying the <inline-formula><tex-math id="M46"><?equation-image-name M46.gif?><?equation-image-status READY?><?equation-image-md5 2f7f5bce9d2bab5220135d13c8f2e5ab?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/2f7f5bce9d2b/M46.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\alpha $\end{document}</tex-math></inline-formula> percentile bin size threshold and<inline-formula><tex-math id="M47"><?equation-image-name M47.gif?><?equation-image-status READY?><?equation-image-md5 b71f2d2217226c6c44199ce517e7372e?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/b71f2d221722/M47.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ \ {n_{p,\ \alpha }} = | S |\ $\end{document}</tex-math></inline-formula>. The truncated average purity <inline-formula><tex-math id="M48"><?equation-image-name M48.gif?><?equation-image-status READY?><?equation-image-md5 313e91e568bccfbba213a8cdb948eae2?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/313e91e568bc/M48.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\bar{p}_\alpha }$\end{document}</tex-math></inline-formula> is calculated as
<disp-formula id="equ8"><label>(8)</label><tex-math id="M49"><?equation-image-name M49.gif?><?equation-image-status READY?><?equation-image-md5 7fb7d0e8a05e32c2e34ce6b37609c4de?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/7fb7d0e8a05e/M49.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
{\bar{p}_\alpha } = \frac{1}{{{n_{p,\alpha }}}}\ \mathop \sum \limits_{x \in S} {p_x}.
\end{eqnarray*}
\end{document}</tex-math></disp-formula></p><p>AMBER also allows exclusion of other subsets of bins, such as bins representing viruses or circular elements.</p><p>While the average purity is calculated by averaging over all predicted genome bins, the average completeness <inline-formula><tex-math id="M50"><?equation-image-name M50.gif?><?equation-image-status READY?><?equation-image-md5 cf3a8677043f81f8bd77e70aaa143c46?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/cf3a8677043f/M50.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\bar{r}\ $\end{document}</tex-math></inline-formula>is averaged over all genomes, including those not mapped to genome bins (for which completeness is zero). More formally, let <inline-formula><tex-math id="M51"><?equation-image-name M51.gif?><?equation-image-status READY?><?equation-image-md5 e3b9f4ef5fdcc7c1e7c6f88e5dc6572a?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e3b9f4ef5fdc/M51.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$Z$\end{document}</tex-math></inline-formula> be the set of unmapped genomes, i.e., <inline-formula><tex-math id="M52"><?equation-image-name M52.gif?><?equation-image-status READY?><?equation-image-md5 fb059c418afec999b6efad2bf4d1e9b7?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/fb059c418afe/M52.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$Z = {\rm{\{ }}y \in Y\ {\rm{|}}\ \forall x \in X:g^*(x) \ne y\} $\end{document}</tex-math></inline-formula>, and <inline-formula><tex-math id="M53"><?equation-image-name M53.gif?><?equation-image-status READY?><?equation-image-md5 5e663272c28497984ebace0593d36b8c?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/5e663272c284/M53.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${n_r} = | X |\ + | Z |$\end{document}</tex-math></inline-formula>, i.e., the sum of the number of predicted genome bins and the number of unmapped genomes. Then <inline-formula><tex-math id="M54"><?equation-image-name M54.gif?><?equation-image-status READY?><?equation-image-md5 cf3a8677043f81f8bd77e70aaa143c46?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/cf3a8677043f/M54.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\bar{r}$\end{document}</tex-math></inline-formula> is calculated as
<disp-formula id="equ9"><label>(9)</label><tex-math id="M55"><?equation-image-name M55.gif?><?equation-image-status READY?><?equation-image-md5 e34542f746ba7a0d5c8c81fdf4821209?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e34542f746ba/M55.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
\bar{r} = \frac{1}{{{n_r}}}\ \mathop \sum \limits_{x \in X} {r_x}.
\end{eqnarray*}
\end{document}</tex-math></disp-formula></p></sec><sec id="sec2-2-2"><title>Assessing binnings of specific samples and in relation to bin sizes</title><p>Generally, it may not only be of interest how well a binning program does for individual bins, or all bins on average, irrespective of their sizes, but also how well it does overall for specific types of samples, where some genomes are more abundant than others. Binners may perform differently for more abundant genomes than for less abundant genomes, or for genomes of particular taxa, whose presence and abundance depend strongly on the sampled environment. To allow assessment of such questions, another set of related metrics exist that either measure the binning performance for the entire sample, the binned portion of a sample, or to which bins contribute proportionally to their sizes.</p><p>To give large bins higher weight than small bins in performance determinations, the <bold>average purity <inline-formula><tex-math id="M56"><?equation-image-name M56.gif?><?equation-image-status READY?><?equation-image-md5 133ad00d09a27044624b6a66a75ba345?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/133ad00d09a2/M56.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${{\boldsymbol{\bar{p}}}_{{\boldsymbol{bp}}}}$\end{document}</tex-math></inline-formula></bold> and <bold>completeness <inline-formula><tex-math id="M57"><?equation-image-name M57.gif?><?equation-image-status READY?><?equation-image-md5 78c97990c5eedf6b70f0f8868e4f4882?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/78c97990c5ee/M57.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${{\boldsymbol{\bar{r}}}_{{\boldsymbol{bp}}}}$\end{document}</tex-math></inline-formula></bold> per base pair can be calculated as
<disp-formula id="equ10"><label>(10)</label><tex-math id="M58"><?equation-image-name M58.gif?><?equation-image-status READY?><?equation-image-md5 db4d69b06f02ea8db13d3d36042a07e1?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/db4d69b06f02/M58.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
{\bar{p}_{bp}} = \frac{{\mathop \sum \nolimits_{x \in X} TPx}}{{\mathop \sum \nolimits_{x \in X} TPx + FPx}}\ = \frac{{\mathop \sum \nolimits_{x \in X} \begin{array}{@{}*{1}{c}@{}} {{\rm{max}}}\\ y \end{array}\left| {x\ \cap \ y} \right|}}{{\mathop \sum \nolimits_{x \in X} \left| x \right|}}\
\end{eqnarray*}
\end{document}</tex-math></disp-formula>and
<disp-formula id="equ11"><label>(11)</label><tex-math id="M59"><?equation-image-name M59.gif?><?equation-image-status READY?><?equation-image-md5 6ef7a4ab05f12e3a606adb5b0daa5029?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/6ef7a4ab05f1/M59.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
{\bar{r}_{bp}} = \frac{{\mathop \sum \nolimits_{y \in Y} \begin{array}{@{}*{1}{c}@{}} {{\rm{max}}}\\ x \end{array}\left| {x\ \cap \ y} \right|}}{{\mathop \sum \nolimits_{y \in Y} \left| y \right|}}\ .
\end{eqnarray*}
\end{document}</tex-math></disp-formula></p><p>Equation (<xref ref-type="disp-formula" rid="equ10">10</xref>) strictly uses the bin-to-genome mapping function<inline-formula><tex-math id="M60"><?equation-image-name M60.gif?><?equation-image-status READY?><?equation-image-md5 cb9959f4d21f138598da3f572ba6686d?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/cb9959f4d21f/M60.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ g$\end{document}</tex-math></inline-formula>. Equation (<xref ref-type="disp-formula" rid="equ11">11</xref>) computes the sum in base pairs of the intersection between each genome and the predicted genome bin that maximizes the intersection, averaged over all genomes. A genome that does not intersect with any bin results in an empty intersection. Binners achieving higher values of <inline-formula><tex-math id="M61"><?equation-image-name M61.gif?><?equation-image-status READY?><?equation-image-md5 6f7d7ae318e970a4c12d402a67df066c?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/6f7d7ae318e9/M61.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\bar{p}_{bp}}$\end{document}</tex-math></inline-formula> and <inline-formula><tex-math id="M62"><?equation-image-name M62.gif?><?equation-image-status READY?><?equation-image-md5 7134a6c5ea97fa62084691be1c2845fc?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/7134a6c5ea97/M62.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\bar{r}_{bp}}$\end{document}</tex-math></inline-formula> than for <inline-formula><tex-math id="M63"><?equation-image-name M63.gif?><?equation-image-status READY?><?equation-image-md5 bc27cd78987e47efc2ee94f14704558a?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/bc27cd78987e/M63.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\bar{p}$\end{document}</tex-math></inline-formula> and <inline-formula><tex-math id="M64"><?equation-image-name M64.gif?><?equation-image-status READY?><?equation-image-md5 cf3a8677043f81f8bd77e70aaa143c46?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/cf3a8677043f/M64.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\bar{r}\ $\end{document}</tex-math></inline-formula>tend to do better for larger bins than for small ones. For those with lower values, it is the other way around.</p><p>The <bold>accuracy <italic toggle="yes">a</italic></bold> measures the average assignment quality per base pair over the entire dataset, including unassigned base pairs. It is calculated as
<disp-formula id="update185118_equ12"><label>(12)</label><tex-math id="M65"><?equation-image-name M65.gif?><?equation-image-status READY?><?equation-image-md5 b25c220803cb46982c762f28128d27ff?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/b25c220803cb/M65.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
a\ = \frac{{\mathop \sum \nolimits_{x \in X} TPx}}{{U + \mathop \sum \nolimits_{x \in X} TPx + FPx}}\ ,
\end{eqnarray*}
\end{document}</tex-math></disp-formula>where <inline-formula><tex-math id="M66"><?equation-image-name M66.gif?><?equation-image-status READY?><?equation-image-md5 7d692246a473f8ef3f94fc9268d00b38?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/7d692246a473/M66.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$U$\end{document}</tex-math></inline-formula> is the number of base pairs that were left unassigned. Like the average purity and completeness per base pair, large bins contribute more strongly to this metric than small bins.</p><p>Genome binners generate groups or clusters of reads and contigs for a given dataset. Instead of calculating performance metrics established with a bin-to-genome mapping, the quality of a clustering can be evaluated by measuring the similarity between the obtained and correct cluster partitions of the dataset, corresponding here to the predicted genome bins and the gold standard contig or read genome assignments, respectively. This is accomplished with the Rand index by comparing how pairs of items are clustered [<xref rid="bib15" ref-type="bibr">15</xref>]. Two contigs or reads of the same genome that are placed in the same predicted genome bin are considered true positives <inline-formula><tex-math id="M67"><?equation-image-name M67.gif?><?equation-image-status READY?><?equation-image-md5 ac86b87ed7dd619abcaf9059d1e13d50?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/ac86b87ed7dd/M67.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ TP$\end{document}</tex-math></inline-formula>. Two contigs or reads of different genomes that are placed in different bins are considered true negatives <inline-formula><tex-math id="M68"><?equation-image-name M68.gif?><?equation-image-status READY?><?equation-image-md5 ed8d4c00eb25b00171a3ef91f0e325de?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/ed8d4c00eb25/M68.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ TN$\end{document}</tex-math></inline-formula>. The Rand index ranges from 0 to 1 and is the number of true pairs,<inline-formula><tex-math id="M69"><?equation-image-name M69.gif?><?equation-image-status READY?><?equation-image-md5 e43d3137db49fcdb4ed7b6d99ad9c76a?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e43d3137db49/M69.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ TP + TN$\end{document}</tex-math></inline-formula>, divided by the total number of pairs. However, for a random clustering of the dataset, the Rand index would be larger than 0. The <bold>adjusted Rand index</bold> (ARI) corrects for this by subtracting the expected value for the Rand index and normalizing the resulting value, such that the values still range from 0 to 1.</p><p>More formally, following [<xref rid="bib16" ref-type="bibr">16</xref>], let <inline-formula><tex-math id="M70"><?equation-image-name M70.gif?><?equation-image-status READY?><?equation-image-md5 6be9a0d321942a741aedd4541173b5e6?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/6be9a0d32194/M70.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$m$\end{document}</tex-math></inline-formula> be the total number of base pairs assigned to any predicted genome bin and,<inline-formula><tex-math id="M71"><?equation-image-name M71.gif?><?equation-image-status READY?><?equation-image-md5 9e379a61f585159ea761d87e86cea052?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/9e379a61f585/M71.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ {m_{x,\ y}}$\end{document}</tex-math></inline-formula>, the number of base pairs of genome <inline-formula><tex-math id="M72"><?equation-image-name M72.gif?><?equation-image-status READY?><?equation-image-md5 e5395764aae3a8a3b23943e6b2699b03?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e5395764aae3/M72.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$y$\end{document}</tex-math></inline-formula> assigned to predicted genome bin <inline-formula><tex-math id="M73"><?equation-image-name M73.gif?><?equation-image-status READY?><?equation-image-md5 d52520c2d23b37d5864a7d95ea8e86c8?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d52520c2d23b/M73.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$x$\end{document}</tex-math></inline-formula>. The ARI is computed as
<disp-formula id="update185118_equ13"><label>(13)</label><tex-math id="M74"><?equation-image-name M74.gif?><?equation-image-status READY?><?equation-image-md5 912984be2456cfeb6068caf00c294021?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/912984be2456/M74.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}
\begin{eqnarray*}
ARI\ = \frac{{\mathop \sum \nolimits_{x,y} \left( {\begin{array}{@{}*{1}{c}@{}} {{m_{x,y}}}\\ 2 \end{array}} \right) - \frac{{\mathop \sum \nolimits_x \left( {\begin{array}{@{}*{1}{c}@{}} {{m_{x,.}}}\\ 2 \end{array}} \right)\mathop \sum \nolimits_y \left( {\begin{array}{@{}*{1}{c}@{}} {{m_{.,y}}}\\ 2 \end{array}} \right)}}{{\left( {\begin{array}{@{}*{1}{c}@{}} m\\ 2 \end{array}} \right)}}}}{{\frac{1}{2}\left[ {\mathop \sum \nolimits_x \left( {\begin{array}{@{}*{1}{c}@{}} {{m_{x,.}}}\\ 2 \end{array}} \right) + \mathop \sum \nolimits_y \left( {\begin{array}{@{}*{1}{c}@{}} {{m_{.,y}}}\\ 2 \end{array}} \right)} \right] - \frac{{\mathop \sum \nolimits_x \left( {\begin{array}{@{}*{1}{c}@{}} {{m_{x,.}}}\\ 2 \end{array}} \right)\mathop \sum \nolimits_y \left( {\begin{array}{@{}*{1}{c}@{}} {{m_{.,y}}}\\ 2 \end{array}} \right)}}{{\left( {\begin{array}{@{}*{1}{c}@{}} m\\ 2 \end{array}} \right)}}}},
\end{eqnarray*}
\end{document}</tex-math></disp-formula>where <inline-formula><tex-math id="M75"><?equation-image-name M75.gif?><?equation-image-status READY?><?equation-image-md5 d8d8c9836f5afcbdf74b35c2c6a28a8f?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d8d8c9836f5a/M75.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${m_{.,y}} = \mathop \sum \limits_x {m_{x,y}}\ $\end{document}</tex-math></inline-formula> and <inline-formula><tex-math id="M76"><?equation-image-name M76.gif?><?equation-image-status READY?><?equation-image-md5 21e473a9c46a1494a0c8eec73b29746e?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/21e473a9c46a/M76.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${m_{x,.}} = \mathop \sum \limits_y {m_{x,y}}\ $\end{document}</tex-math></inline-formula>. That is, <inline-formula><tex-math id="M77"><?equation-image-name M77.gif?><?equation-image-status READY?><?equation-image-md5 d8835428ecd6e66e16420ed7a19db405?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d8835428ecd6/M77.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${m_{.,y}}$\end{document}</tex-math></inline-formula> is the number of base pairs of genome<inline-formula><tex-math id="M78"><?equation-image-name M78.gif?><?equation-image-status READY?><?equation-image-md5 e5395764aae3a8a3b23943e6b2699b03?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e5395764aae3/M78.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ y$\end{document}</tex-math></inline-formula> from all bin assignments and<inline-formula><tex-math id="M79"><?equation-image-name M79.gif?><?equation-image-status READY?><?equation-image-md5 c92a1c3eb06915a3037b8083f737738d?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/c92a1c3eb069/M79.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ {m_{x,.}}$\end{document}</tex-math></inline-formula> is the total number of base pairs in predicted genome bin<inline-formula><tex-math id="M80"><?equation-image-name M80.gif?><?equation-image-status READY?><?equation-image-md5 d52520c2d23b37d5864a7d95ea8e86c8?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d52520c2d23b/M80.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ x$\end{document}</tex-math></inline-formula>.</p><p>AMBER also provides ARI as a measure of assignment accuracy per sequence (contig or read) instead of per base pair by considering <inline-formula><tex-math id="M81"><?equation-image-name M81.gif?><?equation-image-status READY?><?equation-image-md5 6be9a0d321942a741aedd4541173b5e6?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/6be9a0d32194/M81.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$m\ $\end{document}</tex-math></inline-formula>to be the total number of sequences assigned to any bin and <inline-formula><tex-math id="M82"><?equation-image-name M82.gif?><?equation-image-status READY?><?equation-image-md5 9dce248b5d308a0c3a891d7c7ed2593a?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/9dce248b5d30/M82.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${m_{x,y}}$\end{document}</tex-math></inline-formula> the number of sequences of genome <inline-formula><tex-math id="M83"><?equation-image-name M83.gif?><?equation-image-status READY?><?equation-image-md5 e5395764aae3a8a3b23943e6b2699b03?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/e5395764aae3/M83.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$y$\end{document}</tex-math></inline-formula> assigned to bin<inline-formula><tex-math id="M84"><?equation-image-name M84.gif?><?equation-image-status READY?><?equation-image-md5 d52520c2d23b37d5864a7d95ea8e86c8?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d52520c2d23b/M84.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ x$\end{document}</tex-math></inline-formula>. The meaning of <inline-formula><tex-math id="M85"><?equation-image-name M85.gif?><?equation-image-status READY?><?equation-image-md5 d8835428ecd6e66e16420ed7a19db405?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d8835428ecd6/M85.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${m_{.,y}}$\end{document}</tex-math></inline-formula> and <inline-formula><tex-math id="M86"><?equation-image-name M86.gif?><?equation-image-status READY?><?equation-image-md5 c92a1c3eb06915a3037b8083f737738d?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/c92a1c3eb069/M86.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${m_{x,.}}$\end{document}</tex-math></inline-formula> changes accordingly.</p><p>Importantly, the ARI is mainly designed for assessing a clustering of an entire dataset, but some genome binning programs exclude sequences from bin assignment, thus assigning only a subset of the sequences from a given dataset. If this unassigned portion is included in the ARI calculation, the ARI becomes meaningless. AMBER, therefore, calculates the ARI only for the assigned portion of the data. For interpretation of these ARI values, the percentage of assigned data should also be considered (provided by AMBER together in plots).</p></sec></sec><sec id="sec2-3"><title>Output and visualization</title><p>AMBER combines the assessment of genome reconstructions from different binning programs or created with varying parameters for one program. The calculated metrics are provided as flat files, in several plots, and in an interactive HTML visualization. An example page is available at [<xref rid="bib17" ref-type="bibr">17</xref>]. The plots visualize the following:
<list list-type="bullet"><list-item><p>(Truncated) purity<inline-formula><tex-math id="M87"><?equation-image-name M87.gif?><?equation-image-status READY?><?equation-image-md5 313e91e568bccfbba213a8cdb948eae2?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/313e91e568bc/M87.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ {\bar{p}_\alpha }$\end{document}</tex-math></inline-formula> per predicted genome bin vs. average completeness <inline-formula><tex-math id="M88"><?equation-image-name M88.gif?><?equation-image-status READY?><?equation-image-md5 cf3a8677043f81f8bd77e70aaa143c46?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/cf3a8677043f/M88.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\bar{r}$\end{document}</tex-math></inline-formula> per genome, with the standard error of the mean</p></list-item><list-item><p>Average purity per base pair <inline-formula><tex-math id="M89"><?equation-image-name M89.gif?><?equation-image-status READY?><?equation-image-md5 6f7d7ae318e970a4c12d402a67df066c?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/6f7d7ae318e9/M89.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\bar{p}_{bp}}$\end{document}</tex-math></inline-formula> vs. average completeness per base pair <inline-formula><tex-math id="M90"><?equation-image-name M90.gif?><?equation-image-status READY?><?equation-image-md5 7134a6c5ea97fa62084691be1c2845fc?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/7134a6c5ea97/M90.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\bar{r}_{bp}}$\end{document}</tex-math></inline-formula></p></list-item><list-item><p>ARI vs. percentage of assigned data</p></list-item><list-item><p>Purity <inline-formula><tex-math id="M91"><?equation-image-name M91.gif?><?equation-image-status READY?><?equation-image-md5 42ba30420e95b2eefae13651162d8c2c?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/42ba30420e95/M91.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${p_x}$\end{document}</tex-math></inline-formula> vs. completeness <inline-formula><tex-math id="M92"><?equation-image-name M92.gif?><?equation-image-status READY?><?equation-image-md5 6802aecc3db9c9bb15844ae63b18077f?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/6802aecc3db9/M92.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${r_x}$\end{document}</tex-math></inline-formula> and box plots for all predicted bins</p></list-item><list-item><p>Heat maps for individual binnings representing base pair assignments to predicted bins vs. their true origins from the underlying genomes</p></list-item></list></p><p>Heat maps are generated from binnings without requiring a mapping, where rows represent the predicted genome bins and columns represent the genomes. The last row includes all unassigned base pairs for every individual genome and, individual entries, the number of base pairs assigned to a bin from a particular genome. Hence, the sum of all entries in a row corresponds to the bin size and the sum of all column entries corresponds to the size of the underlying genome. To facilitate the visualization of the overall binning quality, rows and columns are sorted as follows: for each predicted bin in each row, a bin-to-genome mapping function (<inline-formula><tex-math id="M93"><?equation-image-name M93.gif?><?equation-image-status READY?><?equation-image-md5 cb9959f4d21f138598da3f572ba6686d?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/cb9959f4d21f/M93.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$g$\end{document}</tex-math></inline-formula>, per default) determines the genome (column) that maps to the bin and the true positive base pairs for the bin. Predicted bins are then sorted by the number of true positives in descending order from top to bottom in the matrix, and genomes are sorted from left to right in the same order of the bin-to-genome mappings for the predicted bins. In this way, true positives concentrate in the main diagonal starting at the upper left corner of the matrix.</p><p>AMBER also provides a summary table with the number of genomes recovered with less than a certain threshold (5% and 10% per default) of contamination and more than another threshold (50%, 70%, and 90% per default) of completeness. This is one of the main quality measures used by CheckM [<xref rid="bib3" ref-type="bibr">3</xref>] and in, e.g., [<xref rid="bib18" ref-type="bibr">18</xref>] and [<xref rid="bib19" ref-type="bibr">19</xref>]. In addition, a ranking of different binnings by the highest average purity, average completeness, or the sum of these two metrics is provided as a flat file.</p></sec></sec><sec sec-type="results" id="sec3"><title>Results</title><p>To demonstrate an application of AMBER, we performed an evaluation of the genome binning submissions to the first CAMI challenge together with predictions from four more programs and new program versions on two of the three challenge datasets. These are simulated benchmark datasets representing a single sample dataset from a low-complexity microbial community with 40 genomes and a five-sample time series dataset of a high-complexity microbial community with 596 genome members. Both datasets include bacteria, the high-complexity sample also archaea, high copy circular elements (plasmids and viruses), and substantial strain-level diversity. The samples were sequenced with paired-end 150-bp Illumina reads to a size of 15 GB for each sample. The assessed binners were CONCOCT [<xref rid="bib16" ref-type="bibr">16</xref>], MaxBin 2.0.2 [<xref rid="bib11" ref-type="bibr">11</xref>], MetaBAT [<xref rid="bib20" ref-type="bibr">20</xref>], Metawatt 3.5 [<xref rid="bib21" ref-type="bibr">21</xref>], and MyCC [<xref rid="bib22" ref-type="bibr">22</xref>]. We generated results with newer program versions of MetaBAT and MaxBin. Furthermore, we ran Binsanity, Binsanity-wf [<xref rid="bib23" ref-type="bibr">23</xref>], COCACOLA [<xref rid="bib24" ref-type="bibr">24</xref>], and DAS Tool 1.1 [<xref rid="bib25" ref-type="bibr">25</xref>] on the datasets. DAS Tool combines predictions from multiple binners, aiming to produce consensus high-quality bins. We used as input for DAS Tool the predictions of all binners, except COCACOLA; for MaxBin and MetaBAT, we used the results of the newer versions 2.2.4 and 2.11.2, respectively. The commands and parameters used with the programs are available in the <xref ref-type="supplementary-material" rid="sup8">Supplementary Information</xref>.</p><p>On the low-complexity dataset, MaxBin 2.2.4, as its previous version 2.0.2, performed very well, as did the new MetaBAT version 2.11.2 and DAS Tool 1.1 (Fig. <xref ref-type="fig" rid="fig3">3</xref>, <xref ref-type="supplementary-material" rid="sup8">Supplementary Fig. S1</xref>). Both MaxBin versions achieved the highest average purity per bin, and version 2.0.2 achieved the highest completeness per genome on this dataset. As in the evaluation of the first CAMI challenge, we report the truncated average purity,<inline-formula><tex-math id="M94"><?equation-image-name M94.gif?><?equation-image-status READY?><?equation-image-md5 7ae2ab67fec167ee370ce209be94a217?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/7ae2ab67fec1/M94.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}$\ {\bar{p}_{99}}$\end{document}</tex-math></inline-formula>, with 1% of the smallest bins predicted by each program removed. These small bins are of little practical interest for the analysis of individual bins and distort the average purity, since their purity is usually much lower than that of larger bins (<xref ref-type="supplementary-material" rid="sup8">Supplementary Table S2</xref>) and small and large bins contribute equally to this metric. On the high-complexity dataset, both MaxBin versions assigned less data than other programs, though with the highest purity (Figs. <xref ref-type="fig" rid="fig2">2</xref> and <xref ref-type="fig" rid="fig3">3</xref>). MetaBAT 2.11.2 substantially improved over the previous version with all measures. Apart from DAS Tool 1.1, which created the most high-quality bins from the predictions of the different binners, MetaBAT 2.11.2 recovered the most high-quality bins and showed the highest interquartile range in the purity and completeness box plots for the high-complexity dataset. MetaBAT 2.11.2 and MaxBin 2.0.2 also recovered the most genomes with more than the specified thresholds of completeness and contamination on the high- and low-complexity datasets, respectively (Table <xref rid="tbl1" ref-type="table">1</xref>, <xref ref-type="supplementary-material" rid="sup8">Supplementary Table S1</xref>). DAS Tool 1.1 could further improve on this measure, recovering the most genomes satisfying these conditions on both datasets. Overall, DAS Tool obtained high-quality consensus bins, asserting itself as an option that can be used particularly when it is not clear which binner performs best on a specific dataset. As shown in [<xref rid="bib25" ref-type="bibr">25</xref>], no single binner performs well on all ecosystems and, equivalently, there is no guarantee that the best-performing binners on the analyzed datasets from the first CAMI challenge also perform best on other datasets. For more extensive information on program performances of multiple datasets, we refer the reader to [<xref rid="bib1" ref-type="bibr">1</xref>] and future benchmarking challenges organized by CAMI [<xref rid="bib26" ref-type="bibr">26</xref>]. Notably, some binners, such as CONCOCT, may require more than five samples for optimal performance. In general, the binning performance can also be influenced by parameter settings. These could possibly be fine-tuned to yield better results than the ones presented here. We chose to use default parameters or parameters suggested by the developers of the respective binners during the CAMI challenge (<xref ref-type="supplementary-material" rid="sup8">Supplementary Information</xref>), reproducing a realistic scenario where such fine-tuning is difficult due to the lack of gold standard binnings. To thoroughly and fairly benchmark binners, the CAMI challenge encouraged multiple submissions of the same binner with different parameter settings. Although we present results for binner versions released after the end of the challenge, with noticeable improvements of MetaBAT 2.11.2, the authors of MetaBAT claim that no dataset-specific fine-tuning was performed (direct communication). All results and evaluations are also available in the CAMI benchmarking portal [<xref rid="bib27" ref-type="bibr">27</xref>].</p><fig id="fig2" orientation="portrait" position="float"><label>Figure 2:</label><caption><p>Assessment of genome bins reconstructed from CAMI's high-complexity challenge dataset by different binners. Binner versions participating in CAMI are indicated in the legend in parentheses. <bold>(A)</bold> Average purity per bin (<italic toggle="yes">x</italic>axis), average completeness per genome (<italic toggle="yes">y</italic>axis), and respective standard errors (bars). As in the CAMI challenge, we report <inline-formula><tex-math id="M95"><?equation-image-name M95.gif?><?equation-image-status READY?><?equation-image-md5 7ae2ab67fec167ee370ce209be94a217?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/7ae2ab67fec1/M95.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\bar{p}_{99}}$\end{document}</tex-math></inline-formula> with 1% of the smallest bins predicted by each program removed. <bold>(B)</bold> Average purity per base pair (<italic toggle="yes">x</italic>axis) and average completeness per base pair (<italic toggle="yes">y</italic>axis). <bold>(C)</bold>ARI per base pair (<italic toggle="yes">x</italic>axis) and percentage of assigned base pairs (<italic toggle="yes">y</italic>axis). <bold>(D and E)</bold> Box plots of purity per bin and completeness per genome, respectively.</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" position="float" orientation="portrait" xlink:href="giy069fig2.jpg"><?image-name giy069fig2.jpg?><?image-size 193566?><?image-md5 d7e092b857dcca57d94f695d97a08cea?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 1748?><?image-original-width 2490?><?image-scaled-height 499?><?image-scaled-width 711?><?image-cloudpmc-urn urn:cdn:blobs/994d/6022608/d7e092b857dc/giy069fig2.jpg?><?thumb-name giy069fig2.gif?><?thumb-size 14583?><?thumb-md5 0b2274f671765fcb2e9dfe52cb7bf93e?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 79?><?thumb-scaled-width 113?><?thumb-cloudpmc-urn urn:cdn:blobs/994d/6022608/0b2274f67176/giy069fig2.gif?></graphic></fig><fig id="fig3" orientation="portrait" position="float"><label>Figure 3:</label><caption><p>Heat maps of confusion matrices for four binning results for the low-complexity dataset of the first CAMI challenge representing the base pair assignments to predicted genome bins (<italic toggle="yes">y</italic>axis) vs. their true origin from the underlying genomes or circular elements (<italic toggle="yes">x</italic>axis). Rows and columns are sorted according to the number of true positives per predicted bin (see main text). Row scatter indicates a reduced average purity per base pair and thus underbinning (genomes assigned to one bin), whereas column scatter indicates a lower completeness per base pair and thus overbinning (many bins for one genome). The last row represents the unassigned bases per genome, allowing assessment of the fraction of the sample left unassigned. These views allow a more detailed inspection of binning quality relating to the provided quality metrics (<xref ref-type="supplementary-material" rid="sup8">Supplementary Fig. S1</xref>).</p></caption><graphic xmlns:xlink="http://www.w3.org/1999/xlink" position="float" orientation="portrait" xlink:href="giy069fig3.jpg"><?image-name giy069fig3.jpg?><?image-size 140332?><?image-md5 f2ed61f58dd8777903631cc7156d1545?><?image-image-server-status LOAD_COMPLETED?><?image-original-height 2105?><?image-original-width 2539?><?image-scaled-height 601?><?image-scaled-width 725?><?image-cloudpmc-urn urn:cdn:blobs/994d/6022608/f2ed61f58dd8/giy069fig3.jpg?><?thumb-name giy069fig3.gif?><?thumb-size 14625?><?thumb-md5 88c6ae7afd3ac1f6575f95652304d290?><?thumb-image-server-status NEVER_LOAD?><?thumb-scaled-height 83?><?thumb-scaled-width 100?><?thumb-cloudpmc-urn urn:cdn:blobs/994d/6022608/88c6ae7afd3a/giy069fig3.gif?></graphic></fig><table-wrap id="tbl1" orientation="portrait" position="float"><label>Table 1:</label><caption><p>Respective number of genomes recovered from CAMI's high-complexity dataset with less than 10% and 5% contamination and more than 50%, 70%, and 90% completeness.</p></caption><table frame="hsides" rules="groups"><thead><tr><th colspan="2" align="center" rowspan="1">Genome binner<break/>(% contamination)</th><th colspan="3" align="center" rowspan="1">Predicted bins<break/>(% completeness)</th></tr></thead><tbody><tr><td rowspan="1" colspan="1"/><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&gt;50%</td><td align="left" rowspan="1" colspan="1">&gt;70%</td><td align="left" rowspan="1" colspan="1">&gt;90%</td></tr><tr><td align="left" rowspan="1" colspan="1">Gold standard</td><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">596</td><td align="left" rowspan="1" colspan="1">596</td><td align="left" rowspan="1" colspan="1">596</td></tr><tr><td align="left" rowspan="1" colspan="1">CONCOCT (CAMI)</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">129</td><td align="left" rowspan="1" colspan="1">129</td><td align="left" rowspan="1" colspan="1">123</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">124</td><td align="left" rowspan="1" colspan="1">124</td><td align="left" rowspan="1" colspan="1">118</td></tr><tr><td align="left" rowspan="1" colspan="1">MaxBin 2.0.2 (CAMI)</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">277</td><td align="left" rowspan="1" colspan="1">274</td><td align="left" rowspan="1" colspan="1">244</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">254</td><td align="left" rowspan="1" colspan="1">252</td><td align="left" rowspan="1" colspan="1">224</td></tr><tr><td align="left" rowspan="1" colspan="1">MaxBin 2.2.4</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">274</td><td align="left" rowspan="1" colspan="1">271</td><td align="left" rowspan="1" colspan="1">236</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">249</td><td align="left" rowspan="1" colspan="1">247</td><td align="left" rowspan="1" colspan="1">216</td></tr><tr><td align="left" rowspan="1" colspan="1">MetaBAT (CAMI)</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">173</td><td align="left" rowspan="1" colspan="1">152</td><td align="left" rowspan="1" colspan="1">126</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">159</td><td align="left" rowspan="1" colspan="1">140</td><td align="left" rowspan="1" colspan="1">118</td></tr><tr><td align="left" rowspan="1" colspan="1">MetaBAT 2.11.2</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">
<bold>427</bold>
</td><td align="left" rowspan="1" colspan="1">
<bold>417</bold>
</td><td align="left" rowspan="1" colspan="1">
<bold>361</bold>
</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">
<bold>414</bold>
</td><td align="left" rowspan="1" colspan="1">
<bold>404</bold>
</td><td align="left" rowspan="1" colspan="1">
<bold>353</bold>
</td></tr><tr><td align="left" rowspan="1" colspan="1">Metawatt 3.5 (CAMI)</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">408</td><td align="left" rowspan="1" colspan="1">387</td><td align="left" rowspan="1" colspan="1">338</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">396</td><td align="left" rowspan="1" colspan="1">376</td><td align="left" rowspan="1" colspan="1">330</td></tr><tr><td align="left" rowspan="1" colspan="1">MyCC (CAMI)</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">189</td><td align="left" rowspan="1" colspan="1">182</td><td align="left" rowspan="1" colspan="1">145</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">166</td><td align="left" rowspan="1" colspan="1">159</td><td align="left" rowspan="1" colspan="1">127</td></tr><tr><td align="left" rowspan="1" colspan="1">Binsanity 0.2.5.9</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">9</td><td align="left" rowspan="1" colspan="1">9</td><td align="left" rowspan="1" colspan="1">9</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">6</td><td align="left" rowspan="1" colspan="1">6</td><td align="left" rowspan="1" colspan="1">6</td></tr><tr><td align="left" rowspan="1" colspan="1">Binsanity-refine 0.2.5.9</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">206</td><td align="left" rowspan="1" colspan="1">204</td><td align="left" rowspan="1" colspan="1">192</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">183</td><td align="left" rowspan="1" colspan="1">181</td><td align="left" rowspan="1" colspan="1">171</td></tr><tr><td align="left" rowspan="1" colspan="1">COCACOLA</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">88</td><td align="left" rowspan="1" colspan="1">87</td><td align="left" rowspan="1" colspan="1">75</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">69</td><td align="left" rowspan="1" colspan="1">69</td><td align="left" rowspan="1" colspan="1">60</td></tr><tr><td align="left" rowspan="1" colspan="1">DAS Tool 1.1</td><td align="left" rowspan="1" colspan="1">&lt;10%</td><td align="left" rowspan="1" colspan="1">
<bold>465</bold>
</td><td align="left" rowspan="1" colspan="1">
<bold>462</bold>
</td><td align="left" rowspan="1" colspan="1">
<bold>405</bold>
</td></tr><tr><td rowspan="1" colspan="1"/><td align="left" rowspan="1" colspan="1">&lt;5%</td><td align="left" rowspan="1" colspan="1">
<bold>428</bold>
</td><td align="left" rowspan="1" colspan="1">
<bold>425</bold>
</td><td align="left" rowspan="1" colspan="1">
<bold>376</bold>
</td></tr></tbody></table><table-wrap-foot><fn id="tb1fn1"><p>In bold are the highest number of recovered genomes for a certain level of completeness (column) and contamination (row).</p></fn></table-wrap-foot></table-wrap></sec><sec sec-type="conclusions" id="sec4"><title>Conclusions</title><p>AMBER provides commonly used metrics for assessing the quality of metagenome binnings on benchmark datasets in several convenient output formats, allowing in-depth comparisons of binning results of different programs, software versions, and with varying parameter settings. As such, AMBER facilitates the assessment of genome binning programs on benchmark metagenome datasets for bioinformaticians aiming to optimize data processing pipelines and method developers. The software is available as a stand-alone program [<xref rid="bib12" ref-type="bibr">12</xref>], as a Docker image (automatically built with the provided Dockerfile), and in the CAMI benchmarking portal [<xref rid="bib27" ref-type="bibr">27</xref>]. We will continue to extend the metrics and visualizations according to community requirements and suggestions.</p></sec><sec id="sec5"><title>Availability of source code</title><p>Project name: AMBER: Assessment of Metagenome BinnERs</p><p>Project home page: <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://github.com/CAMI-challenge/AMBER">https://github.com/CAMI-challenge/AMBER</ext-link></p><p>Research Resource Identifier: SCR_016151</p><p>Operating system(s): Platform independent</p><p>Programming language: Python 3.5</p><p>License: Apache 2.0</p></sec><sec id="sec6"><title>Availability of supporting data</title><p>An archive of the CAMI benchmark datasets [<xref rid="bib2" ref-type="bibr">2</xref>] and snapshots of the code [<xref rid="bib28" ref-type="bibr">28</xref>] are available in the <italic toggle="yes">GigaScience</italic> GigaDB repository.</p></sec><sec id="sec7"><title>Additional files</title><p>
<bold><xref ref-type="supplementary-material" rid="sup8">SupplementaryInformation.pdf</xref></bold>. This file contains the following Figures, Tables, and Sections. <xref ref-type="supplementary-material" rid="sup8">Supplementary Fig. S1</xref>: Assessment of genomes reconstructed from CAMI’s low complexity challenge dataset by different binners. <xref ref-type="supplementary-material" rid="sup8">Supplementary Table S1</xref>: Number of genomes recovered from CAMI's low complexity data set. <xref ref-type="supplementary-material" rid="sup8">Supplementary Table S2</xref>: Total number of bins predicted by each binner on CAMI’s high complexity dataset and respective number of bins removed to compute the truncated average purity per bin <inline-formula><tex-math id="M96"><?equation-image-name M96.gif?><?equation-image-status READY?><?equation-image-md5 7ae2ab67fec167ee370ce209be94a217?><?equation-image-cloudpmc-urn urn:cdn:blobs/994d/6022608/7ae2ab67fec1/M96.gif?>\documentclass[12pt]{minimal}
\usepackage{amsmath}
\usepackage{wasysym} 
\usepackage{amsfonts} 
\usepackage{amssymb} 
\usepackage{amsbsy}
\usepackage{upgreek}
\usepackage{mathrsfs}
\setlength{\oddsidemargin}{-69pt}
\begin{document}
}{}${\bar{p}_{99}}$\end{document}</tex-math></inline-formula>. Steps and commands used to run the assessed binning programs.</p></sec><sec id="h1content1529411265998"><title>Abbreviations</title><p>ARI: adjusted Rand index; CAMI: Critical Assessment of Metagenome Interpretation.</p></sec><sec id="sec8"><title>Competing interests</title><p>The authors declare that they have no competing interests.</p></sec><sec id="h1content1529403091948"><title>Authors' contributions</title><p>F.M. implemented most of AMBER, evaluated all presented binners, and wrote the manuscript together with A.C.M. P.H., R.G.O, and A.F. implemented metrics, helped to decide on useful visualizations, and evaluated binners in the first CAMI challenge. P.B. implemented automatic tests, the HTML visualization of AMBER, and integrated it in the CAMI benchmarking portal. A.C.M. and A.S. co-organized the first CAMI challenge and helped to decide on useful metrics. A.C.M. initiated the AMBER project, supervised it, and wrote parts of the manuscript.</p></sec><sec id="sec9"><title>Funding</title><p>This work was supported by Helmholtz Society and the Cluster of Excellence in Plant Sciences funded by the German Research Foundation.</p></sec><sec sec-type="supplementary-material"><title>Supplementary Material</title><supplementary-material content-type="local-data" id="sup1" position="float" orientation="portrait"><label>GIGA-D-18-00016_Original_Submission.pdf</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="giy069_giga-d-18-00016_original_submission.pdf" position="float" orientation="portrait"><?suppdata-name giy069_giga-d-18-00016_original_submission.pdf?><?suppdata-size 1520040?><?suppdata-md5 c1858b1342a3a3262caa08de9322da4e?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:994d/6022608/c1858b1342a3/giy069_giga-d-18-00016_original_submission.pdf?><caption><p>Click here for additional data file.</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="sup2" position="float" orientation="portrait"><label>GIGA-D-18-00016_Revision_1.pdf</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="giy069_giga-d-18-00016_revision_1.pdf" position="float" orientation="portrait"><?suppdata-name giy069_giga-d-18-00016_revision_1.pdf?><?suppdata-size 1542614?><?suppdata-md5 84c33333bf0e46037479de1b02bcc0d8?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:994d/6022608/84c33333bf0e/giy069_giga-d-18-00016_revision_1.pdf?><caption><p>Click here for additional data file.</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="sup3" position="float" orientation="portrait"><label>GIGA-D-18-00016_Revision_2.pdf</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="giy069_giga-d-18-00016_revision_2.pdf" position="float" orientation="portrait"><?suppdata-name giy069_giga-d-18-00016_revision_2.pdf?><?suppdata-size 1538379?><?suppdata-md5 f56920bd37f36886a9a0e2f7c88cec42?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:994d/6022608/f56920bd37f3/giy069_giga-d-18-00016_revision_2.pdf?><caption><p>Click here for additional data file.</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="sup4" position="float" orientation="portrait"><label>Response_to_Reviewer_Comments_Original_Submission.pdf</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="giy069_response_to_reviewer_comments_original_submission.pdf" position="float" orientation="portrait"><?suppdata-name giy069_response_to_reviewer_comments_original_submission.pdf?><?suppdata-size 22394?><?suppdata-md5 6143b9fd0350f364a79af83119a6d9db?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:994d/6022608/6143b9fd0350/giy069_response_to_reviewer_comments_original_submission.pdf?><caption><p>Click here for additional data file.</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="sup5" position="float" orientation="portrait"><label>Response_to_Reviewer_Comments_Revision_1.pdf</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="giy069_response_to_reviewer_comments_revision_1.pdf" position="float" orientation="portrait"><?suppdata-name giy069_response_to_reviewer_comments_revision_1.pdf?><?suppdata-size 41507?><?suppdata-md5 0da13dba21a6a9a8f4d504d70efeadc7?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:994d/6022608/0da13dba21a6/giy069_response_to_reviewer_comments_revision_1.pdf?><caption><p>Click here for additional data file.</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="sup6" position="float" orientation="portrait"><label>Reviewer_1_Report_(Original_Submission) -- Magdalena Calusinska</label><caption><p>1/26/2018 Reviewed</p></caption><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="giy069_reviewer_1_report_(original_submission).pdf" position="float" orientation="portrait"><?suppdata-name giy069_reviewer_1_report_(original_submission).pdf?><?suppdata-size 28226?><?suppdata-md5 84f434771a80012bcb9370e1a78b65f3?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:994d/6022608/84f434771a80/giy069_reviewer_1_report_(original_submission).pdf?><caption><p>Click here for additional data file.</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="sup7" position="float" orientation="portrait"><label>Reviewer_2_Report_(Original_Submission) -- Benjamin Tully</label><caption><p>2/19/2018 Reviewed</p></caption><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="giy069_reviewer_2_report_(original_submission).pdf" position="float" orientation="portrait"><?suppdata-name giy069_reviewer_2_report_(original_submission).pdf?><?suppdata-size 26861?><?suppdata-md5 a699aeab083178337211ec99587c0d76?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:994d/6022608/a699aeab0831/giy069_reviewer_2_report_(original_submission).pdf?><caption><p>Click here for additional data file.</p></caption></media></supplementary-material><supplementary-material content-type="local-data" id="sup8" position="float" orientation="portrait"><label>Supplemental Files</label><media xmlns:xlink="http://www.w3.org/1999/xlink" xlink:href="giy069_supplemental_files.pdf" position="float" orientation="portrait"><?suppdata-name giy069_supplemental_files.pdf?><?suppdata-size 595610?><?suppdata-md5 44d650bd44a2b7dc39401e2f1dc2059c?><?suppdata-image-server-status NEVER_LOAD?><?suppdata-mime-type application?><?suppdata-mime-sub-type pdf?><?suppdata-cloudpmc-urn urn:app:994d/6022608/44d650bd44a2/giy069_supplemental_files.pdf?><caption><p>Click here for additional data file.</p></caption></media></supplementary-material></sec></body><back><ack><title>ACKNOWLEDGEMENTS</title><p>The authors thank Christopher Quince for contributing Python code, all genome binning software developers who participated in the CAMI challenge for their feedback on most relevant metrics, all developers who helped us run their binning software, and the Isaac Newton Institute in Cambridge for its hospitality under the program MTG.</p></ack><ref-list><title>References</title><ref id="bib1"><label>1.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Sczyrba</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Hofmann</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Belmann</surname><given-names>P</given-names></name><etal/></person-group>
<article-title>Critical assessment of metagenome interpretation – a benchmark of metagenomics software</article-title>. <source>Nat Methods</source>. <year>2017</year>;<volume>14</volume>(<issue>11</issue>):<fpage>1063</fpage>–<lpage>71</lpage>.<pub-id pub-id-type="pmid">28967888</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/nmeth.4458</pub-id><pub-id pub-id-type="pmcid">PMC5903868</pub-id></mixed-citation></ref><ref id="bib2"><label>2.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Belmann</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Bremges</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Dahms</surname><given-names>E</given-names></name>, <etal/></person-group>
<article-title>Benchmark data sets, software results and reference data for the first CAMI challenge</article-title>. <source>GigaScience Database</source>. <year>2017</year>
<comment><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://dx.doi.org/10.5524/100344">http://dx.doi.org/10.5524/100344</ext-link></comment>.</mixed-citation></ref><ref id="bib3"><label>3.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Parks</surname><given-names>HD</given-names></name>, <name name-style="western"><surname>Imelfort</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Skennerton</surname><given-names>T</given-names></name><etal/></person-group>
<article-title>CheckM: assessing the quality of microbial genomes recovered from isolates, single cells, and metagenomes</article-title>. <source>Genome Res</source>. <year>2015</year>;<volume>25</volume>(<issue>7</issue>):<fpage>1043</fpage>–<lpage>55</lpage>.<pub-id pub-id-type="pmid">25977477</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1101/gr.186072.114</pub-id><pub-id pub-id-type="pmcid">PMC4484387</pub-id></mixed-citation></ref><ref id="bib4"><label>4.</label><mixed-citation publication-type="journal">
<collab>CAMISIM metagenome simulator</collab>. <comment><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://github.com/CAMI-challenge/CAMISIM">https://github.com/CAMI-challenge/CAMISIM</ext-link>. Accessed 27 Apr 2018</comment>.</mixed-citation></ref><ref id="bib5"><label>5.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Fritz</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Hofmann</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Majda</surname><given-names>S</given-names></name><etal/></person-group>
<article-title>CAMISIM: Simulating metagenomes and microbial communities</article-title>. <source>bioRxiv</source>. <year>2018</year>;<fpage>300970</fpage>.<pub-id pub-id-type="doi" assigning-authority="pmc">10.1186/s40168-019-0633-6</pub-id><pub-id pub-id-type="pmcid">PMC6368784</pub-id><pub-id pub-id-type="pmid">30736849</pub-id></mixed-citation></ref><ref id="bib6"><label>6.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Langmead</surname><given-names>B</given-names></name>, <name name-style="western"><surname>Trapnell</surname><given-names>C</given-names></name>, <name name-style="western"><surname>Pop</surname><given-names>M</given-names></name>, <etal/></person-group>
<article-title>Ultrafast and memory-efficient alignment of short DNA sequences to the human genome</article-title>. <source>Genome Biol</source>. <year>2009</year>;<volume>10</volume>(<issue>3</issue>):<fpage>R25</fpage>–<lpage>10</lpage>.<pub-id pub-id-type="pmid">19261174</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1186/gb-2009-10-3-r25</pub-id><pub-id pub-id-type="pmcid">PMC2690996</pub-id></mixed-citation></ref><ref id="bib7"><label>7.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Li</surname><given-names>H</given-names></name>, <name name-style="western"><surname>Durbin</surname><given-names>R</given-names></name></person-group>
<article-title>Fast and accurate short read alignment with Burrows–Wheeler transform</article-title>. <source>Bioinformatics</source>. <year>2009</year>;<volume>25</volume>(<issue>14</issue>):<fpage>1754</fpage>–<lpage>60</lpage>.<pub-id pub-id-type="pmid">19451168</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bioinformatics/btp324</pub-id><pub-id pub-id-type="pmcid">PMC2705234</pub-id></mixed-citation></ref><ref id="bib8"><label>8.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Mikheenko</surname><given-names>A</given-names></name>, <name name-style="western"><surname>Saveliev</surname><given-names>V</given-names></name>, <name name-style="western"><surname>Gurevich</surname><given-names>A</given-names></name></person-group>
<article-title>MetaQUAST: evaluation of metagenome assemblies</article-title>. <source>Bioinformatics</source>. <year>2016</year>;<volume>32</volume>(<issue>7</issue>):<fpage>1088</fpage>–<lpage>90</lpage>.<pub-id pub-id-type="pmid">26614127</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bioinformatics/btv697</pub-id></mixed-citation></ref><ref id="bib9"><label>9.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Belmann</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Dröge</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Bremges</surname><given-names>A</given-names></name>, <etal/></person-group>
<article-title>Bioboxes: standardised containers for interchangeable bioinformatics software</article-title>. <source>GigaScience</source>. <year>2015</year>;<volume>4</volume>:<fpage>47</fpage>.<pub-id pub-id-type="pmid">26473029</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1186/s13742-015-0087-0</pub-id><pub-id pub-id-type="pmcid">PMC4607242</pub-id></mixed-citation></ref><ref id="bib10"><label>10.</label><mixed-citation publication-type="journal">
<comment>Bioboxes binning format. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://github.com/bioboxes/rfc/tree/master/data-format">https://github.com/bioboxes/rfc/tree/master/data-format</ext-link>. Accessed 27 Apr 2018</comment>.</mixed-citation></ref><ref id="bib11"><label>11.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Wu</surname><given-names>YW</given-names></name>, <name name-style="western"><surname>Simmons</surname><given-names>BA</given-names></name>, <name name-style="western"><surname>Singer</surname><given-names>SW</given-names></name></person-group>
<article-title>MaxBin 2.0: an automated binning algorithm to recover genomes from multiple metagenomic datasets</article-title>. <source>Bioinformatics</source>. <year>2016</year>;<volume>32</volume>:<fpage>605</fpage>–<lpage>7</lpage>.<pub-id pub-id-type="pmid">26515820</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bioinformatics/btv638</pub-id></mixed-citation></ref><ref id="bib12"><label>12.</label><mixed-citation publication-type="journal">
<comment>AMBER GitHub repository. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://github.com/CAMI-challenge/AMBER">https://github.com/CAMI-challenge/AMBER</ext-link>. Accessed 27 Apr 2018</comment>.</mixed-citation></ref><ref id="bib13"><label>13.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Baldi</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Brunak</surname><given-names>S</given-names></name>, <name name-style="western"><surname>Chauvin</surname><given-names>Y</given-names></name>, <etal/></person-group>
<article-title>Assessing the accuracy of prediction algorithms for classification: an overview</article-title>. <source>Bioinformatics</source>. <year>2000</year>;<volume>16</volume>:<fpage>412</fpage>–<lpage>24</lpage>.<pub-id pub-id-type="pmid">10871264</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bioinformatics/16.5.412</pub-id></mixed-citation></ref><ref id="bib14"><label>14.</label><mixed-citation publication-type="book">
<person-group person-group-type="author"><name name-style="western"><surname>Tsoumakas</surname><given-names>G</given-names></name>, <name name-style="western"><surname>Katakis</surname><given-names>I</given-names></name>, <name name-style="western"><surname>Vlahavas</surname><given-names>I</given-names></name></person-group>
<article-title>Mining multi-label data</article-title>. In: <person-group person-group-type="editor"><name name-style="western"><surname>Maimon</surname><given-names>O</given-names></name>, <name name-style="western"><surname>Rokach</surname><given-names>L</given-names></name></person-group>, eds. <source>Data Mining and Knowledge Discovery Handbook</source>. <publisher-name>Springer-Verlag</publisher-name>, <year>2010</year>.</mixed-citation></ref><ref id="bib15"><label>15.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Rand</surname><given-names>WM</given-names></name></person-group>
<article-title>Objective criteria for the evaluation of clustering methods</article-title>. <source>J Am Statist Assoc</source>. <year>1971</year>;<volume>66</volume>(<issue>336</issue>):<fpage>846</fpage>–<lpage>50</lpage>.</mixed-citation></ref><ref id="bib16"><label>16.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Alneberg</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Bjarnason</surname><given-names>BS</given-names></name>, <name name-style="western"><surname>de Bruijn</surname><given-names>I</given-names></name>, <etal/></person-group>
<article-title>Binning metagenomic contigs by coverage and composition</article-title>. <source>Nat Methods</source>. <year>2014</year>;<volume>11</volume>:<fpage>1144</fpage>–<lpage>6</lpage>.<pub-id pub-id-type="pmid">25218180</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/nmeth.3103</pub-id></mixed-citation></ref><ref id="bib17"><label>17.</label><mixed-citation publication-type="journal">
<article-title>AMBER example HTML visualization of calculated metrics</article-title>. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://cami-challenge.github.io/AMBER/">https://cami-challenge.github.io/AMBER/</ext-link>. <comment>Accessed 30 May 2018</comment>.</mixed-citation></ref><ref id="bib18"><label>18.</label><mixed-citation publication-type="book">
<collab>National Center for Biotechnology Information [Internet]</collab>. <publisher-loc>Bethesda, MD</publisher-loc>: <publisher-name>National Library of Medicine (US), National Center for Biotechnology Information</publisher-name>, <year>1988</year>.</mixed-citation></ref><ref id="bib19"><label>19.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Parks</surname><given-names>DH</given-names></name>, <name name-style="western"><surname>Rinke</surname><given-names>C</given-names></name>, <name name-style="western"><surname>Chuvochina</surname><given-names>M</given-names></name>, <etal/></person-group>
<article-title>Recovery of nearly 8,000 metagenome-assembled genomes substantially expands the tree of life</article-title>. <source>Nature Microbiology</source>. <year>2017</year>;<volume>2</volume>:<fpage>1533</fpage>–<lpage>42</lpage>.<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41564-017-0012-7</pub-id><pub-id pub-id-type="pmid">28894102</pub-id></mixed-citation></ref><ref id="bib20"><label>20.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Kang</surname><given-names>DD</given-names></name>, <name name-style="western"><surname>Froula</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Egan</surname><given-names>R</given-names></name>, <etal/></person-group>
<article-title>MetaBAT, an efficient tool for accurately reconstructing single genomes from complex microbial communities</article-title>. <source>PeerJ</source>. <year>2015</year>;<volume>3</volume>:<fpage>e1165</fpage>.<pub-id pub-id-type="pmid">26336640</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.7717/peerj.1165</pub-id><pub-id pub-id-type="pmcid">PMC4556158</pub-id></mixed-citation></ref><ref id="bib21"><label>21.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Strous</surname><given-names>M</given-names></name>, <name name-style="western"><surname>Kraft</surname><given-names>B</given-names></name>, <name name-style="western"><surname>Bisdorf</surname><given-names>R</given-names></name>, <etal/></person-group>
<article-title>The binning of metagenomic contigs for microbial physiology of mixed cultures</article-title>. <source>Frontiers in Microbiology</source>. <year>2012</year>;<volume>3</volume>:<fpage>410</fpage>.<pub-id pub-id-type="pmid">23227024</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.3389/fmicb.2012.00410</pub-id><pub-id pub-id-type="pmcid">PMC3514610</pub-id></mixed-citation></ref><ref id="bib22"><label>22.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Lin</surname><given-names>HH</given-names></name>, <name name-style="western"><surname>Liao</surname><given-names>YC</given-names></name></person-group>
<article-title>Accurate binning of metagenomic contigs via automated clustering sequences using information of genomic signatures and marker genes</article-title>. <source>Sci Rep</source>. <year>2016</year>;<volume>6</volume>:<fpage>24175</fpage>.<pub-id pub-id-type="pmid">27067514</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/srep24175</pub-id><pub-id pub-id-type="pmcid">PMC4828714</pub-id></mixed-citation></ref><ref id="bib23"><label>23.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Graham</surname><given-names>ED</given-names></name>, <name name-style="western"><surname>Heidelberg</surname><given-names>JF</given-names></name>, <name name-style="western"><surname>Tully</surname><given-names>BJ</given-names></name></person-group>
<article-title>BinSanity: unsupervised clustering of environmental microbial assemblies using coverage and affinity propagation</article-title>. <source>PeerJ</source>. <year>2016</year>;<volume>5</volume>:<fpage>e3035</fpage>.<pub-id pub-id-type="doi" assigning-authority="pmc">10.7717/peerj.3035</pub-id><pub-id pub-id-type="pmcid">PMC5345454</pub-id><pub-id pub-id-type="pmid">28289564</pub-id></mixed-citation></ref><ref id="bib24"><label>24.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Lu</surname><given-names>YY</given-names></name>, <name name-style="western"><surname>Chen</surname><given-names>T</given-names></name>, <name name-style="western"><surname>Fuhrman</surname><given-names>JA</given-names></name><etal/></person-group>
<article-title>COCACOLA: binning metagenomic contigs using sequence COmposition, read CoverAge, CO-alignment and paired-end read LinkAge</article-title>. <source>Bioinformatics</source>. <year>2017</year>;<volume>33</volume>(<issue>6</issue>):<fpage>791</fpage>–<lpage>8</lpage>.<pub-id pub-id-type="pmid">27256312</pub-id><pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/bioinformatics/btw290</pub-id></mixed-citation></ref><ref id="bib25"><label>25.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Sieber</surname><given-names>CMK</given-names></name>, <name name-style="western"><surname>Probst</surname><given-names>J</given-names></name>, <name name-style="western"><surname>Sharrar</surname><given-names>A</given-names></name><etal/></person-group>
<article-title>Recovery of genomes from metagenomes via a dereplication, aggregation, and scoring strategy</article-title>. <source>bioRxiv</source>. <year>2017</year>;<fpage>107789</fpage>.<pub-id pub-id-type="doi" assigning-authority="pmc">10.1038/s41564-018-0171-1</pub-id><pub-id pub-id-type="pmcid">PMC6786971</pub-id><pub-id pub-id-type="pmid">29807988</pub-id></mixed-citation></ref><ref id="bib26"><label>26.</label><mixed-citation publication-type="journal">
<collab>Critical Assessment of Metagenome Interpretation (CAMI)</collab>. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://www.cami-challenge.org">http://www.cami-challenge.org</ext-link>. <comment>Accessed 27 Apr 2018</comment>.</mixed-citation></ref><ref id="bib27"><label>27.</label><mixed-citation publication-type="journal">
<comment>CAMI benchmarking portal. <ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="https://data.cami-challenge.org">https://data.cami-challenge.org</ext-link>. Accessed 27 Apr</comment>
<year>2018</year>.</mixed-citation></ref><ref id="bib28"><label>28.</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><name name-style="western"><surname>Meyer</surname><given-names>F</given-names></name>, <name name-style="western"><surname>Hofmann</surname><given-names>P</given-names></name>, <name name-style="western"><surname>Belmann</surname><given-names>P</given-names></name><etal/></person-group>
<article-title>Supporting data for “AMBER: Assessment of Metagenome BinnERs.”</article-title>. <source>GigaScience Database</source>. <year>2018</year>
<comment><ext-link xmlns:xlink="http://www.w3.org/1999/xlink" ext-link-type="uri" xlink:href="http://dx.doi.org/10.5524/100454">http://dx.doi.org/10.5524/100454</ext-link></comment>.<pub-id pub-id-type="doi" assigning-authority="pmc">10.1093/gigascience/giy069</pub-id><pub-id pub-id-type="pmcid">PMC6022608</pub-id><pub-id pub-id-type="pmid">29893851</pub-id></mixed-citation></ref></ref-list></back></article>