r852 -> hybrid correction

This commit is contained in:
chhylp123
2026-02-11 16:38:27 -05:00
parent ec9a8b222d
commit d6ba102d42
73 changed files with 190525 additions and 180074 deletions
+256 -256
View File
@@ -1,257 +1,257 @@
# -*- coding: utf-8 -*-
import sys
import os
# -- General configuration ------------------------------------------------
# If your documentation needs a minimal Sphinx version, state it here.
#needs_sphinx = '1.0'
# Add any Sphinx extension module names here, as strings. They can be
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
# ones.
extensions = [
'sphinx.ext.todo',
'sphinx.ext.mathjax',
'sphinx.ext.ifconfig',
]
# Add any paths that contain templates here, relative to this directory.
templates_path = ['_templates']
# The suffix of source filenames.
source_suffix = '.rst'
# The encoding of source files.
#source_encoding = 'utf-8-sig'
# The master toctree document.
master_doc = 'index'
# General information about the project.
project = u'hifiasm'
copyright = u'2021, Haoyu Cheng, Heng Li'
# The version info for the project you're documenting, acts as replacement for
# |version| and |release|, also used in various other places throughout the
# built documents.
#
# The short X.Y version.
version = '0.16.0-r369'
# The full version, including alpha/beta/rc tags.
release = '0.16.0'
# The language for content autogenerated by Sphinx. Refer to documentation
# for a list of supported languages.
#language = None
# There are two options for replacing |today|: either, you set today to some
# non-false value, then it is used:
#today = ''
# Else, today_fmt is used as the format for a strftime call.
#today_fmt = '%B %d, %Y'
# List of patterns, relative to source directory, that match files and
# directories to ignore when looking for source files.
exclude_patterns = []
# The reST default role (used for this markup: `text`) to use for all
# documents.
#default_role = None
# If true, '()' will be appended to :func: etc. cross-reference text.
#add_function_parentheses = True
# If true, the current module name will be prepended to all description
# unit titles (such as .. function::).
#add_module_names = True
# If true, sectionauthor and moduleauthor directives will be shown in the
# output. They are ignored by default.
#show_authors = False
# The name of the Pygments (syntax highlighting) style to use.
pygments_style = 'sphinx'
# A list of ignored prefixes for module index sorting.
#modindex_common_prefix = []
# If true, keep warnings as "system message" paragraphs in the built documents.
#keep_warnings = False
# -- Options for HTML output ----------------------------------------------
# The theme to use for HTML and HTML Help pages. See the documentation for
# a list of builtin themes.
html_theme = 'default'
# Theme options are theme-specific and customize the look and feel of a theme
# further. For a list of options available for each theme, see the
# documentation.
#html_theme_options = {}
# Add any paths that contain custom themes here, relative to this directory.
#html_theme_path = []
# Build using the RTD theme, if not on RTD.
# https://read-the-docs.readthedocs.org/en/latest/theme.html
# https://github.com/snide/sphinx_rtd_theme
#
on_rtd = os.environ.get('READTHEDOCS', None) == 'True'
if not on_rtd: # only import and set the theme if we're building docs locally
import sphinx_rtd_theme
html_theme = 'sphinx_rtd_theme'
html_theme_path = [ "/usr/local/lib/python2.7/site-packages", ]
# The name for this set of Sphinx documents. If None, it defaults to
# "<project> v<release> documentation".
#html_title = None
# A shorter title for the navigation bar. Default is the same as html_title.
#html_short_title = None
# The name of an image file (relative to this directory) to place at the top
# of the sidebar.
#html_logo = None
# The name of an image file (within the static path) to use as favicon of the
# docs. This file should be a Windows icon file (.ico) being 16x16 or 32x32
# pixels large.
#html_favicon = None
# Add any paths that contain custom static files (such as style sheets) here,
# relative to this directory. They are copied after the builtin static files,
# so a file named "default.css" will overwrite the builtin "default.css".
html_static_path = ['_static']
# Add any extra paths that contain custom files (such as robots.txt or
# .htaccess) here, relative to this directory. These files are copied
# directly to the root of the documentation.
#html_extra_path = []
# If not '', a 'Last updated on:' timestamp is inserted at every page bottom,
# using the given strftime format.
#html_last_updated_fmt = '%b %d, %Y'
# If true, SmartyPants will be used to convert quotes and dashes to
# typographically correct entities.
#html_use_smartypants = True
# Custom sidebar templates, maps document names to template names.
#html_sidebars = {}
# Additional templates that should be rendered to pages, maps page names to
# template names.
#html_additional_pages = {}
# If false, no module index is generated.
#html_domain_indices = True
# If false, no index is generated.
#html_use_index = True
# If true, the index is split into individual pages for each letter.
#html_split_index = False
# If true, links to the reST sources are added to the pages.
#html_show_sourcelink = True
# If true, "Created using Sphinx" is shown in the HTML footer. Default is True.
#html_show_sphinx = True
# If true, "(C) Copyright ..." is shown in the HTML footer. Default is True.
#html_show_copyright = True
# If true, an OpenSearch description file will be output, and all pages will
# contain a <link> tag referring to it. The value of this option must be the
# base URL from which the finished HTML is served.
#html_use_opensearch = ''
# This is the file name suffix for HTML files (e.g. ".xhtml").
#html_file_suffix = None
# Output file base name for HTML help builder.
htmlhelp_basename = 'hifiasm-doc'
# -- Options for LaTeX output ---------------------------------------------
latex_elements = {
# The paper size ('letterpaper' or 'a4paper').
#'papersize': 'letterpaper',
# The font size ('10pt', '11pt' or '12pt').
#'pointsize': '10pt',
# Additional stuff for the LaTeX preamble.
#'preamble': '',
}
# Grouping the document tree into LaTeX files. List of tuples
# (source start file, target name, title,
# author, documentclass [howto, manual, or own class]).
latex_documents = [
('index', 'hifiasm.tex', u'hifiasm Documentation',
u'Haoyu Cheng, Heng Li', 'manual'),
]
# The name of an image file (relative to this directory) to place at the top of
# the title page.
#latex_logo = None
# For "manual" documents, if this is true, then toplevel headings are parts,
# not chapters.
#latex_use_parts = False
# If true, show page references after internal links.
#latex_show_pagerefs = False
# If true, show URL addresses after external links.
#latex_show_urls = False
# Documents to append as an appendix to all manuals.
#latex_appendices = []
# If false, no module index is generated.
#latex_domain_indices = True
# -- Options for manual page output ---------------------------------------
# One entry per manual page. List of tuples
# (source start file, name, description, authors, manual section).
man_pages = [
('index', 'hifiasm', u'hifiasm Documentation',
[u'Haoyu Cheng, Heng Li'], 1)
]
# If true, show URL addresses after external links.
#man_show_urls = False
# -- Options for Texinfo output -------------------------------------------
# Grouping the document tree into Texinfo files. List of tuples
# (source start file, target name, title, author,
# dir menu entry, description, category)
texinfo_documents = [
('index', 'hifiasm', u'hifiasm Documentation',
u'Haoyu Cheng, Heng Li', 'hifiasm', 'One line description of project.',
'Miscellaneous'),
]
# Documents to append as an appendix to all manuals.
#texinfo_appendices = []
# If false, no module index is generated.
#texinfo_domain_indices = True
# How to display URL addresses: 'footnote', 'no', or 'inline'.
#texinfo_show_urls = 'footnote'
# If true, do not generate a @detailmenu in the "Top" node's menu.
# -*- coding: utf-8 -*-
import sys
import os
# -- General configuration ------------------------------------------------
# If your documentation needs a minimal Sphinx version, state it here.
#needs_sphinx = '1.0'
# Add any Sphinx extension module names here, as strings. They can be
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
# ones.
extensions = [
'sphinx.ext.todo',
'sphinx.ext.mathjax',
'sphinx.ext.ifconfig',
]
# Add any paths that contain templates here, relative to this directory.
templates_path = ['_templates']
# The suffix of source filenames.
source_suffix = '.rst'
# The encoding of source files.
#source_encoding = 'utf-8-sig'
# The master toctree document.
master_doc = 'index'
# General information about the project.
project = u'hifiasm'
copyright = u'2021, Haoyu Cheng, Heng Li'
# The version info for the project you're documenting, acts as replacement for
# |version| and |release|, also used in various other places throughout the
# built documents.
#
# The short X.Y version.
version = '0.16.0-r369'
# The full version, including alpha/beta/rc tags.
release = '0.16.0'
# The language for content autogenerated by Sphinx. Refer to documentation
# for a list of supported languages.
#language = None
# There are two options for replacing |today|: either, you set today to some
# non-false value, then it is used:
#today = ''
# Else, today_fmt is used as the format for a strftime call.
#today_fmt = '%B %d, %Y'
# List of patterns, relative to source directory, that match files and
# directories to ignore when looking for source files.
exclude_patterns = []
# The reST default role (used for this markup: `text`) to use for all
# documents.
#default_role = None
# If true, '()' will be appended to :func: etc. cross-reference text.
#add_function_parentheses = True
# If true, the current module name will be prepended to all description
# unit titles (such as .. function::).
#add_module_names = True
# If true, sectionauthor and moduleauthor directives will be shown in the
# output. They are ignored by default.
#show_authors = False
# The name of the Pygments (syntax highlighting) style to use.
pygments_style = 'sphinx'
# A list of ignored prefixes for module index sorting.
#modindex_common_prefix = []
# If true, keep warnings as "system message" paragraphs in the built documents.
#keep_warnings = False
# -- Options for HTML output ----------------------------------------------
# The theme to use for HTML and HTML Help pages. See the documentation for
# a list of builtin themes.
html_theme = 'default'
# Theme options are theme-specific and customize the look and feel of a theme
# further. For a list of options available for each theme, see the
# documentation.
#html_theme_options = {}
# Add any paths that contain custom themes here, relative to this directory.
#html_theme_path = []
# Build using the RTD theme, if not on RTD.
# https://read-the-docs.readthedocs.org/en/latest/theme.html
# https://github.com/snide/sphinx_rtd_theme
#
on_rtd = os.environ.get('READTHEDOCS', None) == 'True'
if not on_rtd: # only import and set the theme if we're building docs locally
import sphinx_rtd_theme
html_theme = 'sphinx_rtd_theme'
html_theme_path = [ "/usr/local/lib/python2.7/site-packages", ]
# The name for this set of Sphinx documents. If None, it defaults to
# "<project> v<release> documentation".
#html_title = None
# A shorter title for the navigation bar. Default is the same as html_title.
#html_short_title = None
# The name of an image file (relative to this directory) to place at the top
# of the sidebar.
#html_logo = None
# The name of an image file (within the static path) to use as favicon of the
# docs. This file should be a Windows icon file (.ico) being 16x16 or 32x32
# pixels large.
#html_favicon = None
# Add any paths that contain custom static files (such as style sheets) here,
# relative to this directory. They are copied after the builtin static files,
# so a file named "default.css" will overwrite the builtin "default.css".
html_static_path = ['_static']
# Add any extra paths that contain custom files (such as robots.txt or
# .htaccess) here, relative to this directory. These files are copied
# directly to the root of the documentation.
#html_extra_path = []
# If not '', a 'Last updated on:' timestamp is inserted at every page bottom,
# using the given strftime format.
#html_last_updated_fmt = '%b %d, %Y'
# If true, SmartyPants will be used to convert quotes and dashes to
# typographically correct entities.
#html_use_smartypants = True
# Custom sidebar templates, maps document names to template names.
#html_sidebars = {}
# Additional templates that should be rendered to pages, maps page names to
# template names.
#html_additional_pages = {}
# If false, no module index is generated.
#html_domain_indices = True
# If false, no index is generated.
#html_use_index = True
# If true, the index is split into individual pages for each letter.
#html_split_index = False
# If true, links to the reST sources are added to the pages.
#html_show_sourcelink = True
# If true, "Created using Sphinx" is shown in the HTML footer. Default is True.
#html_show_sphinx = True
# If true, "(C) Copyright ..." is shown in the HTML footer. Default is True.
#html_show_copyright = True
# If true, an OpenSearch description file will be output, and all pages will
# contain a <link> tag referring to it. The value of this option must be the
# base URL from which the finished HTML is served.
#html_use_opensearch = ''
# This is the file name suffix for HTML files (e.g. ".xhtml").
#html_file_suffix = None
# Output file base name for HTML help builder.
htmlhelp_basename = 'hifiasm-doc'
# -- Options for LaTeX output ---------------------------------------------
latex_elements = {
# The paper size ('letterpaper' or 'a4paper').
#'papersize': 'letterpaper',
# The font size ('10pt', '11pt' or '12pt').
#'pointsize': '10pt',
# Additional stuff for the LaTeX preamble.
#'preamble': '',
}
# Grouping the document tree into LaTeX files. List of tuples
# (source start file, target name, title,
# author, documentclass [howto, manual, or own class]).
latex_documents = [
('index', 'hifiasm.tex', u'hifiasm Documentation',
u'Haoyu Cheng, Heng Li', 'manual'),
]
# The name of an image file (relative to this directory) to place at the top of
# the title page.
#latex_logo = None
# For "manual" documents, if this is true, then toplevel headings are parts,
# not chapters.
#latex_use_parts = False
# If true, show page references after internal links.
#latex_show_pagerefs = False
# If true, show URL addresses after external links.
#latex_show_urls = False
# Documents to append as an appendix to all manuals.
#latex_appendices = []
# If false, no module index is generated.
#latex_domain_indices = True
# -- Options for manual page output ---------------------------------------
# One entry per manual page. List of tuples
# (source start file, name, description, authors, manual section).
man_pages = [
('index', 'hifiasm', u'hifiasm Documentation',
[u'Haoyu Cheng, Heng Li'], 1)
]
# If true, show URL addresses after external links.
#man_show_urls = False
# -- Options for Texinfo output -------------------------------------------
# Grouping the document tree into Texinfo files. List of tuples
# (source start file, target name, title, author,
# dir menu entry, description, category)
texinfo_documents = [
('index', 'hifiasm', u'hifiasm Documentation',
u'Haoyu Cheng, Heng Li', 'hifiasm', 'One line description of project.',
'Miscellaneous'),
]
# Documents to append as an appendix to all manuals.
#texinfo_appendices = []
# If false, no module index is generated.
#texinfo_domain_indices = True
# How to display URL addresses: 'footnote', 'no', or 'inline'.
#texinfo_show_urls = 'footnote'
# If true, do not generate a @detailmenu in the "Top" node's menu.
#texinfo_no_detailmenu = False
+122 -122
View File
@@ -1,123 +1,123 @@
.. _faq:
Hifiasm FAQ
===========
.. contents::
:local:
How do I get contigs in FASTA?
-------------------------------------
The FASTA file can be produced from GFA as follows:
::
awk '/^S/{print ">"$2;print $3}' test.p_ctg.gfa > test.p_ctg.fa
Which types of assemblies should I use?
----------------------------------------
If parental data is available, ``*dip.hap*.p_ctg.gfa`` produced in trio-binning mode should be always preferred. Otherwise if Hi-C data is available, ``*hic.hap*.p_ctg.gfa`` produced in Hi-C mode is the best choice. Both trio-binning mode and Hi-C mode generate fully-phased assemblies.
If you only have HiFi reads, hifiasm in default outputs ``*bp.hap*.p_ctg.gfa``. The primary/alternate assemblies can be also produced by using ``--primary``. All these HiFi-only assemblies are not fully-phased. See `blog <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_ here for more details.
Are inbred/homozygous genomes supported?
--------------------------------------------------------------------------
Yes, please use the ``-l0`` option to disable purge duplication step.
Are diploid genomes supported?
-------------------------------------
Yes, most modules of hifiasm are designed for diploid samples, including purge duplication step, partially phased assembly and fully-phased assembly with trio-binning or Hi-C.
Are polyploid genomes supported?
-------------------------------------
The ``*r_utg.gfa`` and ``*p_utg.gfa`` are lossless so that they also work for polyploid genomes. However, currently the contig-generation modules of hifiasm are designed for diploid samples, which means both the partially phased assembly and the fully-phased assembly does not directly support polyploid genomes. If it is set to >2, the quality of primary assembly for polyploid genomes might be improved. Please use primary assembly for polyploid samples and run multiple rounds of purging steps using third-party tools such as purge_dups.
Why one Hi-C integrated assembly is larger than another one?
------------------------------------------------------------
For some samples like human male, the paternal haplotype should be larger than the maternal haplotype. However, if one assembly is much larger than another one, it should be the issues of hifiasm. To fix it, please set smaller value for ``-s`` (default: 0.55).
Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads. For instance, hifiasm prints the following information during assembly:
::
[M::purge_dups] homozygous read coverage threshold: 36
In this example, hifiasm identifies the coverage threshold for homozygous reads as ``36``. If it is significantly smaller than the homozygous coverage peak, hifiasm will generate two unbalanced assemblies. In this case, please set ``--hom-cov`` to homozygous coverage peak. Please note that tuning ``--hom-cov`` may affect ``*p_utg*gfa`` so that ``*hic*.bin`` should be deleted. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
For Hi-C integrated assembly, why the assembly size of both haplotypes are much larger than the estimated genome size?
------------------------------------------------------------------------------------------------------------------------------
It is likely that hifiasm misidentifies coverage threshold for homozygous reads. Hifiasm prints the following information for debugging:
::
[M::stat] # heterozygous bases: 645155110; # homozygous bases: 1495396634
If most bases of a diploid sample are homozygous, the coverage threshold is wrongly determined by hifiasm. For instance, hifiasm prints the following information during assembly:
::
[M::purge_dups] homozygous read coverage threshold: 36
In this example, hifiasm identifies the coverage threshold for homozygous reads as ``36``. If it is much smaller than homozygous coverage peak, hifiasm thinks most reads are homozygous and assign them to both assemblies, making both of them much larger than the estimated haploid genome size. In this case, please set ``--hom-cov`` to homozygous coverage peak. Please note that tuning ``--hom-cov`` may affect ``*p_utg*gfa`` so that ``*hic*.bin`` should be deleted. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
.. _hic-iss:
How can I tweak parameters to improve Hi-C integrated assembly?
---------------------------------------------------------------
Compared with the HiFi-only assembly or the trio-binning assembly, the Hi-C integrated assembly is a little bit more complex so that you need to take care of the results. See `Why one Hi-C integrated assembly is larger than another one?`_ and `For Hi-C integrated assembly, why the assembly size of both haplotypes are much larger than the estimated genome size?`_ for details on how to fix potential issues.
There are several other options that may affect the Hi-C integrated assembly. Increasing the values of ``--n-weight``, ``--n-perturb`` and ``--f-perturb`` may improve phasing results but takes longer time. However, tuning ``--l-msjoin`` is tricky. All these options do not affect ``*p_utg*gfa`` so that ``*hic*.bin`` can be reused.
.. _p-large:
Why the size of primary assembly or partially phased assembly is much larger than the estimated genome size?
---------------------------------------------------------------------------------------------------------------
It could be because the estimated genome size is incorrect. Another possibility is that hifiasm does not perform enough purging. Setting smaller value for ``-s`` (default: 0.55) or turning ``--hom-cov`` should be helpful. See :ref:`loginter` for more details.
.. _p-hamming:
Why the hamming error rate or the swith error rate of trio-binning assembly is very high?
---------------------------------------------------------------------------------------------------------------
In rare cases, a potential issue is that a few contigs may misjoin two haplotypes. For example, half of a contig come from mother while another half come from father. Such misjoined contigs can be fixed by manually breaking. The coordinates of problematic regions can be found by A-lines in GFA file or ``yak trioeval -e`` (see `issue 37 <https://github.com/chhylp123/hifiasm/issues/37>`_ for more details). However, if there are many misjoined contigs or the switch/hamming error rate reported by ``yak trioeval`` is very high, users should check if the parental data is correct (see `issue 130 <https://github.com/chhylp123/hifiasm/issues/130#issuecomment-862347943>`_ for more details).
Another possibility is that there are some unitigs in unitig graph misjoining two haplotypes. Such problematic unitigs might be ignored by the graph-binning strategy. Set smaller value for ``--t-occ`` forcedly remove unitig including unexpected haplotype-specific reads.
Why does hifiasm stuck or crash?
-------------------------------------
In most cases, it is caused by the low quality HiFi reads. A good HiFi dataset should have a k-mer plot like `issue10 <https://github.com/chhylp123/hifiasm/issues/10#issuecomment-616213684>`_ or `issue49 <https://github.com/chhylp123/hifiasm/issues/49#issue-729106823>`_. In contrast, low quality HiFi data often lead to weird k-mer plot like `issue93 <https://github.com/chhylp123/hifiasm/issues/93#issue-852259042>`_. Such weird k-mer plots usually indicate insufficient coverage or presence of contaminants. See :ref:`loginter` for more details. If the HiFi data look fine, please raise an issue at the `issue page <https://github.com/chhylp123/hifiasm/issues>`_.
What's the usage of different bin files in hifiasm?
----------------------------------------------------
``*ec.bin``, ``*ovlp.reverse.bin`` and ``*ovlp.source.bin`` save the results of error correction step. ``*hic*bin`` saves the results of Hi-C alignment. Please note that ``*hic*.bin`` should be deleted when tuning any parameters affecting ``*p_utg*gfa``. There are several parameters which does not change ``*p_utg*gfa``, including ``-s``, ``--seed``, ``--n-weight``, ``--n-perturb``, ``--f-perturb`` and ``--l-msjoin``. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
Can I generate HiFi-only assembly first, and then add Hi-C or trio data later?
----------------------------------------------------------------------------------------
Yes, the HiFi-only assembly, Hi-C phased assembly and trio-binning assembly share the same ``*ec.bin``, ``*ovlp.reverse.bin`` and ``*ovlp.source.bin``.
What is the minimum read coverage required for hifiasm?
-------------------------------------------------------
Usually >=13x HiFi reads per haplotype. Higher coverage might be able to improve the contiguity of assembly.
Why the primary assembly is more contiguous than the fully-phased assemblies and the partially phased assemblies (i.e. ``*.hap*.p_ctg.gfa``)?
----------------------------------------------------------------------------------------------------------------------------------------------------
For diploid samples, primary assembly usually has greater N50 but at the expense of highly fragmented alternate assembly. From the method view, the primary assembly has an extra joining step, which joins two haplotypes to make primary assembly more contiguous.
When producing fully-phased assemblies and partially phased assemblies, hifiasm is designed to keep both haplotypes contiguous. It is important for many downstream applications like SV calling.
My assembly is fragmented or not contiguous enough, how do I improve it?
--------------------------------------------------------------------------
Raising ``-D`` or ``-N`` may improve the resolution of repetitive regions but takes longer time. These two options affect all types of assemblies and usually do not have a negative impact on the assembly quality. In contrast, ``--purge-max`` only affects primary assembly. Setting larger value for ``--purge-max`` makes primary assembly more contiguous but may collapse repeats or segmental duplications.
If the assembly is too fragmented, users should check if HiFi data is good enough. See `Why does hifiasm stuck or crash?`_ for details.
How do I avoid misassemblies?
--------------------------------------------------------------------------
.. _faq:
Hifiasm FAQ
===========
.. contents::
:local:
How do I get contigs in FASTA?
-------------------------------------
The FASTA file can be produced from GFA as follows:
::
awk '/^S/{print ">"$2;print $3}' test.p_ctg.gfa > test.p_ctg.fa
Which types of assemblies should I use?
----------------------------------------
If parental data is available, ``*dip.hap*.p_ctg.gfa`` produced in trio-binning mode should be always preferred. Otherwise if Hi-C data is available, ``*hic.hap*.p_ctg.gfa`` produced in Hi-C mode is the best choice. Both trio-binning mode and Hi-C mode generate fully-phased assemblies.
If you only have HiFi reads, hifiasm in default outputs ``*bp.hap*.p_ctg.gfa``. The primary/alternate assemblies can be also produced by using ``--primary``. All these HiFi-only assemblies are not fully-phased. See `blog <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_ here for more details.
Are inbred/homozygous genomes supported?
--------------------------------------------------------------------------
Yes, please use the ``-l0`` option to disable purge duplication step.
Are diploid genomes supported?
-------------------------------------
Yes, most modules of hifiasm are designed for diploid samples, including purge duplication step, partially phased assembly and fully-phased assembly with trio-binning or Hi-C.
Are polyploid genomes supported?
-------------------------------------
The ``*r_utg.gfa`` and ``*p_utg.gfa`` are lossless so that they also work for polyploid genomes. However, currently the contig-generation modules of hifiasm are designed for diploid samples, which means both the partially phased assembly and the fully-phased assembly does not directly support polyploid genomes. If it is set to >2, the quality of primary assembly for polyploid genomes might be improved. Please use primary assembly for polyploid samples and run multiple rounds of purging steps using third-party tools such as purge_dups.
Why one Hi-C integrated assembly is larger than another one?
------------------------------------------------------------
For some samples like human male, the paternal haplotype should be larger than the maternal haplotype. However, if one assembly is much larger than another one, it should be the issues of hifiasm. To fix it, please set smaller value for ``-s`` (default: 0.55).
Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads. For instance, hifiasm prints the following information during assembly:
::
[M::purge_dups] homozygous read coverage threshold: 36
In this example, hifiasm identifies the coverage threshold for homozygous reads as ``36``. If it is significantly smaller than the homozygous coverage peak, hifiasm will generate two unbalanced assemblies. In this case, please set ``--hom-cov`` to homozygous coverage peak. Please note that tuning ``--hom-cov`` may affect ``*p_utg*gfa`` so that ``*hic*.bin`` should be deleted. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
For Hi-C integrated assembly, why the assembly size of both haplotypes are much larger than the estimated genome size?
------------------------------------------------------------------------------------------------------------------------------
It is likely that hifiasm misidentifies coverage threshold for homozygous reads. Hifiasm prints the following information for debugging:
::
[M::stat] # heterozygous bases: 645155110; # homozygous bases: 1495396634
If most bases of a diploid sample are homozygous, the coverage threshold is wrongly determined by hifiasm. For instance, hifiasm prints the following information during assembly:
::
[M::purge_dups] homozygous read coverage threshold: 36
In this example, hifiasm identifies the coverage threshold for homozygous reads as ``36``. If it is much smaller than homozygous coverage peak, hifiasm thinks most reads are homozygous and assign them to both assemblies, making both of them much larger than the estimated haploid genome size. In this case, please set ``--hom-cov`` to homozygous coverage peak. Please note that tuning ``--hom-cov`` may affect ``*p_utg*gfa`` so that ``*hic*.bin`` should be deleted. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
.. _hic-iss:
How can I tweak parameters to improve Hi-C integrated assembly?
---------------------------------------------------------------
Compared with the HiFi-only assembly or the trio-binning assembly, the Hi-C integrated assembly is a little bit more complex so that you need to take care of the results. See `Why one Hi-C integrated assembly is larger than another one?`_ and `For Hi-C integrated assembly, why the assembly size of both haplotypes are much larger than the estimated genome size?`_ for details on how to fix potential issues.
There are several other options that may affect the Hi-C integrated assembly. Increasing the values of ``--n-weight``, ``--n-perturb`` and ``--f-perturb`` may improve phasing results but takes longer time. However, tuning ``--l-msjoin`` is tricky. All these options do not affect ``*p_utg*gfa`` so that ``*hic*.bin`` can be reused.
.. _p-large:
Why the size of primary assembly or partially phased assembly is much larger than the estimated genome size?
---------------------------------------------------------------------------------------------------------------
It could be because the estimated genome size is incorrect. Another possibility is that hifiasm does not perform enough purging. Setting smaller value for ``-s`` (default: 0.55) or turning ``--hom-cov`` should be helpful. See :ref:`loginter` for more details.
.. _p-hamming:
Why the hamming error rate or the swith error rate of trio-binning assembly is very high?
---------------------------------------------------------------------------------------------------------------
In rare cases, a potential issue is that a few contigs may misjoin two haplotypes. For example, half of a contig come from mother while another half come from father. Such misjoined contigs can be fixed by manually breaking. The coordinates of problematic regions can be found by A-lines in GFA file or ``yak trioeval -e`` (see `issue 37 <https://github.com/chhylp123/hifiasm/issues/37>`_ for more details). However, if there are many misjoined contigs or the switch/hamming error rate reported by ``yak trioeval`` is very high, users should check if the parental data is correct (see `issue 130 <https://github.com/chhylp123/hifiasm/issues/130#issuecomment-862347943>`_ for more details).
Another possibility is that there are some unitigs in unitig graph misjoining two haplotypes. Such problematic unitigs might be ignored by the graph-binning strategy. Set smaller value for ``--t-occ`` forcedly remove unitig including unexpected haplotype-specific reads.
Why does hifiasm stuck or crash?
-------------------------------------
In most cases, it is caused by the low quality HiFi reads. A good HiFi dataset should have a k-mer plot like `issue10 <https://github.com/chhylp123/hifiasm/issues/10#issuecomment-616213684>`_ or `issue49 <https://github.com/chhylp123/hifiasm/issues/49#issue-729106823>`_. In contrast, low quality HiFi data often lead to weird k-mer plot like `issue93 <https://github.com/chhylp123/hifiasm/issues/93#issue-852259042>`_. Such weird k-mer plots usually indicate insufficient coverage or presence of contaminants. See :ref:`loginter` for more details. If the HiFi data look fine, please raise an issue at the `issue page <https://github.com/chhylp123/hifiasm/issues>`_.
What's the usage of different bin files in hifiasm?
----------------------------------------------------
``*ec.bin``, ``*ovlp.reverse.bin`` and ``*ovlp.source.bin`` save the results of error correction step. ``*hic*bin`` saves the results of Hi-C alignment. Please note that ``*hic*.bin`` should be deleted when tuning any parameters affecting ``*p_utg*gfa``. There are several parameters which does not change ``*p_utg*gfa``, including ``-s``, ``--seed``, ``--n-weight``, ``--n-perturb``, ``--f-perturb`` and ``--l-msjoin``. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
Can I generate HiFi-only assembly first, and then add Hi-C or trio data later?
----------------------------------------------------------------------------------------
Yes, the HiFi-only assembly, Hi-C phased assembly and trio-binning assembly share the same ``*ec.bin``, ``*ovlp.reverse.bin`` and ``*ovlp.source.bin``.
What is the minimum read coverage required for hifiasm?
-------------------------------------------------------
Usually >=13x HiFi reads per haplotype. Higher coverage might be able to improve the contiguity of assembly.
Why the primary assembly is more contiguous than the fully-phased assemblies and the partially phased assemblies (i.e. ``*.hap*.p_ctg.gfa``)?
----------------------------------------------------------------------------------------------------------------------------------------------------
For diploid samples, primary assembly usually has greater N50 but at the expense of highly fragmented alternate assembly. From the method view, the primary assembly has an extra joining step, which joins two haplotypes to make primary assembly more contiguous.
When producing fully-phased assemblies and partially phased assemblies, hifiasm is designed to keep both haplotypes contiguous. It is important for many downstream applications like SV calling.
My assembly is fragmented or not contiguous enough, how do I improve it?
--------------------------------------------------------------------------
Raising ``-D`` or ``-N`` may improve the resolution of repetitive regions but takes longer time. These two options affect all types of assemblies and usually do not have a negative impact on the assembly quality. In contrast, ``--purge-max`` only affects primary assembly. Setting larger value for ``--purge-max`` makes primary assembly more contiguous but may collapse repeats or segmental duplications.
If the assembly is too fragmented, users should check if HiFi data is good enough. See `Why does hifiasm stuck or crash?`_ for details.
How do I avoid misassemblies?
--------------------------------------------------------------------------
Set smaller value for ``--purge-max``, ``-s`` and ``-O``, or use the ``-u`` option.
+16 -16
View File
@@ -1,16 +1,16 @@
.. _hic-assembly:
Hi-C Integrated Assembly
========================
Hifiasm can generate a pair of haplotype-resolved assemblies with paired-end Hi-C reads::
hifiasm -o NA12878.asm -t32 --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
In this mode, each contig is supposed to be a haplotig, which by definition comes from one parental haplotype only. Hifiasm often puts all contigs from the same parental chromosome in one assembly. It has cleanly separated chrX and chrY for a human male dataset. Nonetheless, phasing across centromeres is challenging. Hifiasm is often able to phase entire chromosomes but it may fail in rare cases. Also, contigs from different parental chromosomes are randomly mixed as it is just not possible to phase across chromosomes with Hi-C. Hifiasm does not perform scaffolding for now. You need to run a standalone scaffolder such as SALSA or 3D-DNA to scaffold phased haplotigs.
For samples with high heterozygosity rate, a common issue is that one assembly is much larger than another one. To fix this issue, please set smaller value for ``-s`` (default: 0.55). Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads. In this case, please set ``--hom-cov`` to homozygous coverage peak. See :ref:`hic-iss` for more details.
At the first run, hifiasm saves the alignment of Hi-C reads to disk as ``*hic*.bin``. It reuses the saved results to avoid Hi-C alignment next time. Please note that ``*hic*.bin`` should be deleted when tuning any parameters affecting ``*p_utg*gfa``. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically. There are several parameters which do not change ``*p_utg*gfa``, including ``-s``, ``--seed``, ``--n-weight``, ``--n-perturb``, ``--f-perturb`` and ``--l-msjoin``.
.. _hic-assembly:
Hi-C Integrated Assembly
========================
Hifiasm can generate a pair of haplotype-resolved assemblies with paired-end Hi-C reads::
hifiasm -o NA12878.asm -t32 --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
In this mode, each contig is supposed to be a haplotig, which by definition comes from one parental haplotype only. Hifiasm often puts all contigs from the same parental chromosome in one assembly. It has cleanly separated chrX and chrY for a human male dataset. Nonetheless, phasing across centromeres is challenging. Hifiasm is often able to phase entire chromosomes but it may fail in rare cases. Also, contigs from different parental chromosomes are randomly mixed as it is just not possible to phase across chromosomes with Hi-C. Hifiasm does not perform scaffolding for now. You need to run a standalone scaffolder such as SALSA or 3D-DNA to scaffold phased haplotigs.
For samples with high heterozygosity rate, a common issue is that one assembly is much larger than another one. To fix this issue, please set smaller value for ``-s`` (default: 0.55). Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads. In this case, please set ``--hom-cov`` to homozygous coverage peak. See :ref:`hic-iss` for more details.
At the first run, hifiasm saves the alignment of Hi-C reads to disk as ``*hic*.bin``. It reuses the saved results to avoid Hi-C alignment next time. Please note that ``*hic*.bin`` should be deleted when tuning any parameters affecting ``*p_utg*gfa``. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically. There are several parameters which do not change ``*p_utg*gfa``, including ``-s``, ``--seed``, ``--n-weight``, ``--n-perturb``, ``--f-perturb`` and ``--l-msjoin``.
+80 -80
View File
@@ -1,80 +1,80 @@
Hifiasm
=======
.. toctree::
:hidden:
pa-assembly
trio-assembly
hic-assembly
interpreting-output
faq
parameter-reference
`Hifiasm <https://github.com/chhylp123/hifiasm>`_ is a fast haplotype-resolved de novo assembler for PacBio HiFi reads. It can assemble a human genome in several hours and assemble a ~30Gb California redwood genome in a few days. Hifiasm emits partially phased assemblies of quality competitive with the best assemblers. Given parental short reads or Hi-C data, it produces arguably the best haplotype-resolved assemblies so far.
Publications
============
Hifiasm
Haoyu Cheng, Gregory T. Concepcion, Xiaowen Feng, Haowen Zhang & Heng Li.
`Haplotype-resolved de novo assembly using phased assembly graphs with hifiasm <https://doi.org/10.1038/s41592-020-01056-5>`_. Nature Methods. (2021).
Install
=======
The easiest way to get started is to download a `release <https://github.com/chhylp123/hifiasm/releases>`_. Please report any issues on `github issues <https://github.com/chhylp123/hifiasm/issues>`_ page.
In addition, the latest unreleased version can be found from github:
::
git clone https://github.com/chhylp123/hifiasm
cd hifiasm && make
Another way is to install hifiasm via `bioconda <https://anaconda.org/bioconda/hifiasm>`_:
::
conda install -c bioconda hifiasm
Assembly Concepts
=================
There are different types of assemblies which are commonly used in practice (see
`details <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_).
Hifiasm produces primary/alternate assemblies or partially phased assemblies
only with HiFi reads. Given Hi-C data or trio-binning data, hifiasm produces
contiguous fully-phased assemblies, i.e. haplotype-resolved assemblies.
Why Hifiasm?
============
* Hifiasm delivers high-quality assemblies. It tends to generate longer contigs
and resolve more segmental duplications than other assemblers.
* Given Hi-C reads or short reads from the parents, hifiasm can produce overall the best
haplotype-resolved assembly so far. It is the assembler of choice by the
`Human Pangenome Project <https://humanpangenome.org/>`_ for the first batch of samples.
* Hifiasm can purge duplications between haplotigs without relying on
third-party tools such as purge\_dups. Hifiasm does not need polishing tools
like pilon or racon, either. This simplifies the assembly pipeline and saves
running time.
* Hifiasm is fast. It can assemble a human genome in half a day and assemble a
~30Gb redwood genome in three days. No genome is too large for hifiasm.
* Hifiasm is trivial to install and easy to use. It does not required Python,
R or C++11 compilers, and can be compiled into a single executable. The
default setting works well with a variety of genomes.
Learn
=====
* :ref:`HiFi-only Assembly <pa-assembly>` - Assembling HiFi reads without additional data types
* :ref:`Trio-binning Assembly <trio-assembly>` - Producing fully phased assemblies with HiFi and trio-binning data
* :ref:`Hi-C Integrated Assembly <hic-assembly>` - Producing fully phased assemblies with HiFi and Hi-C data
* :ref:`Hifiasm Output <interpreting-output>` - Interpreting results
* :ref:`Hifiasm FAQ <faq>` - Frequently asked questions
* :ref:`Hifiasm Parameters <parameter-reference>` - Parameter reference of hifiasm
Hifiasm
=======
.. toctree::
:hidden:
pa-assembly
trio-assembly
hic-assembly
interpreting-output
faq
parameter-reference
`Hifiasm <https://github.com/chhylp123/hifiasm>`_ is a fast haplotype-resolved de novo assembler for PacBio HiFi reads. It can assemble a human genome in several hours and assemble a ~30Gb California redwood genome in a few days. Hifiasm emits partially phased assemblies of quality competitive with the best assemblers. Given parental short reads or Hi-C data, it produces arguably the best haplotype-resolved assemblies so far.
Publications
============
Hifiasm
Haoyu Cheng, Gregory T. Concepcion, Xiaowen Feng, Haowen Zhang & Heng Li.
`Haplotype-resolved de novo assembly using phased assembly graphs with hifiasm <https://doi.org/10.1038/s41592-020-01056-5>`_. Nature Methods. (2021).
Install
=======
The easiest way to get started is to download a `release <https://github.com/chhylp123/hifiasm/releases>`_. Please report any issues on `github issues <https://github.com/chhylp123/hifiasm/issues>`_ page.
In addition, the latest unreleased version can be found from github:
::
git clone https://github.com/chhylp123/hifiasm
cd hifiasm && make
Another way is to install hifiasm via `bioconda <https://anaconda.org/bioconda/hifiasm>`_:
::
conda install -c bioconda hifiasm
Assembly Concepts
=================
There are different types of assemblies which are commonly used in practice (see
`details <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_).
Hifiasm produces primary/alternate assemblies or partially phased assemblies
only with HiFi reads. Given Hi-C data or trio-binning data, hifiasm produces
contiguous fully-phased assemblies, i.e. haplotype-resolved assemblies.
Why Hifiasm?
============
* Hifiasm delivers high-quality assemblies. It tends to generate longer contigs
and resolve more segmental duplications than other assemblers.
* Given Hi-C reads or short reads from the parents, hifiasm can produce overall the best
haplotype-resolved assembly so far. It is the assembler of choice by the
`Human Pangenome Project <https://humanpangenome.org/>`_ for the first batch of samples.
* Hifiasm can purge duplications between haplotigs without relying on
third-party tools such as purge\_dups. Hifiasm does not need polishing tools
like pilon or racon, either. This simplifies the assembly pipeline and saves
running time.
* Hifiasm is fast. It can assemble a human genome in half a day and assemble a
~30Gb redwood genome in three days. No genome is too large for hifiasm.
* Hifiasm is trivial to install and easy to use. It does not required Python,
R or C++11 compilers, and can be compiled into a single executable. The
default setting works well with a variety of genomes.
Learn
=====
* :ref:`HiFi-only Assembly <pa-assembly>` - Assembling HiFi reads without additional data types
* :ref:`Trio-binning Assembly <trio-assembly>` - Producing fully phased assemblies with HiFi and trio-binning data
* :ref:`Hi-C Integrated Assembly <hic-assembly>` - Producing fully phased assemblies with HiFi and Hi-C data
* :ref:`Hifiasm Output <interpreting-output>` - Interpreting results
* :ref:`Hifiasm FAQ <faq>` - Frequently asked questions
* :ref:`Hifiasm Parameters <parameter-reference>` - Parameter reference of hifiasm
+102 -102
View File
@@ -1,102 +1,102 @@
.. _interpreting-output:
Hifiasm Output
===============
.. _outfile:
Output files
---------------------------------------
In general, hifiasm generates the following assembly graphs in the GFA format:
* ```prefix`.r_utg.gfa``: haplotype-resolved raw unitig graph. This graph keeps all haplotype information.
* ```prefix`.p_utg.gfa``: haplotype-resolved processed unitig graph without small bubbles. Small bubbles might be caused by somatic mutations or noise in data, which are not the real haplotype information. Hifiasm automatically pops such small bubbles based on coverage. The option ``--hom-cov`` affects the result. See :ref:`homozygous coverage setting <homcov>` for more details. In addition, the option ``-p`` forcedly pops bubbles.
* ```prefix`.p_ctg.gfa``: assembly graph of primary contigs. This graph includes a complete assembly with long stretches of phased blocks.
* ```prefix`.a_ctg.gfa``: assembly graph of alternate contigs. This graph consists of all contigs that are discarded in primary contig graph.
* ```prefix`.*hap*.p_ctg.gfa``: phased contig graph. This graph keeps the phased contigs.
Hifiasm outputs ``*.r_utg.gfa`` and ``*.p_utg.gfa`` in any cases. Specifically, hifiasm outputs the following assembly graphs in trio-binning mode:
* ```prefix`.dip.hap1.p_ctg.gfa``: fully phased paternal/haplotype1 contig graph keeping the phased paternal/haplotype1 assembly.
* ```prefix`.dip.hap2.p_ctg.gfa``: fully phased maternal/haplotype2 contig graph keeping the phased maternal/haplotype2 assembly.
With Hi-C partition options, hifiasm outputs:
* ```prefix`.hic.p_ctg.gfa``: assembly graph of primary contigs.
* ```prefix`.hic.hap1.p_ctg.gfa``: fully phased contig graph of haplotype1 where each contig is fully phased.
* ```prefix`.hic.hap2.p_ctg.gfa``: fully phased contig graph of haplotype2 where each contig is fully phased.
* ```prefix`.hic.a_ctg.gfa`` (optional with ``--primary``): assembly graph of alternate contigs.
Hifiasm generates the following assembly graphs only with HiFi reads in default:
* ```prefix`.bp.p_ctg.gfa``: assembly graph of primary contigs.
* ```prefix`.bp.hap1.p_ctg.gfa``: partially phased contig graph of haplotype1.
* ```prefix`.bp.hap2.p_ctg.gfa``: partially phased contig graph of haplotype2.
If the option ``--primary`` or ``-l0`` is specified, hifiasm outputs:
* ```prefix`.p_ctg.gfa``: assembly graph of primary contigs.
* ```prefix`.a_ctg.gfa``: assembly graph of alternate contigs.
For each graph, hifiasm also outputs a simplified version (``*noseq*gfa``) without sequences for the ease of visualization. The coordinates of low quality regions are written to ``*lowQ.bed`` in BED format.
The concepts of different types of assemblies can be found `here <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_.
.. _outformat:
Output file formats
---------------------------------------
Hifiasm broadly follows the specification for `GFA 1.0 <https://github.com/GFA-spec/GFA-spec/blob/master/GFA1.md>`_. There are several fields that are specifically used by hifiasm. For ``S`` segment line:
* ``rd:i:``: read coverage. It is calculated by the reads coming from the same contig/unitig.
Hifiasm outputs ``A`` lines including the information of reads which are used to construct contig/unitig. Each ``A`` line is plain-text, tab-separated, and the columns appear in the following order:
.. list-table::
:widths: 10 25 50
:header-rows: 1
* - Col
- Type
- Description
* - 1
- string
- Should be always ``A``
* - 2
- string
- Contig/unitig name
* - 3
- int
- Contig/unitig start coordinate of subregion constructed by read
* - 4
- char
- Read strand: "+" or "-"
* - 5
- string
- Read name
* - 6
- int
- Read start coordinate of subregion which is used to construct contig/unitig
* - 7
- int
- Read end coordinate of subregion which is used to construct contig/unitig
* - 8
- id:i:int
- Read ID
* - 9
- HG:A:char
- Haplotype status of read. ``HG:A:a``, ``HG:A:p``, ``HG:A:m`` indicate read is non-binnable, father/hap1-specific and mother/hap2-specific, respectively.
.. _loginter:
Hifiasm log interpretation
---------------------------------------
Hifiasm prints several information for quick debugging, including:
.. _homcov:
* k-mer plot: showing how many k-mers appear a certain number of times. For homozygous samples, there should be one peak around read coverage. For heterozygous samples, there should two peaks, where the smaller peak is around the heterozygous read coverage and the larger peak is around the homozygous read coverage. For example, `issue10 <https://github.com/chhylp123/hifiasm/issues/10#issuecomment-616213684>`_ indicates the heterozygous read coverage and the homozygous read coverage are 28 and 57, respectively. `Issue49 <https://github.com/chhylp123/hifiasm/issues/49#issue-729106823>`_ is another good example. Weird k-mer plot like `issue93 <https://github.com/chhylp123/hifiasm/issues/93#issue-852259042>`_ is often caused by insufficient coverage or presence of contaminants.
* homozygous coverage: coverage threshold for homozygous reads. Hifiasm prints it as: ``[M::purge_dups] homozygous read coverage threshold: X``. If it is not around homozygous coverage, the final assembly might be either too large or too small. To fix this issue, please set ``--hom-cov`` to homozygous coverage.
* number of het/hom bases: how many bases in unitig graph are heterozygous and homozygous during Hi-C phased assembly. Hifiasm prints it as: ``[M::stat] # heterozygous bases: X; # homozygous bases: Y``. Given a heterozygous sample, if there are much more homozygous bases than heterozygous bases, hifiasm fails to identify correct coverage threshold for homozygous reads. In this case, please set ``--hom-cov`` to homozygous coverage.
.. _interpreting-output:
Hifiasm Output
===============
.. _outfile:
Output files
---------------------------------------
In general, hifiasm generates the following assembly graphs in the GFA format:
* ```prefix`.r_utg.gfa``: haplotype-resolved raw unitig graph. This graph keeps all haplotype information.
* ```prefix`.p_utg.gfa``: haplotype-resolved processed unitig graph without small bubbles. Small bubbles might be caused by somatic mutations or noise in data, which are not the real haplotype information. Hifiasm automatically pops such small bubbles based on coverage. The option ``--hom-cov`` affects the result. See :ref:`homozygous coverage setting <homcov>` for more details. In addition, the option ``-p`` forcedly pops bubbles.
* ```prefix`.p_ctg.gfa``: assembly graph of primary contigs. This graph includes a complete assembly with long stretches of phased blocks.
* ```prefix`.a_ctg.gfa``: assembly graph of alternate contigs. This graph consists of all contigs that are discarded in primary contig graph.
* ```prefix`.*hap*.p_ctg.gfa``: phased contig graph. This graph keeps the phased contigs.
Hifiasm outputs ``*.r_utg.gfa`` and ``*.p_utg.gfa`` in any cases. Specifically, hifiasm outputs the following assembly graphs in trio-binning mode:
* ```prefix`.dip.hap1.p_ctg.gfa``: fully phased paternal/haplotype1 contig graph keeping the phased paternal/haplotype1 assembly.
* ```prefix`.dip.hap2.p_ctg.gfa``: fully phased maternal/haplotype2 contig graph keeping the phased maternal/haplotype2 assembly.
With Hi-C partition options, hifiasm outputs:
* ```prefix`.hic.p_ctg.gfa``: assembly graph of primary contigs.
* ```prefix`.hic.hap1.p_ctg.gfa``: fully phased contig graph of haplotype1 where each contig is fully phased.
* ```prefix`.hic.hap2.p_ctg.gfa``: fully phased contig graph of haplotype2 where each contig is fully phased.
* ```prefix`.hic.a_ctg.gfa`` (optional with ``--primary``): assembly graph of alternate contigs.
Hifiasm generates the following assembly graphs only with HiFi reads in default:
* ```prefix`.bp.p_ctg.gfa``: assembly graph of primary contigs.
* ```prefix`.bp.hap1.p_ctg.gfa``: partially phased contig graph of haplotype1.
* ```prefix`.bp.hap2.p_ctg.gfa``: partially phased contig graph of haplotype2.
If the option ``--primary`` or ``-l0`` is specified, hifiasm outputs:
* ```prefix`.p_ctg.gfa``: assembly graph of primary contigs.
* ```prefix`.a_ctg.gfa``: assembly graph of alternate contigs.
For each graph, hifiasm also outputs a simplified version (``*noseq*gfa``) without sequences for the ease of visualization. The coordinates of low quality regions are written to ``*lowQ.bed`` in BED format.
The concepts of different types of assemblies can be found `here <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_.
.. _outformat:
Output file formats
---------------------------------------
Hifiasm broadly follows the specification for `GFA 1.0 <https://github.com/GFA-spec/GFA-spec/blob/master/GFA1.md>`_. There are several fields that are specifically used by hifiasm. For ``S`` segment line:
* ``rd:i:``: read coverage. It is calculated by the reads coming from the same contig/unitig.
Hifiasm outputs ``A`` lines including the information of reads which are used to construct contig/unitig. Each ``A`` line is plain-text, tab-separated, and the columns appear in the following order:
.. list-table::
:widths: 10 25 50
:header-rows: 1
* - Col
- Type
- Description
* - 1
- string
- Should be always ``A``
* - 2
- string
- Contig/unitig name
* - 3
- int
- Contig/unitig start coordinate of subregion constructed by read
* - 4
- char
- Read strand: "+" or "-"
* - 5
- string
- Read name
* - 6
- int
- Read start coordinate of subregion which is used to construct contig/unitig
* - 7
- int
- Read end coordinate of subregion which is used to construct contig/unitig
* - 8
- id:i:int
- Read ID
* - 9
- HG:A:char
- Haplotype status of read. ``HG:A:a``, ``HG:A:p``, ``HG:A:m`` indicate read is non-binnable, father/hap1-specific and mother/hap2-specific, respectively.
.. _loginter:
Hifiasm log interpretation
---------------------------------------
Hifiasm prints several information for quick debugging, including:
.. _homcov:
* k-mer plot: showing how many k-mers appear a certain number of times. For homozygous samples, there should be one peak around read coverage. For heterozygous samples, there should two peaks, where the smaller peak is around the heterozygous read coverage and the larger peak is around the homozygous read coverage. For example, `issue10 <https://github.com/chhylp123/hifiasm/issues/10#issuecomment-616213684>`_ indicates the heterozygous read coverage and the homozygous read coverage are 28 and 57, respectively. `Issue49 <https://github.com/chhylp123/hifiasm/issues/49#issue-729106823>`_ is another good example. Weird k-mer plot like `issue93 <https://github.com/chhylp123/hifiasm/issues/93#issue-852259042>`_ is often caused by insufficient coverage or presence of contaminants.
* homozygous coverage: coverage threshold for homozygous reads. Hifiasm prints it as: ``[M::purge_dups] homozygous read coverage threshold: X``. If it is not around homozygous coverage, the final assembly might be either too large or too small. To fix this issue, please set ``--hom-cov`` to homozygous coverage.
* number of het/hom bases: how many bases in unitig graph are heterozygous and homozygous during Hi-C phased assembly. Hifiasm prints it as: ``[M::stat] # heterozygous bases: X; # homozygous bases: Y``. Given a heterozygous sample, if there are much more homozygous bases than heterozygous bases, hifiasm fails to identify correct coverage threshold for homozygous reads. In this case, please set ``--hom-cov`` to homozygous coverage.
+47 -47
View File
@@ -1,47 +1,47 @@
.. _pa-assembly:
HiFi-only Assembly
==================
A typical hifiasm command line looks like::
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
where ``NA12878.fq.gz`` provides the input reads, ``-t`` sets the number of CPUs in
use and ``-o`` specifies the prefix of output files. Input sequences should be FASTA
or FASTQ format, uncompressed or compressed with gzip (.gz). The quality scores of reads
in FASTQ are ignored by hifiasm. Hifiasm outputs assemblies in `GFA <https://github.com/pmelsted/GFA-spec/blob/master/GFA-spec.md>`_ format.
At the first run, hifiasm saves corrected reads and overlaps to disk as ``NA12878.asm.*.bin``. It reuses the saved results to avoid the time-consuming all-vs-all overlap calculation next time. You may specify ``-i`` to ignore precomputed overlaps and redo overlapping from raw reads. You can also dump error corrected reads in FASTA and read overlaps in PAF with::
hifiasm -o NA12878.asm -t 32 --write-paf --write-ec /dev/null
Hifiasm purges haplotig duplications by default. For inbred or homozygous genomes, you may disable purging with option ``-l0``. Old HiFi reads may contain short adapter sequences at the ends of reads. You can specify ``-z20`` to trim both ends of reads by 20bp. For small genomes, use ``-f0`` to disable the initial bloom filter which takes 16GB memory at the beginning. For genomes much larger than human, applying ``-f38`` or even ``-f39`` is preferred to save memory on k-mer counting.
Produce two partially phased assemblies
---------------------------------------
Since v0.15, hifiasm produces two sets of partially phased contigs in default like::
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
In this example, the partially phased contigs are written to ``NA12878.asm.bp.hap*.p_ctg.gfa``.
This pair of files can be thought to represent the two haplotypes in a diploid genome, though with occasional switch errors. The frequency of switches is determined by the heterozygosity of the input sample. Hifiasm also writes the primary contigs to ``NA12878.asm.bp.p_ctg.gfa``.
For samples with high heterozygosity rate, a common issue is that one set of partially phased contigs is much larger than another set. To fix this issue, please set smaller value for ``-s`` (default: 0.55). Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads.
In this case, please set ``--hom-cov`` to homozygous coverage. See :ref:`p-large` for more details.
Produce primary/alternate assemblies
------------------------------------
To get primary/alternate assemblies, the option ``--primary`` should be set::
hifiasm -o NA12878.asm --primary -t 32 NA12878.fq.gz
The primary contigs and the alternate contigs are written to ``NA12878.asm.p_ctg.gfa`` and ``NA12878.asm.a_ctg.gfa``, respectively. For inbred or homozygous genomes, the primary/alternate assemblies can be also produced by ``-l0``. Similarly, turning ``-s`` or ``--hom-cov`` should
be helpful if the primary assembly is too large. See :ref:`p-large` for more details.
.. _pa-assembly:
HiFi-only Assembly
==================
A typical hifiasm command line looks like::
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
where ``NA12878.fq.gz`` provides the input reads, ``-t`` sets the number of CPUs in
use and ``-o`` specifies the prefix of output files. Input sequences should be FASTA
or FASTQ format, uncompressed or compressed with gzip (.gz). The quality scores of reads
in FASTQ are ignored by hifiasm. Hifiasm outputs assemblies in `GFA <https://github.com/pmelsted/GFA-spec/blob/master/GFA-spec.md>`_ format.
At the first run, hifiasm saves corrected reads and overlaps to disk as ``NA12878.asm.*.bin``. It reuses the saved results to avoid the time-consuming all-vs-all overlap calculation next time. You may specify ``-i`` to ignore precomputed overlaps and redo overlapping from raw reads. You can also dump error corrected reads in FASTA and read overlaps in PAF with::
hifiasm -o NA12878.asm -t 32 --write-paf --write-ec /dev/null
Hifiasm purges haplotig duplications by default. For inbred or homozygous genomes, you may disable purging with option ``-l0``. Old HiFi reads may contain short adapter sequences at the ends of reads. You can specify ``-z20`` to trim both ends of reads by 20bp. For small genomes, use ``-f0`` to disable the initial bloom filter which takes 16GB memory at the beginning. For genomes much larger than human, applying ``-f38`` or even ``-f39`` is preferred to save memory on k-mer counting.
Produce two partially phased assemblies
---------------------------------------
Since v0.15, hifiasm produces two sets of partially phased contigs in default like::
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
In this example, the partially phased contigs are written to ``NA12878.asm.bp.hap*.p_ctg.gfa``.
This pair of files can be thought to represent the two haplotypes in a diploid genome, though with occasional switch errors. The frequency of switches is determined by the heterozygosity of the input sample. Hifiasm also writes the primary contigs to ``NA12878.asm.bp.p_ctg.gfa``.
For samples with high heterozygosity rate, a common issue is that one set of partially phased contigs is much larger than another set. To fix this issue, please set smaller value for ``-s`` (default: 0.55). Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads.
In this case, please set ``--hom-cov`` to homozygous coverage. See :ref:`p-large` for more details.
Produce primary/alternate assemblies
------------------------------------
To get primary/alternate assemblies, the option ``--primary`` should be set::
hifiasm -o NA12878.asm --primary -t 32 NA12878.fq.gz
The primary contigs and the alternate contigs are written to ``NA12878.asm.p_ctg.gfa`` and ``NA12878.asm.a_ctg.gfa``, respectively. For inbred or homozygous genomes, the primary/alternate assemblies can be also produced by ``-l0``. Similarly, turning ``-s`` or ``--hom-cov`` should
be helpful if the primary assembly is too large. See :ref:`p-large` for more details.
+298 -298
View File
@@ -1,298 +1,298 @@
.. _parameter-reference:
Hifiasm Parameter Reference
============================
Synopsis
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Assembly only with HiFi reads:
::
hifiasm -o [prefix] -t [nThreads] [options] input1.fq [input2.fq [...]]
Trio binning assembly with yak dumps:
::
yak count -o paternal.yak -b37 [-t nThreads] [-k kmerLen] paternal.fq.gz
yak count -o maternal.yak -b37 [-t nThreads] [-k kmerLen] maternal.fq.gz
hifiasm [-o prefix] [-t nThreads] [options] -1 paternal.yak -2 maternal.yak child.hifi.fq.gz
Hi-C integrated assembly:
::
hifiasm -o [prefix] -t [nThreads] --h1 [hic_r1.fq.gz,...] --h2 [hic_r2.fq.gz,...] [options] HiFi.read.fq.gz
To get detailed description of options, run:
::
hifiasm -h
or:
::
man ./hifiasm.1
General options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _oopt:
**\-o <FILE=hifiasm.asm>**
Prefix of output files. See :ref:`outfile` and :ref:`outformat` for more details.
.. _topt:
**\-t <INT=1>**
Number of CPU threads used by hifiasm.
.. _hopt:
**\-h**
Show help information.
.. _versionopt:
**\-\-version**
Show version number.
Error correction options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _kopt:
**\-k <INT=51>**
K-mer length. This option must be less than 64.
.. _wopt:
**\-w <INT=51>**
Minimizer window size.
.. _fopt:
**\-f <INT=37>**
Number of bits for bloom filter; 0 to disable. This bloom filter is used to filter out singleton k-mers when counting all k-mers. It takes 2\ :sup:`(INT-3)` bytes of memory. A proper setting saves memory. ``-f37`` is recommended for human assembly. For small genomes, use ``-f0`` to disable the initial bloom filter which takes 16GB memory at the beginning. For genomes much larger than human, applying ``-f38`` or even ``-f39`` is preferred to save memory on k-mer counting.
.. _Dopt:
**\-D <FLOAT=5.0>**
Drop k-mers occurring ``>FLOAT*coverage`` times. Hifiasm discards these high-frequency k-mers during error correction to reduce running time. The ``coverage`` is determined automatically by hifiasm based on k-mer plot, representing homozygous read coverage. Raising this option may improve the resolution of repetitive regions but takes longer time.
.. _NEopt:
**\-N <INT=100>**
Consider up to ``max(-D*coverage,-N)`` overlaps for each oriented read. The ``coverage`` is determined automatically by hifiasm based on k-mer plot, representing homozygous read coverage. Raising this option may improve the resolution of repetitive regions but takes longer time.
.. _ropt:
**\-r <INT=3>**
Rounds of haplotype-aware error correction. This option affects all outputs of hifiasm. Odd rounds of correction are preferred in practice.
.. _zopt:
**\-z <INT=0>**
Length of adapters that should be removed. This option remove ``INT`` bases from both ends of each read. Some old HiFi reads may consist of short adapters (e.g. 20bp adapter at one end). For such data, trimming short adapters would significantly improve the assembly quality.
.. _max-kocc-opt:
**\-\-max-kocc <INT=2000>**
Employ k-mers occurring < ``INT`` times to rescue repetitive overlaps. This option may improve the resolution of repeats.
.. _hg-size-opt:
**\-\-hg-size <INT(k/m/g)>**
Estimated haploid genome size used for inferring read coverage. This option is used to get accurate homozygous read coverage during error correction. Common suffices are required, for example, 100m or 3g.
.. _min-hist-cnt-opt:
**\-\-min-hist-cnt <INT=5>**
When analyzing the k-mer spectrum, ignore counts below ``INT``. For very low coverage of HiFi data, set smaller value for this option. See `issue 45 <https://github.com/chhylp123/hifiasm/issues/49>`_ for example.
Assembly options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _aopt:
**\-a <INT=4>**
Rounds of assembly graph cleaning. This option is used with ``-x`` and ``-y``. Note that unlike -r, this option does not affect error corrected reads and all-to-all overlaps.
.. _mopt:
**\-m <INT=10000000>**
Maximal probing distance for bubble popping when generating primary/alternate contig graphs. Bubbles longer than ``INT`` bases will not be popped.
.. _popt:
**\-p <INT=0>**
Maximal probing distance for bubble popping when generating haplotype-resolved processed unitig graph without small bubbles. Bubbles longer than ``INT`` bases will not be popped. Small bubbles might be caused by somatic mutations or noise in data. Please note that hifiasm automatically pops small bubbles based on coverage, which can be tweaked by ``--hom-cov``.
.. _nopt:
**\-n <INT=3>**
A unitig is considered small if it is composed of less than ``INT`` reads. Hifiasm may try to remove small unitigs at various steps.
.. _xyopt:
**\-x <FLOAT1=0.8>, \-y <FLOAT2=0.2>**
Max and min overlap drop ratio. This option is used with ``-a``. Given a node N in the assembly graph, let max(N) be the length of the longest overlap of N. Hifiasm iteratively drops overlaps of N if their length/max(N) is below a threshold controlled by ``-x`` and ``-y``. Hifiasm applies ``-a`` rounds of short overlap removal with an increasing threshold between ``FLOAT1`` and ``FLOAT2``.
.. _iopt:
**\-i**
Ignore all bin files so that hifiasm will start again from scratch.
.. _uopt:
**\-u**
Disable post-join step for contigs which may improve N50. The post-join step of hifiasm improves contig N50 but may introduce misassemblies.
.. _hom-cov-opt:
**\-\-hom-cov <INT>**
Homozygous read coverage inferred automatically in default. This option affects different types of outputs, including Hi-C phased assembly and HiFi-only assembly. For more details, see :ref:`hic-iss`, :ref:`p-large` and :ref:`loginter`.
.. _pri-range-opt:
**\-\-pri-range <INT1[,INT2]>**
Min and max coverage cutoffs of primary contigs. Keep contigs with coverage in this range at p_ctg.gfa. Inferred automatically in default. If ``INT2`` is not specified, it is set to infinity. Set -1 to disable.
.. _lowQ-opt:
**\-\-lowQ <INT=70>**
Output contig regions with ``>=INT%`` inconsistency to the bed file with suffix lowQ.bed. Set 0 to disable.
.. _b-cov-opt:
**\-\-b-cov <INT=0>**
Break contigs at potential misassemblies with ``<INT``-fold coverage. Work with ``--m-rate``. Set 0 to disable.
.. _h-cov-opt:
**\-\-h-cov <INT=-1>**
Break contigs at potential misassemblies with ``>INT``-fold coverage. Work with ``--m-rate``. Set -1 to disable.
.. _m-rate-opt:
**\-\-m-rate <FLOAT=0.75>**
Break contigs with ``<=FLOAT*coverage`` exact overlaps. Only work when ``--b-cov`` and ``--h-cov`` are specified.
.. _primary-opt:
**\-\-primary**
Output a primary assembly and an alternate assembly. Enable this option or ``-l0`` outputs a primary assembly and an alternate assembly.
Trio-binning options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _1opt:
**\-1 <FILE>**
K-mer dump generated by `yak count <https://github.com/lh3/yak>`_ from the paternal/haplotype1 reads.
.. _2opt:
**\-2 <FILE>**
K-mer dump generated by `yak count <https://github.com/lh3/yak>`_ from the maternal/haplotype2 reads.
.. _3opt:
**\-3 <FILE>**
List of paternal/haplotype1 read names.
.. _4opt:
**\-4 <FILE>**
List of maternal/haplotype2 read names.
.. _cdopt:
**\-c <INT1=2>, -d <INT2=5>**
Lower bound and upper bound of the binned k-mer's frequency. When doing trio binning, a k-mer is said to be differentiating if it occurs >= ``INT2`` times in one sample but occurs < ``INT1`` times in the other sample.
.. _t-occ-opt:
**\-\-t-occ <INT=60>**
Forcedly remove unitig including ``>INT`` unexpected haplotype-specific reads without considering graph topology. For more details, see :ref:`p-hamming`.
Purge duplication options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _ldopt:
**\-l <INT=3>**
Level of purge duplication. 0 to disable, 1 to only purge contained haplotigs, 2 to purge all types of haplotigs, 3 to purge all types of haplotigs in the most aggressive way. In default, 3 for non-trio assembly, 0 for trio-binning assembly. For trio-binning assembly, only level 0 and level 1 are allowed.
.. _sdopt:
**\-s <FLOAT=0.55>**
Similarity threshold for duplicate haplotigs that should be purged. In default, 0.75 for ``-l1/-l2``, 0.55 for ``-l3``. This option affects both HiFi-only assembly and Hi-C phased assembly. For more details, see :ref:`hic-iss` and :ref:`p-large`.
.. _ovlpdopt:
**\-O <INT=1>**
Min number of overlapped reads for duplicate haplotigs that should be purged.
.. _purgeopt:
**\-\-purge-max <INT>**
Coverage upper bound of purge duplication, which is inferred automatically in default. If the coverage of a contig is higher than this bound, don't apply purge duplication. Larger value makes assembly more contiguous but may collapse repeats or segmental duplications.
.. _nhapopt:
**\-\-n\-hap <INT=2>**
Assumption of haplotype number. If it is set to >2, the quality of primary assembly for polyploid genomes might be improved.
Hi-C integration options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _h1opt:
**\-\-h1 <FILEs>**
File names of input Hi-C R1 ``[r1_1.fq,r1_2.fq,...]``.
.. _h2opt:
**\-\-h2 <FILEs>**
File names of input Hi-C R2 ``[r2_1.fq,r2_2.fq,...]``.
.. _n-weightopt:
**\-\-n-weight <INT=3>**
Rounds of reweighting Hi-C links. Raising this option may improve phasing results but takes longer time.
.. _n-perturbopt:
**\-\-n-perturb <INT=10000>**
Rounds of perturbation. Increasing this option may improve phasing results but takes longer time.
.. _f-perturbopt:
**\-\-f-perturb <FLOAT=0.1>**
Fraction to flip for perturbation. Increasing this option may improve phasing results but takes longer time.
.. _seedopt:
**\-\-seed <INT=11>**
RNG seed.
.. _l-msjoin:
**\-\-l-msjoin <INT=500000>**
Detect misjoined unitigs of ``>=INT`` in size; 0 to disable.
.. _parameter-reference:
Hifiasm Parameter Reference
============================
Synopsis
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
Assembly only with HiFi reads:
::
hifiasm -o [prefix] -t [nThreads] [options] input1.fq [input2.fq [...]]
Trio binning assembly with yak dumps:
::
yak count -o paternal.yak -b37 [-t nThreads] [-k kmerLen] paternal.fq.gz
yak count -o maternal.yak -b37 [-t nThreads] [-k kmerLen] maternal.fq.gz
hifiasm [-o prefix] [-t nThreads] [options] -1 paternal.yak -2 maternal.yak child.hifi.fq.gz
Hi-C integrated assembly:
::
hifiasm -o [prefix] -t [nThreads] --h1 [hic_r1.fq.gz,...] --h2 [hic_r2.fq.gz,...] [options] HiFi.read.fq.gz
To get detailed description of options, run:
::
hifiasm -h
or:
::
man ./hifiasm.1
General options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _oopt:
**\-o <FILE=hifiasm.asm>**
Prefix of output files. See :ref:`outfile` and :ref:`outformat` for more details.
.. _topt:
**\-t <INT=1>**
Number of CPU threads used by hifiasm.
.. _hopt:
**\-h**
Show help information.
.. _versionopt:
**\-\-version**
Show version number.
Error correction options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _kopt:
**\-k <INT=51>**
K-mer length. This option must be less than 64.
.. _wopt:
**\-w <INT=51>**
Minimizer window size.
.. _fopt:
**\-f <INT=37>**
Number of bits for bloom filter; 0 to disable. This bloom filter is used to filter out singleton k-mers when counting all k-mers. It takes 2\ :sup:`(INT-3)` bytes of memory. A proper setting saves memory. ``-f37`` is recommended for human assembly. For small genomes, use ``-f0`` to disable the initial bloom filter which takes 16GB memory at the beginning. For genomes much larger than human, applying ``-f38`` or even ``-f39`` is preferred to save memory on k-mer counting.
.. _Dopt:
**\-D <FLOAT=5.0>**
Drop k-mers occurring ``>FLOAT*coverage`` times. Hifiasm discards these high-frequency k-mers during error correction to reduce running time. The ``coverage`` is determined automatically by hifiasm based on k-mer plot, representing homozygous read coverage. Raising this option may improve the resolution of repetitive regions but takes longer time.
.. _NEopt:
**\-N <INT=100>**
Consider up to ``max(-D*coverage,-N)`` overlaps for each oriented read. The ``coverage`` is determined automatically by hifiasm based on k-mer plot, representing homozygous read coverage. Raising this option may improve the resolution of repetitive regions but takes longer time.
.. _ropt:
**\-r <INT=3>**
Rounds of haplotype-aware error correction. This option affects all outputs of hifiasm. Odd rounds of correction are preferred in practice.
.. _zopt:
**\-z <INT=0>**
Length of adapters that should be removed. This option remove ``INT`` bases from both ends of each read. Some old HiFi reads may consist of short adapters (e.g. 20bp adapter at one end). For such data, trimming short adapters would significantly improve the assembly quality.
.. _max-kocc-opt:
**\-\-max-kocc <INT=2000>**
Employ k-mers occurring < ``INT`` times to rescue repetitive overlaps. This option may improve the resolution of repeats.
.. _hg-size-opt:
**\-\-hg-size <INT(k/m/g)>**
Estimated haploid genome size used for inferring read coverage. This option is used to get accurate homozygous read coverage during error correction. Common suffices are required, for example, 100m or 3g.
.. _min-hist-cnt-opt:
**\-\-min-hist-cnt <INT=5>**
When analyzing the k-mer spectrum, ignore counts below ``INT``. For very low coverage of HiFi data, set smaller value for this option. See `issue 45 <https://github.com/chhylp123/hifiasm/issues/49>`_ for example.
Assembly options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _aopt:
**\-a <INT=4>**
Rounds of assembly graph cleaning. This option is used with ``-x`` and ``-y``. Note that unlike -r, this option does not affect error corrected reads and all-to-all overlaps.
.. _mopt:
**\-m <INT=10000000>**
Maximal probing distance for bubble popping when generating primary/alternate contig graphs. Bubbles longer than ``INT`` bases will not be popped.
.. _popt:
**\-p <INT=0>**
Maximal probing distance for bubble popping when generating haplotype-resolved processed unitig graph without small bubbles. Bubbles longer than ``INT`` bases will not be popped. Small bubbles might be caused by somatic mutations or noise in data. Please note that hifiasm automatically pops small bubbles based on coverage, which can be tweaked by ``--hom-cov``.
.. _nopt:
**\-n <INT=3>**
A unitig is considered small if it is composed of less than ``INT`` reads. Hifiasm may try to remove small unitigs at various steps.
.. _xyopt:
**\-x <FLOAT1=0.8>, \-y <FLOAT2=0.2>**
Max and min overlap drop ratio. This option is used with ``-a``. Given a node N in the assembly graph, let max(N) be the length of the longest overlap of N. Hifiasm iteratively drops overlaps of N if their length/max(N) is below a threshold controlled by ``-x`` and ``-y``. Hifiasm applies ``-a`` rounds of short overlap removal with an increasing threshold between ``FLOAT1`` and ``FLOAT2``.
.. _iopt:
**\-i**
Ignore all bin files so that hifiasm will start again from scratch.
.. _uopt:
**\-u**
Disable post-join step for contigs which may improve N50. The post-join step of hifiasm improves contig N50 but may introduce misassemblies.
.. _hom-cov-opt:
**\-\-hom-cov <INT>**
Homozygous read coverage inferred automatically in default. This option affects different types of outputs, including Hi-C phased assembly and HiFi-only assembly. For more details, see :ref:`hic-iss`, :ref:`p-large` and :ref:`loginter`.
.. _pri-range-opt:
**\-\-pri-range <INT1[,INT2]>**
Min and max coverage cutoffs of primary contigs. Keep contigs with coverage in this range at p_ctg.gfa. Inferred automatically in default. If ``INT2`` is not specified, it is set to infinity. Set -1 to disable.
.. _lowQ-opt:
**\-\-lowQ <INT=70>**
Output contig regions with ``>=INT%`` inconsistency to the bed file with suffix lowQ.bed. Set 0 to disable.
.. _b-cov-opt:
**\-\-b-cov <INT=0>**
Break contigs at potential misassemblies with ``<INT``-fold coverage. Work with ``--m-rate``. Set 0 to disable.
.. _h-cov-opt:
**\-\-h-cov <INT=-1>**
Break contigs at potential misassemblies with ``>INT``-fold coverage. Work with ``--m-rate``. Set -1 to disable.
.. _m-rate-opt:
**\-\-m-rate <FLOAT=0.75>**
Break contigs with ``<=FLOAT*coverage`` exact overlaps. Only work when ``--b-cov`` and ``--h-cov`` are specified.
.. _primary-opt:
**\-\-primary**
Output a primary assembly and an alternate assembly. Enable this option or ``-l0`` outputs a primary assembly and an alternate assembly.
Trio-binning options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _1opt:
**\-1 <FILE>**
K-mer dump generated by `yak count <https://github.com/lh3/yak>`_ from the paternal/haplotype1 reads.
.. _2opt:
**\-2 <FILE>**
K-mer dump generated by `yak count <https://github.com/lh3/yak>`_ from the maternal/haplotype2 reads.
.. _3opt:
**\-3 <FILE>**
List of paternal/haplotype1 read names.
.. _4opt:
**\-4 <FILE>**
List of maternal/haplotype2 read names.
.. _cdopt:
**\-c <INT1=2>, -d <INT2=5>**
Lower bound and upper bound of the binned k-mer's frequency. When doing trio binning, a k-mer is said to be differentiating if it occurs >= ``INT2`` times in one sample but occurs < ``INT1`` times in the other sample.
.. _t-occ-opt:
**\-\-t-occ <INT=60>**
Forcedly remove unitig including ``>INT`` unexpected haplotype-specific reads without considering graph topology. For more details, see :ref:`p-hamming`.
Purge duplication options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _ldopt:
**\-l <INT=3>**
Level of purge duplication. 0 to disable, 1 to only purge contained haplotigs, 2 to purge all types of haplotigs, 3 to purge all types of haplotigs in the most aggressive way. In default, 3 for non-trio assembly, 0 for trio-binning assembly. For trio-binning assembly, only level 0 and level 1 are allowed.
.. _sdopt:
**\-s <FLOAT=0.55>**
Similarity threshold for duplicate haplotigs that should be purged. In default, 0.75 for ``-l1/-l2``, 0.55 for ``-l3``. This option affects both HiFi-only assembly and Hi-C phased assembly. For more details, see :ref:`hic-iss` and :ref:`p-large`.
.. _ovlpdopt:
**\-O <INT=1>**
Min number of overlapped reads for duplicate haplotigs that should be purged.
.. _purgeopt:
**\-\-purge-max <INT>**
Coverage upper bound of purge duplication, which is inferred automatically in default. If the coverage of a contig is higher than this bound, don't apply purge duplication. Larger value makes assembly more contiguous but may collapse repeats or segmental duplications.
.. _nhapopt:
**\-\-n\-hap <INT=2>**
Assumption of haplotype number. If it is set to >2, the quality of primary assembly for polyploid genomes might be improved.
Hi-C integration options
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
.. _h1opt:
**\-\-h1 <FILEs>**
File names of input Hi-C R1 ``[r1_1.fq,r1_2.fq,...]``.
.. _h2opt:
**\-\-h2 <FILEs>**
File names of input Hi-C R2 ``[r2_1.fq,r2_2.fq,...]``.
.. _n-weightopt:
**\-\-n-weight <INT=3>**
Rounds of reweighting Hi-C links. Raising this option may improve phasing results but takes longer time.
.. _n-perturbopt:
**\-\-n-perturb <INT=10000>**
Rounds of perturbation. Increasing this option may improve phasing results but takes longer time.
.. _f-perturbopt:
**\-\-f-perturb <FLOAT=0.1>**
Fraction to flip for perturbation. Increasing this option may improve phasing results but takes longer time.
.. _seedopt:
**\-\-seed <INT=11>**
RNG seed.
.. _l-msjoin:
**\-\-l-msjoin <INT=500000>**
Detect misjoined unitigs of ``>=INT`` in size; 0 to disable.
+29 -29
View File
@@ -1,29 +1,29 @@
.. _trio-assembly:
Trio-binning Assembly
=====================
When parental short reads are available, hifiasm can also generate a pair of haplotype-resolved assemblies with trio binning. To perform such assembly, you need to count k-mers first with `yak <https://github.com/lh3/yak>`_ and then do assembly::
yak count -k31 -b37 -t16 -o pat.yak paternal.fq.gz
yak count -k31 -b37 -t16 -o mat.yak maternal.fq.gz
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak NA12878.fq.gz
Here ``NA12878.asm.hap1.p_ctg.gfa`` and ``NA12878.asm.hap2.p_ctg.gfa`` give the assemblies for two haplotypes. In the binning mode, hifiasm does not purge haplotig duplicates by default. Because hifiasm reuses saved overlaps, you can generate both primary/alternate assemblies and trio binning assemblies with::
hifiasm -o NA12878.asm --primary -t 32 NA12878.fq.gz 2> NA12878.asm.pri.log
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak /dev/null 2> NA12878.asm.trio.log
The second command line will run much faster than the first. The phasing switch error rate and hamming error rate are able to be evaluated quickly by `yak <https://github.com/lh3/yak>`_::
yak trioeval -t16 pat.yak mat.yak assembly.fa
The W-line and H-line reported by ``yak trioeval`` indicate switch error rate and hamming error rate respectively::
W 26714 3029448 0.008818
H 24315 3029885 0.008025
For this example, the switch error rate is 0.8818% and the hamming error rate is 0.8025%. If the hamming error rate or the swith error rate of trio-binning assembly is very high, it might be caused by hifiasm or the incorrect parental data. To fix it, see :ref:`p-hamming` for more details.
.. _trio-assembly:
Trio-binning Assembly
=====================
When parental short reads are available, hifiasm can also generate a pair of haplotype-resolved assemblies with trio binning. To perform such assembly, you need to count k-mers first with `yak <https://github.com/lh3/yak>`_ and then do assembly::
yak count -k31 -b37 -t16 -o pat.yak paternal.fq.gz
yak count -k31 -b37 -t16 -o mat.yak maternal.fq.gz
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak NA12878.fq.gz
Here ``NA12878.asm.hap1.p_ctg.gfa`` and ``NA12878.asm.hap2.p_ctg.gfa`` give the assemblies for two haplotypes. In the binning mode, hifiasm does not purge haplotig duplicates by default. Because hifiasm reuses saved overlaps, you can generate both primary/alternate assemblies and trio binning assemblies with::
hifiasm -o NA12878.asm --primary -t 32 NA12878.fq.gz 2> NA12878.asm.pri.log
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak /dev/null 2> NA12878.asm.trio.log
The second command line will run much faster than the first. The phasing switch error rate and hamming error rate are able to be evaluated quickly by `yak <https://github.com/lh3/yak>`_::
yak trioeval -t16 pat.yak mat.yak assembly.fa
The W-line and H-line reported by ``yak trioeval`` indicate switch error rate and hamming error rate respectively::
W 26714 3029448 0.008818
H 24315 3029885 0.008025
For this example, the switch error rate is 0.8818% and the hamming error rate is 0.8025%. If the hamming error rate or the swith error rate of trio-binning assembly is very high, it might be caused by hifiasm or the incorrect parental data. To fix it, see :ref:`p-hamming` for more details.