mirror of
https://github.com/lh3/minimap2.git
synced 2026-09-25 09:28:12 +08:00
Compare commits
24
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d6e6811a0f | ||
|
|
b385748c40 | ||
|
|
7339801629 | ||
|
|
7fd30e15b8 | ||
|
|
4df2d259ee | ||
|
|
9cabb4a2b9 | ||
|
|
a5c14dd5f9 | ||
|
|
38075e82cc | ||
|
|
1ee40b0c32 | ||
|
|
b403cf3e6f | ||
|
|
84b1c201c8 | ||
|
|
f557d7fbd9 | ||
|
|
4bc645c31d | ||
|
|
609b430866 | ||
|
|
1c21888e94 | ||
|
|
34e273c8ee | ||
|
|
f68b4b22df | ||
|
|
e2e494de67 | ||
|
|
558be6b729 | ||
|
|
6da640e551 | ||
|
|
448341c96c | ||
|
|
a9ac74ffe1 | ||
|
|
b2ff8fbe92 | ||
|
|
0369874d4e |
@@ -1,21 +0,0 @@
|
|||||||
name: CI
|
|
||||||
|
|
||||||
on:
|
|
||||||
push:
|
|
||||||
branches:
|
|
||||||
- master
|
|
||||||
pull_request:
|
|
||||||
|
|
||||||
jobs:
|
|
||||||
build:
|
|
||||||
runs-on: ubuntu-latest
|
|
||||||
strategy:
|
|
||||||
matrix:
|
|
||||||
compiler: [gcc, clang]
|
|
||||||
|
|
||||||
steps:
|
|
||||||
- name: Checkout minimap2
|
|
||||||
uses: actions/checkout@v2
|
|
||||||
|
|
||||||
- name: Compile with ${{ matrix.compiler }}
|
|
||||||
run: make CC=${{ matrix.compiler }}
|
|
||||||
@@ -1,3 +1,6 @@
|
|||||||
[submodule "lib/simde"]
|
[submodule "lib/simde"]
|
||||||
path = lib/simde
|
path = lib/simde
|
||||||
url = https://github.com/nemequ/simde.git
|
url = https://github.com/nemequ/simde.git
|
||||||
|
[submodule "ext/TAL"]
|
||||||
|
path = ext/TAL
|
||||||
|
url = https://github.com/IntelLabs/Trans-Omics-Acceleration-Library.git
|
||||||
|
|||||||
+24
@@ -0,0 +1,24 @@
|
|||||||
|
matrix:
|
||||||
|
include:
|
||||||
|
- language: c
|
||||||
|
compiler: gcc
|
||||||
|
script: make
|
||||||
|
- language: c
|
||||||
|
compiler: clang
|
||||||
|
script: make
|
||||||
|
- arch: arm64
|
||||||
|
language: c
|
||||||
|
compiler: gcc
|
||||||
|
script: make arm_neon=1 aarch64=1
|
||||||
|
- language: python
|
||||||
|
python: "2.7"
|
||||||
|
before_install: pip install cython
|
||||||
|
script: python setup.py build_ext
|
||||||
|
- language: python
|
||||||
|
python: "3.5"
|
||||||
|
before_install: pip install cython
|
||||||
|
script: python setup.py build_ext
|
||||||
|
- language: python
|
||||||
|
python: "3.9"
|
||||||
|
before_install: pip install cython
|
||||||
|
script: python setup.py build_ext
|
||||||
@@ -4,6 +4,7 @@ include ksw2_dispatch.c
|
|||||||
include main.c
|
include main.c
|
||||||
include README.md
|
include README.md
|
||||||
include sse2neon/emmintrin.h
|
include sse2neon/emmintrin.h
|
||||||
|
include python/mappy.c
|
||||||
include python/cmappy.h
|
include python/cmappy.h
|
||||||
include python/cmappy.pxd
|
include python/cmappy.pxd
|
||||||
include python/mappy.pyx
|
include python/mappy.pyx
|
||||||
|
|||||||
@@ -1,20 +1,74 @@
|
|||||||
CFLAGS= -g -Wall -O2 -Wc++-compat #-Wextra
|
|
||||||
CPPFLAGS= -DHAVE_KALLOC
|
## /* The MIT License
|
||||||
INCLUDES=
|
##
|
||||||
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o \
|
## Copyright (c) 2018- Dana-Farber Cancer Institute
|
||||||
lchain.o align.o hit.o seed.o map.o format.o pe.o esterr.o splitidx.o \
|
## 2017-2018 Broad Institute, Inc.
|
||||||
ksw2_ll_sse.o
|
##
|
||||||
|
## Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
## a copy of this software and associated documentation files (the
|
||||||
|
## "Software"), to deal in the Software without restriction, including
|
||||||
|
## without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
## distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
## permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
## the following conditions:
|
||||||
|
##
|
||||||
|
## The above copyright notice and this permission notice shall be
|
||||||
|
## included in all copies or substantial portions of the Software.
|
||||||
|
##
|
||||||
|
## THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
## EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
## MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
|
## NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
|
## BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
|
## ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
|
## CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
## SOFTWARE.
|
||||||
|
## Modified Copyright (C) 2021 Intel Corporation
|
||||||
|
## Contacts: Saurabh Kalikar <saurabh.kalikar@intel.com>;
|
||||||
|
## Vasimuddin Md <vasimuddin.md@intel.com>; Sanchit Misra <sanchit.misra@intel.com>;
|
||||||
|
## Chirag Jain <chirag@iisc.ac.in>; Heng Li <hli@jimmy.harvard.edu>
|
||||||
|
## */
|
||||||
|
##
|
||||||
|
CFLAGS= -Wall -O2 -Wc++-compat #-Wextra
|
||||||
|
CPPFLAGS= -DHAVE_KALLOC -march=native
|
||||||
|
|
||||||
|
OPT_FLAGS= -DVECTORIZED_CHAINING -DALIGN_AVX
|
||||||
|
|
||||||
|
ifeq ($(lhash), 1)
|
||||||
|
OPT_FLAGS+= -DLISA_HASH -DUINT64 -DVECTORIZE
|
||||||
|
endif
|
||||||
|
|
||||||
|
ifeq ($(manual_profile), 1)
|
||||||
|
CPPFLAGS+= -DMANUAL_PROFILING
|
||||||
|
endif
|
||||||
|
|
||||||
|
ifeq ($(use_avx2), 1)
|
||||||
|
OPT_FLAGS+= -DAPPLY_AVX2
|
||||||
|
endif
|
||||||
|
|
||||||
|
ifeq ($(disable_output), 1)
|
||||||
|
CPPFLAGS+= -DDISABLE_OUTPUT
|
||||||
|
endif
|
||||||
|
|
||||||
|
ifeq ($(no_opt),)
|
||||||
|
CPPFLAGS+= $(OPT_FLAGS)
|
||||||
|
endif
|
||||||
|
|
||||||
|
INCLUDES= -I./ext/TAL/src/LISA-hash -I./ext/TAL/src/dynamic-programming
|
||||||
|
|
||||||
|
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o chain.o align.o hit.o map.o format.o pe.o esterr.o splitidx.o ksw2_ll_sse.o
|
||||||
PROG= minimap2
|
PROG= minimap2
|
||||||
PROG_EXTRA= sdust minimap2-lite
|
PROG_EXTRA= sdust minimap2-lite
|
||||||
LIBS= -lm -lz -lpthread
|
LIBS= -lm -lz -lpthread
|
||||||
|
|
||||||
ifneq ($(aarch64),)
|
CC=$(CXX)
|
||||||
arm_neon=1
|
ifeq ($(CC), g++)
|
||||||
|
CC=g++ -std=c++11
|
||||||
endif
|
endif
|
||||||
|
|
||||||
ifeq ($(arm_neon),) # if arm_neon is not defined
|
ifeq ($(arm_neon),) # if arm_neon is not defined
|
||||||
ifeq ($(sse2only),) # if sse2only is not defined
|
ifeq ($(sse2only),) # if sse2only is not defined
|
||||||
OBJS+=ksw2_extz2_sse41.o ksw2_extd2_sse41.o ksw2_exts2_sse41.o ksw2_extz2_sse2.o ksw2_extd2_sse2.o ksw2_exts2_sse2.o ksw2_dispatch.o
|
OBJS+=ksw2_extz2_sse41.o ksw2_extd2_sse41.o ksw2_exts2_sse41.o ksw2_extz2_sse2.o ksw2_extd2_sse2.o ksw2_exts2_sse2.o ksw2_dispatch.o ksw2_extd2_avx.o
|
||||||
else # if sse2only is defined
|
else # if sse2only is defined
|
||||||
OBJS+=ksw2_extz2_sse.o ksw2_extd2_sse.o ksw2_exts2_sse.o
|
OBJS+=ksw2_extz2_sse.o ksw2_extd2_sse.o ksw2_exts2_sse.o
|
||||||
endif
|
endif
|
||||||
@@ -30,12 +84,12 @@ endif
|
|||||||
|
|
||||||
ifneq ($(asan),)
|
ifneq ($(asan),)
|
||||||
CFLAGS+=-fsanitize=address
|
CFLAGS+=-fsanitize=address
|
||||||
LIBS+=-fsanitize=address -ldl
|
LIBS+=-fsanitize=address
|
||||||
endif
|
endif
|
||||||
|
|
||||||
ifneq ($(tsan),)
|
ifneq ($(tsan),)
|
||||||
CFLAGS+=-fsanitize=thread
|
CFLAGS+=-fsanitize=thread
|
||||||
LIBS+=-fsanitize=thread -ldl
|
LIBS+=-fsanitize=thread
|
||||||
endif
|
endif
|
||||||
|
|
||||||
.PHONY:all extra clean depend
|
.PHONY:all extra clean depend
|
||||||
@@ -60,6 +114,17 @@ libminimap2.a:$(OBJS)
|
|||||||
sdust:sdust.c kalloc.o kalloc.h kdq.h kvec.h kseq.h ketopt.h sdust.h
|
sdust:sdust.c kalloc.o kalloc.h kdq.h kvec.h kseq.h ketopt.h sdust.h
|
||||||
$(CC) -D_SDUST_MAIN $(CFLAGS) $< kalloc.o -o $@ -lz
|
$(CC) -D_SDUST_MAIN $(CFLAGS) $< kalloc.o -o $@ -lz
|
||||||
|
|
||||||
|
multi:
|
||||||
|
$(MAKE) clean
|
||||||
|
$(MAKE)
|
||||||
|
mv minimap2 mm2-fast
|
||||||
|
$(MAKE) clean
|
||||||
|
$(MAKE) lhash=1
|
||||||
|
mv minimap2 mm2-fast-lhash
|
||||||
|
$(MAKE) clean
|
||||||
|
$(MAKE) no_opt=1
|
||||||
|
mv minimap2 mm2-fast-no-opt
|
||||||
|
|
||||||
# SSE-specific targets on x86/x86_64
|
# SSE-specific targets on x86/x86_64
|
||||||
|
|
||||||
ifeq ($(arm_neon),) # if arm_neon is defined, compile this target with the default setting (i.e. no -msse2)
|
ifeq ($(arm_neon),) # if arm_neon is defined, compile this target with the default setting (i.e. no -msse2)
|
||||||
@@ -109,28 +174,26 @@ depend:
|
|||||||
|
|
||||||
# DO NOT DELETE
|
# DO NOT DELETE
|
||||||
|
|
||||||
align.o: minimap.h mmpriv.h bseq.h kseq.h ksw2.h kalloc.h
|
align.o: minimap.h mmpriv.h bseq.h ksw2.h kalloc.h
|
||||||
bseq.o: bseq.h kvec.h kalloc.h kseq.h
|
bseq.o: bseq.h kvec.h kalloc.h kseq.h
|
||||||
esterr.o: mmpriv.h minimap.h bseq.h kseq.h
|
chain.o: minimap.h mmpriv.h bseq.h kalloc.h
|
||||||
|
esterr.o: mmpriv.h minimap.h bseq.h
|
||||||
example.o: minimap.h kseq.h
|
example.o: minimap.h kseq.h
|
||||||
format.o: kalloc.h mmpriv.h minimap.h bseq.h kseq.h
|
format.o: kalloc.h mmpriv.h minimap.h bseq.h
|
||||||
hit.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h khash.h
|
hit.o: mmpriv.h minimap.h bseq.h kalloc.h khash.h
|
||||||
index.o: kthread.h bseq.h minimap.h mmpriv.h kseq.h kvec.h kalloc.h khash.h
|
index.o: kthread.h bseq.h minimap.h mmpriv.h kvec.h kalloc.h khash.h
|
||||||
index.o: ksort.h
|
|
||||||
kalloc.o: kalloc.h
|
kalloc.o: kalloc.h
|
||||||
ksw2_extd2_sse.o: ksw2.h kalloc.h
|
ksw2_extd2_sse.o: ksw2.h kalloc.h
|
||||||
ksw2_exts2_sse.o: ksw2.h kalloc.h
|
ksw2_exts2_sse.o: ksw2.h kalloc.h
|
||||||
ksw2_extz2_sse.o: ksw2.h kalloc.h
|
ksw2_extz2_sse.o: ksw2.h kalloc.h
|
||||||
ksw2_ll_sse.o: ksw2.h kalloc.h
|
ksw2_ll_sse.o: ksw2.h kalloc.h
|
||||||
kthread.o: kthread.h
|
kthread.o: kthread.h
|
||||||
lchain.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h krmq.h
|
main.o: bseq.h minimap.h mmpriv.h ketopt.h
|
||||||
main.o: bseq.h minimap.h mmpriv.h kseq.h ketopt.h
|
map.o: kthread.h kvec.h kalloc.h sdust.h mmpriv.h minimap.h bseq.h khash.h
|
||||||
map.o: kthread.h kvec.h kalloc.h sdust.h mmpriv.h minimap.h bseq.h kseq.h
|
map.o: ksort.h
|
||||||
map.o: khash.h ksort.h
|
misc.o: mmpriv.h minimap.h bseq.h ksort.h
|
||||||
misc.o: mmpriv.h minimap.h bseq.h kseq.h ksort.h
|
options.o: mmpriv.h minimap.h bseq.h
|
||||||
options.o: mmpriv.h minimap.h bseq.h kseq.h
|
pe.o: mmpriv.h minimap.h bseq.h kvec.h kalloc.h ksort.h
|
||||||
pe.o: mmpriv.h minimap.h bseq.h kseq.h kvec.h kalloc.h ksort.h
|
sdust.o: kalloc.h kdq.h kvec.h ketopt.h sdust.h
|
||||||
sdust.o: kalloc.h kdq.h kvec.h sdust.h
|
sketch.o: kvec.h kalloc.h mmpriv.h minimap.h bseq.h
|
||||||
seed.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h ksort.h
|
splitidx.o: mmpriv.h minimap.h bseq.h
|
||||||
sketch.o: kvec.h kalloc.h mmpriv.h minimap.h bseq.h kseq.h
|
|
||||||
splitidx.o: mmpriv.h minimap.h bseq.h kseq.h
|
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
CFLAGS= -g -Wall -O2 -Wc++-compat #-Wextra
|
CFLAGS= -g -Wall -O2 -Wc++-compat #-Wextra
|
||||||
CPPFLAGS= -DHAVE_KALLOC -DUSE_SIMDE -DSIMDE_ENABLE_NATIVE_ALIASES
|
CPPFLAGS= -DHAVE_KALLOC -DUSE_SIMDE -DSIMDE_ENABLE_NATIVE_ALIASES
|
||||||
INCLUDES= -Ilib/simde
|
INCLUDES= -Ilib/simde
|
||||||
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o lchain.o align.o hit.o map.o format.o pe.o seed.o esterr.o splitidx.o \
|
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o chain.o align.o hit.o map.o format.o pe.o esterr.o splitidx.o \
|
||||||
ksw2_extz2_simde.o ksw2_extd2_simde.o ksw2_exts2_simde.o ksw2_ll_simde.o
|
ksw2_extz2_simde.o ksw2_extd2_simde.o ksw2_exts2_simde.o ksw2_ll_simde.o
|
||||||
PROG= minimap2
|
PROG= minimap2
|
||||||
PROG_EXTRA= sdust minimap2-lite
|
PROG_EXTRA= sdust minimap2-lite
|
||||||
|
|||||||
@@ -1,160 +1,3 @@
|
|||||||
Release 2.25-r1173 (25 April 2023)
|
|
||||||
----------------------------------
|
|
||||||
|
|
||||||
Notable changes:
|
|
||||||
|
|
||||||
* Improvement: use the miniprot splice model for RNA-seq alignment by default.
|
|
||||||
This model considers non-GT-AG splice sites and leads to slightly higher
|
|
||||||
(<0.1%) accuracy and sensitivity on real human data.
|
|
||||||
|
|
||||||
* Change: increased the default `-I` to `8G` such that minimap2 would create a
|
|
||||||
uni-part index for a pair of mammalian genomes. This change may increase the
|
|
||||||
memory for all-vs-all read overlap alignment given large datasets.
|
|
||||||
|
|
||||||
* New feature: output the sequences in secondary alignments with option
|
|
||||||
`--secondary-seq` (#687).
|
|
||||||
|
|
||||||
* Bugfix: --rmq was not parsed correctly (#1010)
|
|
||||||
|
|
||||||
* Bugfix: possibly incorrect coordinate when applying end bonus to the target
|
|
||||||
sequence (#1025). This is a ksw2 bug. It does not affect minimap2 as
|
|
||||||
minimap2 is not using the affected feature.
|
|
||||||
|
|
||||||
* Improvement: incorporated several changes for better compatibility with
|
|
||||||
Windows (#1051) and for minimap2 integration at Oxford Nanopore Technologies
|
|
||||||
(#1048 and #1033).
|
|
||||||
|
|
||||||
* Improvement: output the HD-line in SAM output (#1019).
|
|
||||||
|
|
||||||
* Improvement: check minimap2 index file in mappy to prevent segmentation
|
|
||||||
fault for certain indices (#1008).
|
|
||||||
|
|
||||||
For genomic sequences, minimap2 should give identical output to v2.24.
|
|
||||||
Long-read RNA-seq alignment may occasionally differ from previous versions.
|
|
||||||
|
|
||||||
(2.25: 25 April 2023, r1173)
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Release 2.24-r1122 (26 December 2021)
|
|
||||||
-------------------------------------
|
|
||||||
|
|
||||||
This release improves alignment around long poorly aligned regions. Older
|
|
||||||
minimap2 may chain through such regions in rare cases which may result in
|
|
||||||
missing alignments later. The issue has become worse since the the change of
|
|
||||||
the chaining algorithm in v2.19. v2.23 implements an incomplete remedy. This
|
|
||||||
release provides a better solution with a X-drop-like heuristic and by enabling
|
|
||||||
two-bandwidth chaining in the assembly mode.
|
|
||||||
|
|
||||||
(2.24: 26 December 2021, r1122)
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Release 2.23-r1111 (18 November 2021)
|
|
||||||
-------------------------------------
|
|
||||||
|
|
||||||
Notable changes:
|
|
||||||
|
|
||||||
* Bugfix: fixed missing alignments around long inversions (#806 and #816).
|
|
||||||
This bug affected v2.19 through v2.22.
|
|
||||||
|
|
||||||
* Improvement: avoid extremely long mapping time for pathologic reads with
|
|
||||||
highly repeated k-mers not in the reference (#771). Use --q-occ-frac=0
|
|
||||||
to disable the new heuristic.
|
|
||||||
|
|
||||||
* Change: use --cap-kalloc=1g by default.
|
|
||||||
|
|
||||||
(2.23: 18 November 2021, r1111)
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Release 2.22-r1101 (7 August 2021)
|
|
||||||
----------------------------------
|
|
||||||
|
|
||||||
When choosing the best alignment, this release uses logarithm gap penalty and
|
|
||||||
query-specific mismatch penalty. It improves the sensitivity to long INDELs in
|
|
||||||
repetitive regions.
|
|
||||||
|
|
||||||
Other notable changes:
|
|
||||||
|
|
||||||
* Bugfix: fixed an indirect memory leak that may waste a large amount of
|
|
||||||
memory given highly repetitive reference such as a 16S RNA database (#749).
|
|
||||||
All versions of minimap2 have this issue.
|
|
||||||
|
|
||||||
* New feature: added --cap-kalloc to reduce the peak memory. This option is
|
|
||||||
not enabled by default but may become the default in future releases.
|
|
||||||
|
|
||||||
Known issue:
|
|
||||||
|
|
||||||
* Minimap2 may take a long time to map a read (#771). So far it is not clear
|
|
||||||
if this happens to v2.18 and earlier versions.
|
|
||||||
|
|
||||||
(2.22: 7 August 2021, r1101)
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Release 2.21-r1071 (6 July 2021)
|
|
||||||
--------------------------------
|
|
||||||
|
|
||||||
This release fixed a regression in short-read mapping introduced in v2.19
|
|
||||||
(#776). It also fixed invalid comparisons of uninitialized variables, though
|
|
||||||
these are harmless (#752). Long-read alignment should be identical to v2.20.
|
|
||||||
|
|
||||||
(2.21: 6 July 2021, r1071)
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Release 2.20-r1061 (27 May 2021)
|
|
||||||
--------------------------------
|
|
||||||
|
|
||||||
This release fixed a bug in the Python module and improves the command-line
|
|
||||||
compatibiliity with v2.18. In v2.19, if `-r` is specified with an `asm*` preset,
|
|
||||||
users would get alignments more fragmented than v2.18. This could be an issue
|
|
||||||
for existing pipelines specifying `-r`. This release resolves this issue.
|
|
||||||
|
|
||||||
(2.20: 27 May 2021, r1061)
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Release 2.19-r1057 (26 May 2021)
|
|
||||||
--------------------------------
|
|
||||||
|
|
||||||
This release includes a few important improvements backported from unimap:
|
|
||||||
|
|
||||||
* Improvement: more contiguous alignment through long INDELs. This is enabled
|
|
||||||
by the minigraph chaining algorithm. All `asm*` presets now use the new
|
|
||||||
algorithm. They can find INDELs up to 100kb and may be faster for
|
|
||||||
chromosome-long contigs. The default mode and `map*` presets use this
|
|
||||||
algorithm to replace the long-join heuristic.
|
|
||||||
|
|
||||||
* Improvement: better alignment in highly repetitive regions by rescuing
|
|
||||||
high-occurrence seeds. If the distance between two adjacent seeds is too
|
|
||||||
large, attempt to choose a fraction of high-occurrence seeds in-between.
|
|
||||||
Minimap2 now produces fewer clippings and alignment break points in long
|
|
||||||
satellite regions.
|
|
||||||
|
|
||||||
* Improvement: allow to specify an interval of k-mer occurrences with `-U`.
|
|
||||||
For repeat-rich genomes, the automatic k-mer occurrence threshold determined
|
|
||||||
by `-f` may be too large and makes alignment impractically slow. The new
|
|
||||||
option protects against such cases. Enabled for `asm*` and `map-hifi`.
|
|
||||||
|
|
||||||
* New feature: added the `map-hifi` preset for maping PacBio High-Fidelity
|
|
||||||
(HiFi) reads.
|
|
||||||
|
|
||||||
* Change to the default: apply `--cap-sw-mem=100m` for genomic alignment.
|
|
||||||
|
|
||||||
* Bugfix: minimap2 could not generate an index file with `-xsr` (#734).
|
|
||||||
|
|
||||||
This release represents the most signficant algorithmic change since v2.1 in
|
|
||||||
2017. With features backported from unimap, minimap2 now has similar power to
|
|
||||||
unimap for contig alignment. Unimap will remain an experimental project and is
|
|
||||||
no longer recommended over minimap2. Sorry for reverting the recommendation in
|
|
||||||
short time.
|
|
||||||
|
|
||||||
(2.19: 26 May 2021, r1057)
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Release 2.18-r1015 (9 April 2021)
|
Release 2.18-r1015 (9 April 2021)
|
||||||
---------------------------------
|
---------------------------------
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,77 @@
|
|||||||
|
## mm2-fast
|
||||||
|
### Introduction
|
||||||
|
mm2-fast is an accelerated implementation of minimap2 on modern CPUs. mm2-fast accelerates all the three major modules of minimap2: (a) seeding, (b) chaining, and (c) pairwise alignment, achieving up to 3.5x speedup over minimap2.
|
||||||
|
mm2-fast is a drop-in replacement of minimap2, providing the same functionality with the exact same output.
|
||||||
|
In the current version, all the modules are optimized using **AVX-512** vectorization. Detailed benchmark results are available in our [preprint](https://doi.org/10.1101/2021.07.21.453294).
|
||||||
|
|
||||||
|
### System requirement
|
||||||
|
Operating System: Linux
|
||||||
|
mm2-fast was tested using g++ (GCC) 9.2.0 and icpc version 19.1.3.304
|
||||||
|
Architecture: x86\_64 CPUs with [AVX512](https://en.wikipedia.org/wiki/AVX-512)
|
||||||
|
Memory requirement: ~30GB for human genome
|
||||||
|
|
||||||
|
### Installation
|
||||||
|
Clone the *fast-contrib* branch from minimap2 github page. The source code can be compiled by simple using *make* command. It only takes a few seconds.
|
||||||
|
```
|
||||||
|
git clone --recursive https://github.com/lh3/minimap2.git -b fast-contrib mm2-fast
|
||||||
|
cd mm2-fast
|
||||||
|
make
|
||||||
|
```
|
||||||
|
|
||||||
|
### Usage
|
||||||
|
The usage of mm2-fast is same as minimap2. Here is an example of mapping ONT reads with test data.
|
||||||
|
```sh
|
||||||
|
./minimap2 -ax map-ont test/MT-human.fa test/MT-orang.fa > mm2-fast_output
|
||||||
|
```
|
||||||
|
|
||||||
|
### Accuracy evaluation
|
||||||
|
As mm2-fast is an accelerated version of minimap2-v2.18, the output of mm2-fast can be verified against minimap2-v2.18. Note that AVX512-based chaining in mm2-fast by default runs with a chaining parameter *max-skip=infinity* for higher chaining precision. Therefore, for correctness verification, minimap2 should run with a larger value of *max-skip* parameter. Follow the below steps to verify the accuracy of mm2-fast.
|
||||||
|
```sh
|
||||||
|
git clone https://github.com/lh3/minimap2.git -b v2.18
|
||||||
|
cd minimap2 && make
|
||||||
|
./minimap2 -ax map-ont test/MT-human.fa test/MT-orang.fa --max-chain-skip=1000000 > minimap2_output
|
||||||
|
```
|
||||||
|
The output generated by minimap2 and mm2-fast should match.
|
||||||
|
```sh
|
||||||
|
diff minimap2_output mm2-fast_output > diff_result
|
||||||
|
```
|
||||||
|
The file diff\_result should show a clean-diff with the difference of 2 lines, i.e., the lines containing the command-line parameters for minimap2 and mm2-fast.
|
||||||
|
|
||||||
|
### Advanced options
|
||||||
|
The default compilation using make applies two optimizations: AVX512 vectorized chaining and alignment, and learned-indexes based seeding is disabled by default as it requires availability of [Rust](https://en.wikipedia.org/wiki/Rust_(programming_language)). This is because the learned hash-table uses an external training library that runs on Rust. Rust is trivial to install, see https://rustup.rs/ and add its path to .bashrc file. Rust installation only takes a few seconds. Following are the steps to enable learned hash table optimization in mm2-fast:
|
||||||
|
```sh
|
||||||
|
# Start by building learned hash table index for optimized seeding module
|
||||||
|
./build_rmi.sh test/MT-human.fa map-ont ##Takes two arguments: 1. path-to-reference-seq-file 2. preset.
|
||||||
|
##For human genome, this step should take around 20-30 minutes to finish.
|
||||||
|
|
||||||
|
# Next, compile and run the mapping phase
|
||||||
|
make clean && make lhash=1
|
||||||
|
./minimap2 -ax map-ont test/MT-human.fa test/MT-orang.fa > mm2-fast-lhash_output
|
||||||
|
```
|
||||||
|
To compile mm2-fast with all optimizations turned off and switch back to default minimap2, use the following command during compilation. This could be useful for debugging.
|
||||||
|
```sh
|
||||||
|
make clean && make no_opt=1
|
||||||
|
```
|
||||||
|
mm2-fast includes preliminary support for AVX2 architecture. Currently, chaining step is not optimized for AVX2 but the seeding and alignment steps are available. To try mm2-fast on AVX2 systems, use the following command to compile.
|
||||||
|
```sh
|
||||||
|
make clean && make lhash=1 use_avx2=1
|
||||||
|
```
|
||||||
|
### Performance
|
||||||
|
We have observed up to 3.5x speedup across datasets (please refer to the paper for more details). For example, for the randomly sampled 100K reads from ["HG002\_GM24385\_1\_2\_3\_Guppy\_3.6.0\_prom.fastq.gz"](https://precision.fda.gov/challenges/10/view), minimap2 takes 80 seconds, while mm2-fast takes 38 seconds to map against the human genome on a 28 cores Intel® Xeon® Platinum 8280 CPUs. Our sampled datasets with 100K reads are available [here](https://drive.google.com/drive/folders/1131j7ejHdT7QZnjxLcTLi5qqwYcfFbuv).
|
||||||
|
|
||||||
|
### Future Plans
|
||||||
|
The current version of mm2-fast is based on minimap2-v2.18. We are planning to apply our optimizations to minimap2 master branch.
|
||||||
|
### Citations
|
||||||
|
["Accelerating long-read analysis on modern CPUs"](https://doi.org/10.1101/2021.07.21.453294); Saurabh Kalikar, Chirag Jain, Vasimuddin Md, Sanchit Misra; BioRxiv 2021
|
||||||
|
|
||||||
|
---
|
||||||
|
The original README content of minimap2 follows.
|
||||||
|
|
||||||
|
|
||||||
[](https://github.com/lh3/minimap2/releases)
|
[](https://github.com/lh3/minimap2/releases)
|
||||||
[](https://anaconda.org/bioconda/minimap2)
|
[](https://anaconda.org/bioconda/minimap2)
|
||||||
[](https://pypi.python.org/pypi/mappy)
|
[](https://pypi.python.org/pypi/mappy)
|
||||||
[](https://github.com/lh3/minimap2/actions)
|
[](https://travis-ci.org/lh3/minimap2)
|
||||||
## <a name="started"></a>Getting Started
|
## <a name="started"></a>Getting Started
|
||||||
```sh
|
```sh
|
||||||
git clone https://github.com/lh3/minimap2
|
git clone https://github.com/lh3/minimap2
|
||||||
@@ -12,10 +82,9 @@ cd minimap2 && make
|
|||||||
./minimap2 -x map-ont -d MT-human-ont.mmi test/MT-human.fa
|
./minimap2 -x map-ont -d MT-human-ont.mmi test/MT-human.fa
|
||||||
./minimap2 -a MT-human-ont.mmi test/MT-orang.fa > test.sam
|
./minimap2 -a MT-human-ont.mmi test/MT-orang.fa > test.sam
|
||||||
# use presets (no test data)
|
# use presets (no test data)
|
||||||
./minimap2 -ax map-pb ref.fa pacbio.fq.gz > aln.sam # PacBio CLR genomic reads
|
./minimap2 -ax map-pb ref.fa pacbio.fq.gz > aln.sam # PacBio genomic reads
|
||||||
./minimap2 -ax map-ont ref.fa ont.fq.gz > aln.sam # Oxford Nanopore genomic reads
|
./minimap2 -ax map-ont ref.fa ont.fq.gz > aln.sam # Oxford Nanopore genomic reads
|
||||||
./minimap2 -ax map-hifi ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.19 or later)
|
./minimap2 -ax asm20 ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio CCS genomic reads
|
||||||
./minimap2 -ax asm20 ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.18 or earlier)
|
|
||||||
./minimap2 -ax sr ref.fa read1.fa read2.fa > aln.sam # short genomic paired-end reads
|
./minimap2 -ax sr ref.fa read1.fa read2.fa > aln.sam # short genomic paired-end reads
|
||||||
./minimap2 -ax splice ref.fa rna-reads.fa > aln.sam # spliced long reads (strand unknown)
|
./minimap2 -ax splice ref.fa rna-reads.fa > aln.sam # spliced long reads (strand unknown)
|
||||||
./minimap2 -ax splice -uf -k14 ref.fa reads.fa > aln.sam # noisy Nanopore Direct RNA-seq
|
./minimap2 -ax splice -uf -k14 ref.fa reads.fa > aln.sam # noisy Nanopore Direct RNA-seq
|
||||||
@@ -27,6 +96,9 @@ cd minimap2 && make
|
|||||||
# man page for detailed command line options
|
# man page for detailed command line options
|
||||||
man ./minimap2.1
|
man ./minimap2.1
|
||||||
```
|
```
|
||||||
|
[Unimap][unimap] is recommended for aligning long contigs against a reference
|
||||||
|
genome. It often takes less wall-clock time and is much more sensitive to long
|
||||||
|
insertions and deletions.
|
||||||
|
|
||||||
## Table of Contents
|
## Table of Contents
|
||||||
|
|
||||||
@@ -74,8 +146,8 @@ Detailed evaluations are available from the [minimap2 paper][doi] or the
|
|||||||
Minimap2 is optimized for x86-64 CPUs. You can acquire precompiled binaries from
|
Minimap2 is optimized for x86-64 CPUs. You can acquire precompiled binaries from
|
||||||
the [release page][release] with:
|
the [release page][release] with:
|
||||||
```sh
|
```sh
|
||||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.25/minimap2-2.25_x64-linux.tar.bz2 | tar -jxvf -
|
curl -L https://github.com/lh3/minimap2/releases/download/v2.18/minimap2-2.18_x64-linux.tar.bz2 | tar -jxvf -
|
||||||
./minimap2-2.25_x64-linux/minimap2
|
./minimap2-2.18_x64-linux/minimap2
|
||||||
```
|
```
|
||||||
If you want to compile from the source, you need to have a C compiler, GNU make
|
If you want to compile from the source, you need to have a C compiler, GNU make
|
||||||
and zlib development files installed. Then type `make` in the source code
|
and zlib development files installed. Then type `make` in the source code
|
||||||
@@ -96,7 +168,7 @@ with the ARM related command lines given above.
|
|||||||
|
|
||||||
Without any options, minimap2 takes a reference database and a query sequence
|
Without any options, minimap2 takes a reference database and a query sequence
|
||||||
file as input and produce approximate mapping, without base-level alignment
|
file as input and produce approximate mapping, without base-level alignment
|
||||||
(i.e. coordinates are only approximate and no CIGAR in output), in the [PAF format][paf]:
|
(i.e. no CIGAR), in the [PAF format][paf]:
|
||||||
```sh
|
```sh
|
||||||
minimap2 ref.fa query.fq > approx-mapping.paf
|
minimap2 ref.fa query.fq > approx-mapping.paf
|
||||||
```
|
```
|
||||||
@@ -137,13 +209,13 @@ parameters at the same time. The default setting is the same as `map-ont`.
|
|||||||
#### <a name="map-long-genomic"></a>Map long noisy genomic reads
|
#### <a name="map-long-genomic"></a>Map long noisy genomic reads
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
minimap2 -ax map-pb ref.fa pacbio-reads.fq > aln.sam # for PacBio CLR reads
|
minimap2 -ax map-pb ref.fa pacbio-reads.fq > aln.sam # for PacBio subreads
|
||||||
minimap2 -ax map-ont ref.fa ont-reads.fq > aln.sam # for Oxford Nanopore reads
|
minimap2 -ax map-ont ref.fa ont-reads.fq > aln.sam # for Oxford Nanopore reads
|
||||||
```
|
```
|
||||||
The difference between `map-pb` and `map-ont` is that `map-pb` uses
|
The difference between `map-pb` and `map-ont` is that `map-pb` uses
|
||||||
homopolymer-compressed (HPC) minimizers as seeds, while `map-ont` uses ordinary
|
homopolymer-compressed (HPC) minimizers as seeds, while `map-ont` uses ordinary
|
||||||
minimizers as seeds. Emperical evaluation suggests HPC minimizers improve
|
minimizers as seeds. Emperical evaluation suggests HPC minimizers improve
|
||||||
performance and sensitivity when aligning PacBio CLR reads, but hurt when aligning
|
performance and sensitivity when aligning PacBio reads, but hurt when aligning
|
||||||
Nanopore reads.
|
Nanopore reads.
|
||||||
|
|
||||||
#### <a name="map-long-splice"></a>Map long mRNA/cDNA reads
|
#### <a name="map-long-splice"></a>Map long mRNA/cDNA reads
|
||||||
@@ -204,7 +276,7 @@ strand field. In this case, each line indicates an oriented junction.
|
|||||||
#### <a name="long-overlap"></a>Find overlaps between long reads
|
#### <a name="long-overlap"></a>Find overlaps between long reads
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
minimap2 -x ava-pb reads.fq reads.fq > ovlp.paf # PacBio CLR read overlap
|
minimap2 -x ava-pb reads.fq reads.fq > ovlp.paf # PacBio read overlap
|
||||||
minimap2 -x ava-ont reads.fq reads.fq > ovlp.paf # Oxford Nanopore read overlap
|
minimap2 -x ava-ont reads.fq reads.fq > ovlp.paf # Oxford Nanopore read overlap
|
||||||
```
|
```
|
||||||
Similarly, `ava-pb` uses HPC minimizers while `ava-ont` uses ordinary
|
Similarly, `ava-pb` uses HPC minimizers while `ava-ont` uses ordinary
|
||||||
@@ -251,7 +323,7 @@ To avoid this issue, you can add option `-L` at the minimap2 command line.
|
|||||||
This option moves a long CIGAR to the `CG` tag and leaves a fully clipped CIGAR
|
This option moves a long CIGAR to the `CG` tag and leaves a fully clipped CIGAR
|
||||||
at the SAM CIGAR column. Current tools that don't read CIGAR (e.g. merging and
|
at the SAM CIGAR column. Current tools that don't read CIGAR (e.g. merging and
|
||||||
sorting) still work with such BAM records; tools that read CIGAR will
|
sorting) still work with such BAM records; tools that read CIGAR will
|
||||||
effectively ignore these records. It has been decided that future tools
|
effectively ignore these records. It has been decided that future tools will
|
||||||
will seamlessly recognize long-cigar records generated by option `-L`.
|
will seamlessly recognize long-cigar records generated by option `-L`.
|
||||||
|
|
||||||
**TL;DR**: if you work with ultra-long reads and use tools that only process
|
**TL;DR**: if you work with ultra-long reads and use tools that only process
|
||||||
@@ -272,7 +344,7 @@ CGATCGATAAATAGAGTAG---GAATAGCA
|
|||||||
CGATCG---AATAGAGTAGGTCGAATtGCA
|
CGATCG---AATAGAGTAGGTCGAATtGCA
|
||||||
```
|
```
|
||||||
is represented as `:6-ata:10+gtc:4*at:3`, where `:[0-9]+` represents an
|
is represented as `:6-ata:10+gtc:4*at:3`, where `:[0-9]+` represents an
|
||||||
identical block, `-ata` represents a deletion, `+gtc` an insertion and `*at`
|
identical block, `-ata` represents a deltion, `+gtc` an insertion and `*at`
|
||||||
indicates reference base `a` is substituted with a query base `t`. It is
|
indicates reference base `a` is substituted with a query base `t`. It is
|
||||||
similar to the `MD` SAM tag but is standalone and easier to parse.
|
similar to the `MD` SAM tag but is standalone and easier to parse.
|
||||||
|
|
||||||
@@ -350,11 +422,6 @@ If you use minimap2 in your work, please cite:
|
|||||||
> Li, H. (2018). Minimap2: pairwise alignment for nucleotide sequences.
|
> Li, H. (2018). Minimap2: pairwise alignment for nucleotide sequences.
|
||||||
> *Bioinformatics*, **34**:3094-3100. [doi:10.1093/bioinformatics/bty191][doi]
|
> *Bioinformatics*, **34**:3094-3100. [doi:10.1093/bioinformatics/bty191][doi]
|
||||||
|
|
||||||
and/or:
|
|
||||||
|
|
||||||
> Li, H. (2021). New strategies to improve minimap2 alignment accuracy.
|
|
||||||
> *Bioinformatics*, **37**:4572-4574. [doi:10.1093/bioinformatics/btab705][doi2]
|
|
||||||
|
|
||||||
## <a name="dguide"></a>Developers' Guide
|
## <a name="dguide"></a>Developers' Guide
|
||||||
|
|
||||||
Minimap2 is not only a command line tool, but also a programming library.
|
Minimap2 is not only a command line tool, but also a programming library.
|
||||||
@@ -404,6 +471,5 @@ mappy` or [from BioConda][mappyconda] via `conda install -c bioconda mappy`.
|
|||||||
[manpage]: https://lh3.github.io/minimap2/minimap2.html
|
[manpage]: https://lh3.github.io/minimap2/minimap2.html
|
||||||
[manpage-cs]: https://lh3.github.io/minimap2/minimap2.html#10
|
[manpage-cs]: https://lh3.github.io/minimap2/minimap2.html#10
|
||||||
[doi]: https://doi.org/10.1093/bioinformatics/bty191
|
[doi]: https://doi.org/10.1093/bioinformatics/bty191
|
||||||
[doi2]: https://doi.org/10.1093/bioinformatics/btab705
|
[smide]: https://github.com/nemequ/simde
|
||||||
[simde]: https://github.com/nemequ/simde
|
|
||||||
[unimap]: https://github.com/lh3/unimap
|
[unimap]: https://github.com/lh3/unimap
|
||||||
|
|||||||
@@ -1,3 +1,33 @@
|
|||||||
|
/* The MIT License
|
||||||
|
|
||||||
|
Copyright (c) 2018- Dana-Farber Cancer Institute
|
||||||
|
2017-2018 Broad Institute, Inc.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
|
Modified Copyright (C) 2021 Intel Corporation
|
||||||
|
Contacts: Saurabh Kalikar <saurabh.kalikar@intel.com>;
|
||||||
|
Vasimuddin Md <vasimuddin.md@intel.com>; Sanchit Misra <sanchit.misra@intel.com>;
|
||||||
|
Chirag Jain <chirag@iisc.ac.in>; Heng Li <hli@jimmy.harvard.edu>
|
||||||
|
*/
|
||||||
|
|
||||||
#include <assert.h>
|
#include <assert.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
@@ -5,7 +35,9 @@
|
|||||||
#include "minimap.h"
|
#include "minimap.h"
|
||||||
#include "mmpriv.h"
|
#include "mmpriv.h"
|
||||||
#include "ksw2.h"
|
#include "ksw2.h"
|
||||||
|
#include "ksw2_extd2_avx.h"
|
||||||
|
#include <x86intrin.h>
|
||||||
|
extern uint64_t alignment_time;
|
||||||
static void ksw_gen_simple_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t sc_ambi)
|
static void ksw_gen_simple_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t sc_ambi)
|
||||||
{
|
{
|
||||||
int i, j;
|
int i, j;
|
||||||
@@ -53,16 +85,16 @@ static int mm_test_zdrop(void *km, const mm_mapopt_t *opt, const uint8_t *qseq,
|
|||||||
// find the score and the region where score drops most along diagonal
|
// find the score and the region where score drops most along diagonal
|
||||||
for (k = 0, score = 0; k < n_cigar; ++k) {
|
for (k = 0, score = 0; k < n_cigar; ++k) {
|
||||||
uint32_t l, op = cigar[k]&0xf, len = cigar[k]>>4;
|
uint32_t l, op = cigar[k]&0xf, len = cigar[k]>>4;
|
||||||
if (op == MM_CIGAR_MATCH) {
|
if (op == 0) {
|
||||||
for (l = 0; l < len; ++l) {
|
for (l = 0; l < len; ++l) {
|
||||||
score += mat[tseq[i + l] * 5 + qseq[j + l]];
|
score += mat[tseq[i + l] * 5 + qseq[j + l]];
|
||||||
update_max_zdrop(score, i+l, j+l, &max, &max_i, &max_j, opt->e, &max_zdrop, pos);
|
update_max_zdrop(score, i+l, j+l, &max, &max_i, &max_j, opt->e, &max_zdrop, pos);
|
||||||
}
|
}
|
||||||
i += len, j += len;
|
i += len, j += len;
|
||||||
} else if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL || op == MM_CIGAR_N_SKIP) {
|
} else if (op == 1 || op == 2 || op == 3) {
|
||||||
score -= opt->q + opt->e * len;
|
score -= opt->q + opt->e * len;
|
||||||
if (op == MM_CIGAR_INS) j += len;
|
if (op == 1) j += len; // insertion
|
||||||
else i += len;
|
else i += len; // deletion
|
||||||
update_max_zdrop(score, i, j, &max, &max_i, &max_j, opt->e, &max_zdrop, pos);
|
update_max_zdrop(score, i, j, &max, &max_i, &max_j, opt->e, &max_zdrop, pos);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -98,12 +130,12 @@ static void mm_fix_cigar(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq,
|
|||||||
for (k = 0; k < p->n_cigar; ++k) { // indel left alignment
|
for (k = 0; k < p->n_cigar; ++k) { // indel left alignment
|
||||||
uint32_t op = p->cigar[k]&0xf, len = p->cigar[k]>>4;
|
uint32_t op = p->cigar[k]&0xf, len = p->cigar[k]>>4;
|
||||||
if (len == 0) to_shrink = 1;
|
if (len == 0) to_shrink = 1;
|
||||||
if (op == MM_CIGAR_MATCH) {
|
if (op == 0) {
|
||||||
toff += len, qoff += len;
|
toff += len, qoff += len;
|
||||||
} else if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL) {
|
} else if (op == 1 || op == 2) { // insertion or deletion
|
||||||
if (k > 0 && k < p->n_cigar - 1 && (p->cigar[k-1]&0xf) == 0 && (p->cigar[k+1]&0xf) == 0) {
|
if (k > 0 && k < p->n_cigar - 1 && (p->cigar[k-1]&0xf) == 0 && (p->cigar[k+1]&0xf) == 0) {
|
||||||
int l, prev_len = p->cigar[k-1] >> 4;
|
int l, prev_len = p->cigar[k-1] >> 4;
|
||||||
if (op == MM_CIGAR_INS) {
|
if (op == 1) {
|
||||||
for (l = 0; l < prev_len; ++l)
|
for (l = 0; l < prev_len; ++l)
|
||||||
if (qseq[qoff - 1 - l] != qseq[qoff + len - 1 - l])
|
if (qseq[qoff - 1 - l] != qseq[qoff + len - 1 - l])
|
||||||
break;
|
break;
|
||||||
@@ -116,9 +148,9 @@ static void mm_fix_cigar(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq,
|
|||||||
p->cigar[k-1] -= l<<4, p->cigar[k+1] += l<<4, qoff -= l, toff -= l;
|
p->cigar[k-1] -= l<<4, p->cigar[k+1] += l<<4, qoff -= l, toff -= l;
|
||||||
if (l == prev_len) to_shrink = 1;
|
if (l == prev_len) to_shrink = 1;
|
||||||
}
|
}
|
||||||
if (op == MM_CIGAR_INS) qoff += len;
|
if (op == 1) qoff += len;
|
||||||
else toff += len;
|
else toff += len;
|
||||||
} else if (op == MM_CIGAR_N_SKIP) {
|
} else if (op == 3) {
|
||||||
toff += len;
|
toff += len;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -128,13 +160,13 @@ static void mm_fix_cigar(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq,
|
|||||||
uint32_t l, s[3] = {0,0,0};
|
uint32_t l, s[3] = {0,0,0};
|
||||||
for (l = k; l < p->n_cigar; ++l) { // count number of adjacent I and D
|
for (l = k; l < p->n_cigar; ++l) { // count number of adjacent I and D
|
||||||
uint32_t op = p->cigar[l]&0xf;
|
uint32_t op = p->cigar[l]&0xf;
|
||||||
if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL || p->cigar[l]>>4 == 0)
|
if (op == 1 || op == 2 || p->cigar[l]>>4 == 0)
|
||||||
s[op] += p->cigar[l] >> 4;
|
s[op] += p->cigar[l] >> 4;
|
||||||
else break;
|
else break;
|
||||||
}
|
}
|
||||||
if (s[1] > 0 && s[2] > 0 && l - k > 2) { // turn to a single I and a single D
|
if (s[1] > 0 && s[2] > 0 && l - k > 2) { // turn to a single I and a single D
|
||||||
p->cigar[k] = s[1]<<4|MM_CIGAR_INS;
|
p->cigar[k] = s[1]<<4|1;
|
||||||
p->cigar[k+1] = s[2]<<4|MM_CIGAR_DEL;
|
p->cigar[k+1] = s[2]<<4|2;
|
||||||
for (k += 2; k < l; ++k)
|
for (k += 2; k < l; ++k)
|
||||||
p->cigar[k] &= 0xf;
|
p->cigar[k] &= 0xf;
|
||||||
to_shrink = 1;
|
to_shrink = 1;
|
||||||
@@ -154,9 +186,9 @@ static void mm_fix_cigar(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq,
|
|||||||
else p->cigar[k+1] += p->cigar[k]>>4<<4; // add length to the next CIGAR operator
|
else p->cigar[k+1] += p->cigar[k]>>4<<4; // add length to the next CIGAR operator
|
||||||
p->n_cigar = l;
|
p->n_cigar = l;
|
||||||
}
|
}
|
||||||
if ((p->cigar[0]&0xf) == MM_CIGAR_INS || (p->cigar[0]&0xf) == MM_CIGAR_DEL) { // get rid of leading I or D
|
if ((p->cigar[0]&0xf) == 1 || (p->cigar[0]&0xf) == 2) { // get rid of leading I or D
|
||||||
int32_t l = p->cigar[0] >> 4;
|
int32_t l = p->cigar[0] >> 4;
|
||||||
if ((p->cigar[0]&0xf) == MM_CIGAR_INS) {
|
if ((p->cigar[0]&0xf) == 1) {
|
||||||
if (r->rev) r->qe -= l;
|
if (r->rev) r->qe -= l;
|
||||||
else r->qs += l;
|
else r->qs += l;
|
||||||
*qshift = l;
|
*qshift = l;
|
||||||
@@ -174,7 +206,7 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
|||||||
if (r->p == 0) return;
|
if (r->p == 0) return;
|
||||||
for (k = 0; k < r->p->n_cigar; ++k) {
|
for (k = 0; k < r->p->n_cigar; ++k) {
|
||||||
uint32_t op = r->p->cigar[k]&0xf, len = r->p->cigar[k]>>4;
|
uint32_t op = r->p->cigar[k]&0xf, len = r->p->cigar[k]>>4;
|
||||||
if (op == MM_CIGAR_MATCH) {
|
if (op == 0) {
|
||||||
while (len > 0) {
|
while (len > 0) {
|
||||||
for (l = 0; l < len && qseq[qoff + l] == tseq[toff + l]; ++l) {} // run of "="; TODO: N<=>N is converted to "="
|
for (l = 0; l < len && qseq[qoff + l] == tseq[toff + l]; ++l) {} // run of "="; TODO: N<=>N is converted to "="
|
||||||
if (l > 0) { ++n_EQX; len -= l; toff += l; qoff += l; }
|
if (l > 0) { ++n_EQX; len -= l; toff += l; qoff += l; }
|
||||||
@@ -183,11 +215,11 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
|||||||
if (l > 0) { ++n_EQX; len -= l; toff += l; qoff += l; }
|
if (l > 0) { ++n_EQX; len -= l; toff += l; qoff += l; }
|
||||||
}
|
}
|
||||||
++n_M;
|
++n_M;
|
||||||
} else if (op == MM_CIGAR_INS) {
|
} else if (op == 1) { // insertion
|
||||||
qoff += len;
|
qoff += len;
|
||||||
} else if (op == MM_CIGAR_DEL) {
|
} else if (op == 2) { // deletion
|
||||||
toff += len;
|
toff += len;
|
||||||
} else if (op == MM_CIGAR_N_SKIP) {
|
} else if (op == 3) { // intron
|
||||||
toff += len;
|
toff += len;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -195,7 +227,7 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
|||||||
if (n_EQX == n_M) {
|
if (n_EQX == n_M) {
|
||||||
for (k = 0; k < r->p->n_cigar; ++k) {
|
for (k = 0; k < r->p->n_cigar; ++k) {
|
||||||
uint32_t op = r->p->cigar[k]&0xf, len = r->p->cigar[k]>>4;
|
uint32_t op = r->p->cigar[k]&0xf, len = r->p->cigar[k]>>4;
|
||||||
if (op == MM_CIGAR_MATCH) r->p->cigar[k] = len << 4 | MM_CIGAR_EQ_MATCH;
|
if (op == 0) r->p->cigar[k] = len << 4 | 7;
|
||||||
}
|
}
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
@@ -209,25 +241,25 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
|||||||
toff = qoff = m = 0;
|
toff = qoff = m = 0;
|
||||||
for (k = 0; k < r->p->n_cigar; ++k) {
|
for (k = 0; k < r->p->n_cigar; ++k) {
|
||||||
uint32_t op = r->p->cigar[k]&0xf, len = r->p->cigar[k]>>4;
|
uint32_t op = r->p->cigar[k]&0xf, len = r->p->cigar[k]>>4;
|
||||||
if (op == MM_CIGAR_MATCH) {
|
if (op == 0) { // match/mismatch
|
||||||
while (len > 0) {
|
while (len > 0) {
|
||||||
// match
|
// match
|
||||||
for (l = 0; l < len && qseq[qoff + l] == tseq[toff + l]; ++l) {}
|
for (l = 0; l < len && qseq[qoff + l] == tseq[toff + l]; ++l) {}
|
||||||
if (l > 0) p->cigar[m++] = l << 4 | MM_CIGAR_EQ_MATCH;
|
if (l > 0) p->cigar[m++] = l << 4 | 7;
|
||||||
len -= l;
|
len -= l;
|
||||||
toff += l, qoff += l;
|
toff += l, qoff += l;
|
||||||
// mismatch
|
// mismatch
|
||||||
for (l = 0; l < len && qseq[qoff + l] != tseq[toff + l]; ++l) {}
|
for (l = 0; l < len && qseq[qoff + l] != tseq[toff + l]; ++l) {}
|
||||||
if (l > 0) p->cigar[m++] = l << 4 | MM_CIGAR_X_MISMATCH;
|
if (l > 0) p->cigar[m++] = l << 4 | 8;
|
||||||
len -= l;
|
len -= l;
|
||||||
toff += l, qoff += l;
|
toff += l, qoff += l;
|
||||||
}
|
}
|
||||||
continue;
|
continue;
|
||||||
} else if (op == MM_CIGAR_INS) {
|
} else if (op == 1) { // insertion
|
||||||
qoff += len;
|
qoff += len;
|
||||||
} else if (op == MM_CIGAR_DEL) {
|
} else if (op == 2) { // deletion
|
||||||
toff += len;
|
toff += len;
|
||||||
} else if (op == MM_CIGAR_N_SKIP) {
|
} else if (op == 3) { // intron
|
||||||
toff += len;
|
toff += len;
|
||||||
}
|
}
|
||||||
p->cigar[m++] = r->p->cigar[k];
|
p->cigar[m++] = r->p->cigar[k];
|
||||||
@@ -237,11 +269,10 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
|||||||
r->p = p;
|
r->p = p;
|
||||||
}
|
}
|
||||||
|
|
||||||
static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq, const int8_t *mat, int8_t q, int8_t e, int is_eqx, int log_gap)
|
static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq, const int8_t *mat, int8_t q, int8_t e, int is_eqx)
|
||||||
{
|
{
|
||||||
uint32_t k, l;
|
uint32_t k, l;
|
||||||
int32_t qshift, tshift, toff = 0, qoff = 0;
|
int32_t s = 0, max = 0, qshift, tshift, toff = 0, qoff = 0;
|
||||||
double s = 0.0, max = 0.0;
|
|
||||||
mm_extra_t *p = r->p;
|
mm_extra_t *p = r->p;
|
||||||
if (p == 0) return;
|
if (p == 0) return;
|
||||||
mm_fix_cigar(r, qseq, tseq, &qshift, &tshift);
|
mm_fix_cigar(r, qseq, tseq, &qshift, &tshift);
|
||||||
@@ -249,7 +280,7 @@ static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *ts
|
|||||||
r->blen = r->mlen = 0;
|
r->blen = r->mlen = 0;
|
||||||
for (k = 0; k < p->n_cigar; ++k) {
|
for (k = 0; k < p->n_cigar; ++k) {
|
||||||
uint32_t op = p->cigar[k]&0xf, len = p->cigar[k]>>4;
|
uint32_t op = p->cigar[k]&0xf, len = p->cigar[k]>>4;
|
||||||
if (op == MM_CIGAR_MATCH) {
|
if (op == 0) { // match/mismatch
|
||||||
int n_ambi = 0, n_diff = 0;
|
int n_ambi = 0, n_diff = 0;
|
||||||
for (l = 0; l < len; ++l) {
|
for (l = 0; l < len; ++l) {
|
||||||
int cq = qseq[qoff + l], ct = tseq[toff + l];
|
int cq = qseq[qoff + l], ct = tseq[toff + l];
|
||||||
@@ -261,29 +292,27 @@ static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *ts
|
|||||||
}
|
}
|
||||||
r->blen += len - n_ambi, r->mlen += len - (n_ambi + n_diff), p->n_ambi += n_ambi;
|
r->blen += len - n_ambi, r->mlen += len - (n_ambi + n_diff), p->n_ambi += n_ambi;
|
||||||
toff += len, qoff += len;
|
toff += len, qoff += len;
|
||||||
} else if (op == MM_CIGAR_INS) {
|
} else if (op == 1) { // insertion
|
||||||
int n_ambi = 0;
|
int n_ambi = 0;
|
||||||
for (l = 0; l < len; ++l)
|
for (l = 0; l < len; ++l)
|
||||||
if (qseq[qoff + l] > 3) ++n_ambi;
|
if (qseq[qoff + l] > 3) ++n_ambi;
|
||||||
r->blen += len - n_ambi, p->n_ambi += n_ambi;
|
r->blen += len - n_ambi, p->n_ambi += n_ambi;
|
||||||
if (log_gap) s -= q + (double)e * mg_log2(1.0 + len);
|
s -= q + e * len;
|
||||||
else s -= q + e;
|
|
||||||
if (s < 0) s = 0;
|
if (s < 0) s = 0;
|
||||||
qoff += len;
|
qoff += len;
|
||||||
} else if (op == MM_CIGAR_DEL) {
|
} else if (op == 2) { // deletion
|
||||||
int n_ambi = 0;
|
int n_ambi = 0;
|
||||||
for (l = 0; l < len; ++l)
|
for (l = 0; l < len; ++l)
|
||||||
if (tseq[toff + l] > 3) ++n_ambi;
|
if (tseq[toff + l] > 3) ++n_ambi;
|
||||||
r->blen += len - n_ambi, p->n_ambi += n_ambi;
|
r->blen += len - n_ambi, p->n_ambi += n_ambi;
|
||||||
if (log_gap) s -= q + (double)e * mg_log2(1.0 + len);
|
s -= q + e * len;
|
||||||
else s -= q + e;
|
|
||||||
if (s < 0) s = 0;
|
if (s < 0) s = 0;
|
||||||
toff += len;
|
toff += len;
|
||||||
} else if (op == MM_CIGAR_N_SKIP) {
|
} else if (op == 3) { // intron
|
||||||
toff += len;
|
toff += len;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
p->dp_max = (int32_t)(max + .499);
|
p->dp_max = max;
|
||||||
assert(qoff == r->qe - r->qs && toff == r->re - r->rs);
|
assert(qoff == r->qe - r->qs && toff == r->re - r->rs);
|
||||||
if (is_eqx) mm_update_cigar_eqx(r, qseq, tseq); // NB: it has to be called here as changes to qseq and tseq are not returned
|
if (is_eqx) mm_update_cigar_eqx(r, qseq, tseq); // NB: it has to be called here as changes to qseq and tseq are not returned
|
||||||
}
|
}
|
||||||
@@ -315,6 +344,10 @@ static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, uint32_t *cigar) //
|
|||||||
|
|
||||||
static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc, const int8_t *mat, int w, int end_bonus, int zdrop, int flag, ksw_extz_t *ez)
|
static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc, const int8_t *mat, int w, int end_bonus, int zdrop, int flag, ksw_extz_t *ez)
|
||||||
{
|
{
|
||||||
|
#ifdef MANUAL_PROFILING
|
||||||
|
uint64_t align_start = __rdtsc();
|
||||||
|
#endif
|
||||||
|
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
||||||
int i;
|
int i;
|
||||||
fprintf(stderr, "===> q=(%d,%d), e=(%d,%d), bw=%d, flag=%d, zdrop=%d <===\n", opt->q, opt->q2, opt->e, opt->e2, w, flag, opt->zdrop);
|
fprintf(stderr, "===> q=(%d,%d), e=(%d,%d), bw=%d, flag=%d, zdrop=%d <===\n", opt->q, opt->q2, opt->e, opt->e2, w, flag, opt->zdrop);
|
||||||
@@ -326,21 +359,32 @@ static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint
|
|||||||
if (opt->max_sw_mat > 0 && (int64_t)tlen * qlen > opt->max_sw_mat) {
|
if (opt->max_sw_mat > 0 && (int64_t)tlen * qlen > opt->max_sw_mat) {
|
||||||
ksw_reset_extz(ez);
|
ksw_reset_extz(ez);
|
||||||
ez->zdropped = 1;
|
ez->zdropped = 1;
|
||||||
} else if (opt->flag & MM_F_SPLICE) {
|
} else if (opt->flag & MM_F_SPLICE)
|
||||||
int flag_tmp = flag;
|
ksw_exts2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->noncan, zdrop, opt->junc_bonus, flag, junc, ez);
|
||||||
if (!(opt->flag & MM_F_SPLICE_OLD)) flag_tmp |= KSW_EZ_SPLICE_CMPLX;
|
else if (opt->q == opt->q2 && opt->e == opt->e2)
|
||||||
ksw_exts2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->noncan, zdrop, opt->junc_bonus, flag_tmp, junc, ez);
|
|
||||||
} else if (opt->q == opt->q2 && opt->e == opt->e2)
|
|
||||||
ksw_extz2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, w, zdrop, end_bonus, flag, ez);
|
ksw_extz2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, w, zdrop, end_bonus, flag, ez);
|
||||||
else
|
else{
|
||||||
|
#if defined (ALIGN_AVX) && (defined(__AVX512BW__) || (defined(__AVX2__) && defined(APPLY_AVX2)))
|
||||||
|
#ifdef __AVX512BW__
|
||||||
|
|
||||||
|
ksw_extd2_avx512(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, flag, ez);
|
||||||
|
#elif __AVX2__
|
||||||
|
ksw_extd2_avx2(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, flag, ez);
|
||||||
|
#endif
|
||||||
|
#else
|
||||||
ksw_extd2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, flag, ez);
|
ksw_extd2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, flag, ez);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
||||||
int i;
|
int i;
|
||||||
fprintf(stderr, "score=%d, cigar=", ez->score);
|
fprintf(stderr, "score=%d, cigar=", ez->score);
|
||||||
for (i = 0; i < ez->n_cigar; ++i)
|
for (i = 0; i < ez->n_cigar; ++i)
|
||||||
fprintf(stderr, "%d%c", ez->cigar[i]>>4, MM_CIGAR_STR[ez->cigar[i]&0xf]);
|
fprintf(stderr, "%d%c", ez->cigar[i]>>4, "MIDN"[ez->cigar[i]&0xf]);
|
||||||
fprintf(stderr, "\n");
|
fprintf(stderr, "\n");
|
||||||
}
|
}
|
||||||
|
#ifdef MANUAL_PROFILING
|
||||||
|
alignment_time += (__rdtsc() - align_start);
|
||||||
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline int mm_get_hplen_back(const mm_idx_t *mi, uint32_t rid, uint32_t x)
|
static inline int mm_get_hplen_back(const mm_idx_t *mi, uint32_t rid, uint32_t x)
|
||||||
@@ -538,13 +582,8 @@ static int mm_seed_ext_score(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
|||||||
re = re + ext_len < (int32_t)mi->seq[rid].len? re + ext_len : mi->seq[rid].len;
|
re = re + ext_len < (int32_t)mi->seq[rid].len? re + ext_len : mi->seq[rid].len;
|
||||||
qe = qe + ext_len < qlen? qe + ext_len : qlen;
|
qe = qe + ext_len < qlen? qe + ext_len : qlen;
|
||||||
tseq = (uint8_t*)kmalloc(km, re - rs);
|
tseq = (uint8_t*)kmalloc(km, re - rs);
|
||||||
if (opt->flag & MM_F_QSTRAND) {
|
mm_idx_getseq(mi, rid, rs, re, tseq);
|
||||||
qseq = qseq0[0] + qs;
|
qseq = qseq0[a->x>>63] + qs;
|
||||||
mm_idx_getseq2(mi, a->x>>63, rid, rs, re, tseq);
|
|
||||||
} else {
|
|
||||||
qseq = qseq0[a->x>>63] + qs;
|
|
||||||
mm_idx_getseq(mi, rid, rs, re, tseq);
|
|
||||||
}
|
|
||||||
qp = ksw_ll_qinit(km, 2, qe - qs, qseq, 5, mat);
|
qp = ksw_ll_qinit(km, 2, qe - qs, qseq, 5, mat);
|
||||||
score = ksw_ll_i16(qp, re - rs, tseq, opt->q, opt->e, &q_off, &t_off);
|
score = ksw_ll_i16(qp, re - rs, tseq, opt->q, opt->e, &q_off, &t_off);
|
||||||
kfree(km, tseq);
|
kfree(km, tseq);
|
||||||
@@ -577,7 +616,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
int is_sr = !!(opt->flag & MM_F_SR), is_splice = !!(opt->flag & MM_F_SPLICE);
|
int is_sr = !!(opt->flag & MM_F_SR), is_splice = !!(opt->flag & MM_F_SPLICE);
|
||||||
int32_t rid = a[r->as].x<<1>>33, rev = a[r->as].x>>63, as1, cnt1;
|
int32_t rid = a[r->as].x<<1>>33, rev = a[r->as].x>>63, as1, cnt1;
|
||||||
uint8_t *tseq, *qseq, *junc;
|
uint8_t *tseq, *qseq, *junc;
|
||||||
int32_t i, l, bw, bw_long, dropped = 0, extra_flag = 0, rs0, re0, qs0, qe0;
|
int32_t i, l, bw, dropped = 0, extra_flag = 0, rs0, re0, qs0, qe0;
|
||||||
int32_t rs, re, qs, qe;
|
int32_t rs, re, qs, qe;
|
||||||
int32_t rs1, qs1, re1, qe1;
|
int32_t rs1, qs1, re1, qe1;
|
||||||
int8_t mat[25];
|
int8_t mat[25];
|
||||||
@@ -588,8 +627,6 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
if (r->cnt == 0) return;
|
if (r->cnt == 0) return;
|
||||||
ksw_gen_simple_mat(5, mat, opt->a, opt->b, opt->sc_ambi);
|
ksw_gen_simple_mat(5, mat, opt->a, opt->b, opt->sc_ambi);
|
||||||
bw = (int)(opt->bw * 1.5 + 1.);
|
bw = (int)(opt->bw * 1.5 + 1.);
|
||||||
bw_long = (int)(opt->bw_long * 1.5 + 1.);
|
|
||||||
if (bw_long < bw) bw_long = bw;
|
|
||||||
|
|
||||||
if (is_sr && !(mi->flag & MM_I_HPC)) {
|
if (is_sr && !(mi->flag & MM_I_HPC)) {
|
||||||
mm_max_stretch(r, a, &as1, &cnt1);
|
mm_max_stretch(r, a, &as1, &cnt1);
|
||||||
@@ -700,13 +737,8 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
junc = (uint8_t*)kmalloc(km, re0 - rs0);
|
junc = (uint8_t*)kmalloc(km, re0 - rs0);
|
||||||
|
|
||||||
if (qs > 0 && rs > 0) { // left extension; probably the condition can be changed to "qs > qs0 && rs > rs0"
|
if (qs > 0 && rs > 0) { // left extension; probably the condition can be changed to "qs > qs0 && rs > rs0"
|
||||||
if (opt->flag & MM_F_QSTRAND) {
|
qseq = &qseq0[rev][qs0];
|
||||||
qseq = &qseq0[0][qs0];
|
mm_idx_getseq(mi, rid, rs0, rs, tseq);
|
||||||
mm_idx_getseq2(mi, rev, rid, rs0, rs, tseq);
|
|
||||||
} else {
|
|
||||||
qseq = &qseq0[rev][qs0];
|
|
||||||
mm_idx_getseq(mi, rid, rs0, rs, tseq);
|
|
||||||
}
|
|
||||||
mm_idx_bed_junc(mi, rid, rs0, rs, junc);
|
mm_idx_bed_junc(mi, rid, rs0, rs, junc);
|
||||||
mm_seq_rev(qs - qs0, qseq);
|
mm_seq_rev(qs - qs0, qseq);
|
||||||
mm_seq_rev(rs - rs0, tseq);
|
mm_seq_rev(rs - rs0, tseq);
|
||||||
@@ -731,17 +763,12 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
} else mm_adjust_minier(mi, qseq0, &a[as1 + i], &re, &qe);
|
} else mm_adjust_minier(mi, qseq0, &a[as1 + i], &re, &qe);
|
||||||
re1 = re, qe1 = qe;
|
re1 = re, qe1 = qe;
|
||||||
if (i == cnt1 - 1 || (a[as1+i].y&MM_SEED_LONG_JOIN) || (qe - qs >= opt->min_ksw_len && re - rs >= opt->min_ksw_len)) {
|
if (i == cnt1 - 1 || (a[as1+i].y&MM_SEED_LONG_JOIN) || (qe - qs >= opt->min_ksw_len && re - rs >= opt->min_ksw_len)) {
|
||||||
int j, bw1 = bw_long, zdrop_code;
|
int j, bw1 = bw, zdrop_code;
|
||||||
if (a[as1+i].y & MM_SEED_LONG_JOIN)
|
if (a[as1+i].y & MM_SEED_LONG_JOIN)
|
||||||
bw1 = qe - qs > re - rs? qe - qs : re - rs;
|
bw1 = qe - qs > re - rs? qe - qs : re - rs;
|
||||||
// perform alignment
|
// perform alignment
|
||||||
if (opt->flag & MM_F_QSTRAND) {
|
qseq = &qseq0[rev][qs];
|
||||||
qseq = &qseq0[0][qs];
|
mm_idx_getseq(mi, rid, rs, re, tseq);
|
||||||
mm_idx_getseq2(mi, rev, rid, rs, re, tseq);
|
|
||||||
} else {
|
|
||||||
qseq = &qseq0[rev][qs];
|
|
||||||
mm_idx_getseq(mi, rid, rs, re, tseq);
|
|
||||||
}
|
|
||||||
mm_idx_bed_junc(mi, rid, rs, re, junc);
|
mm_idx_bed_junc(mi, rid, rs, re, junc);
|
||||||
if (is_sr) { // perform ungapped alignment
|
if (is_sr) { // perform ungapped alignment
|
||||||
assert(qe - qs == re - rs);
|
assert(qe - qs == re - rs);
|
||||||
@@ -750,7 +777,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
if (qseq[j] >= 4 || tseq[j] >= 4) ez->score += opt->e2;
|
if (qseq[j] >= 4 || tseq[j] >= 4) ez->score += opt->e2;
|
||||||
else ez->score += qseq[j] == tseq[j]? opt->a : -opt->b;
|
else ez->score += qseq[j] == tseq[j]? opt->a : -opt->b;
|
||||||
}
|
}
|
||||||
ez->cigar = ksw_push_cigar(km, &ez->n_cigar, &ez->m_cigar, ez->cigar, MM_CIGAR_MATCH, qe - qs);
|
ez->cigar = ksw_push_cigar(km, &ez->n_cigar, &ez->m_cigar, ez->cigar, 0, qe - qs);
|
||||||
} else { // perform normal gapped alignment
|
} else { // perform normal gapped alignment
|
||||||
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, opt->zdrop, extra_flag|KSW_EZ_APPROX_MAX, ez); // first pass: with approximate Z-drop
|
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, opt->zdrop, extra_flag|KSW_EZ_APPROX_MAX, ez); // first pass: with approximate Z-drop
|
||||||
}
|
}
|
||||||
@@ -777,7 +804,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
re1 = rs + (ez->max_t + 1);
|
re1 = rs + (ez->max_t + 1);
|
||||||
qe1 = qs + (ez->max_q + 1);
|
qe1 = qs + (ez->max_q + 1);
|
||||||
if (cnt1 - (j + 1) >= opt->min_cnt) {
|
if (cnt1 - (j + 1) >= opt->min_cnt) {
|
||||||
mm_split_reg(r, r2, as1 + j + 1 - r->as, qlen, a, !!(opt->flag&MM_F_QSTRAND));
|
mm_split_reg(r, r2, as1 + j + 1 - r->as, qlen, a);
|
||||||
if (zdrop_code == 2) r2->split_inv = 1;
|
if (zdrop_code == 2) r2->split_inv = 1;
|
||||||
}
|
}
|
||||||
break;
|
break;
|
||||||
@@ -787,13 +814,8 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
}
|
}
|
||||||
|
|
||||||
if (!dropped && qe < qe0 && re < re0) { // right extension
|
if (!dropped && qe < qe0 && re < re0) { // right extension
|
||||||
if (opt->flag & MM_F_QSTRAND) {
|
qseq = &qseq0[rev][qe];
|
||||||
qseq = &qseq0[0][qe];
|
mm_idx_getseq(mi, rid, re, re0, tseq);
|
||||||
mm_idx_getseq2(mi, rev, rid, re, re0, tseq);
|
|
||||||
} else {
|
|
||||||
qseq = &qseq0[rev][qe];
|
|
||||||
mm_idx_getseq(mi, rid, re, re0, tseq);
|
|
||||||
}
|
|
||||||
mm_idx_bed_junc(mi, rid, re, re0, junc);
|
mm_idx_bed_junc(mi, rid, re, re0, junc);
|
||||||
mm_align_pair(km, opt, qe0 - qe, qseq, re0 - re, tseq, junc, mat, bw, opt->end_bonus, opt->zdrop, extra_flag|KSW_EZ_EXTZ_ONLY, ez);
|
mm_align_pair(km, opt, qe0 - qe, qseq, re0 - re, tseq, junc, mat, bw, opt->end_bonus, opt->zdrop, extra_flag|KSW_EZ_EXTZ_ONLY, ez);
|
||||||
if (ez->n_cigar > 0) {
|
if (ez->n_cigar > 0) {
|
||||||
@@ -806,19 +828,13 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
assert(qe1 <= qlen);
|
assert(qe1 <= qlen);
|
||||||
|
|
||||||
r->rs = rs1, r->re = re1;
|
r->rs = rs1, r->re = re1;
|
||||||
if (!rev || (opt->flag & MM_F_QSTRAND)) r->qs = qs1, r->qe = qe1;
|
if (rev) r->qs = qlen - qe1, r->qe = qlen - qs1;
|
||||||
else r->qs = qlen - qe1, r->qe = qlen - qs1;
|
else r->qs = qs1, r->qe = qe1;
|
||||||
|
|
||||||
assert(re1 - rs1 <= re0 - rs0);
|
assert(re1 - rs1 <= re0 - rs0);
|
||||||
if (r->p) {
|
if (r->p) {
|
||||||
if (opt->flag & MM_F_QSTRAND) {
|
mm_idx_getseq(mi, rid, rs1, re1, tseq);
|
||||||
mm_idx_getseq2(mi, r->rev, rid, rs1, re1, tseq);
|
mm_update_extra(r, &qseq0[r->rev][qs1], tseq, mat, opt->q, opt->e, opt->flag & MM_F_EQX);
|
||||||
qseq = &qseq0[0][qs1];
|
|
||||||
} else {
|
|
||||||
mm_idx_getseq(mi, rid, rs1, re1, tseq);
|
|
||||||
qseq = &qseq0[r->rev][qs1];
|
|
||||||
}
|
|
||||||
mm_update_extra(r, qseq, tseq, mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & MM_F_SR));
|
|
||||||
if (rev && r->p->trans_strand)
|
if (rev && r->p->trans_strand)
|
||||||
r->p->trans_strand ^= 3; // flip to the read strand
|
r->p->trans_strand ^= 3; // flip to the read strand
|
||||||
}
|
}
|
||||||
@@ -828,7 +844,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
}
|
}
|
||||||
|
|
||||||
static int mm_align1_inv(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, uint8_t *qseq0[2], const mm_reg1_t *r1, const mm_reg1_t *r2, mm_reg1_t *r_inv, ksw_extz_t *ez)
|
static int mm_align1_inv(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, uint8_t *qseq0[2], const mm_reg1_t *r1, const mm_reg1_t *r2, mm_reg1_t *r_inv, ksw_extz_t *ez)
|
||||||
{ // NB: this doesn't work with the qstrand mode
|
{
|
||||||
int tl, ql, score, ret = 0, q_off, t_off;
|
int tl, ql, score, ret = 0, q_off, t_off;
|
||||||
uint8_t *tseq, *qseq;
|
uint8_t *tseq, *qseq;
|
||||||
int8_t mat[25];
|
int8_t mat[25];
|
||||||
@@ -877,7 +893,7 @@ static int mm_align1_inv(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, i
|
|||||||
}
|
}
|
||||||
r_inv->rs = r1->re + t_off;
|
r_inv->rs = r1->re + t_off;
|
||||||
r_inv->re = r_inv->rs + ez->max_t + 1;
|
r_inv->re = r_inv->rs + ez->max_t + 1;
|
||||||
mm_update_extra(r_inv, &qseq[q_off], &tseq[t_off], mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & MM_F_SR));
|
mm_update_extra(r_inv, &qseq[q_off], &tseq[t_off], mat, opt->q, opt->e, opt->flag & MM_F_EQX);
|
||||||
ret = 1;
|
ret = 1;
|
||||||
end_align1_inv:
|
end_align1_inv:
|
||||||
kfree(km, tseq);
|
kfree(km, tseq);
|
||||||
@@ -894,71 +910,6 @@ static inline mm_reg1_t *mm_insert_reg(const mm_reg1_t *r, int i, int *n_regs, m
|
|||||||
return regs;
|
return regs;
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline void mm_count_gaps(const mm_reg1_t *r, int32_t *n_gap_, int32_t *n_gapo_)
|
|
||||||
{
|
|
||||||
uint32_t i;
|
|
||||||
int32_t n_gapo = 0, n_gap = 0;
|
|
||||||
*n_gap_ = *n_gapo_ = -1;
|
|
||||||
if (r->p == 0) return;
|
|
||||||
for (i = 0; i < r->p->n_cigar; ++i) {
|
|
||||||
int32_t op = r->p->cigar[i] & 0xf, len = r->p->cigar[i] >> 4;
|
|
||||||
if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL)
|
|
||||||
++n_gapo, n_gap += len;
|
|
||||||
}
|
|
||||||
*n_gap_ = n_gap, *n_gapo_ = n_gapo;
|
|
||||||
}
|
|
||||||
|
|
||||||
double mm_event_identity(const mm_reg1_t *r)
|
|
||||||
{
|
|
||||||
int32_t n_gap, n_gapo;
|
|
||||||
if (r->p == 0) return -1.0f;
|
|
||||||
mm_count_gaps(r, &n_gap, &n_gapo);
|
|
||||||
return (double)r->mlen / (r->blen + r->p->n_ambi - n_gap + n_gapo);
|
|
||||||
}
|
|
||||||
|
|
||||||
static int32_t mm_recal_max_dp(const mm_reg1_t *r, double b2, int32_t match_sc)
|
|
||||||
{
|
|
||||||
uint32_t i;
|
|
||||||
int32_t n_gap = 0, n_gapo = 0, n_mis;
|
|
||||||
double gap_cost = 0.0;
|
|
||||||
if (r->p == 0) return -1;
|
|
||||||
for (i = 0; i < r->p->n_cigar; ++i) {
|
|
||||||
int32_t op = r->p->cigar[i] & 0xf, len = r->p->cigar[i] >> 4;
|
|
||||||
if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL) {
|
|
||||||
gap_cost += b2 + (double)mg_log2(1.0 + len);
|
|
||||||
++n_gapo, n_gap += len;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
n_mis = r->blen + r->p->n_ambi - r->mlen - n_gap;
|
|
||||||
return (int32_t)(match_sc * (r->mlen - b2 * n_mis - gap_cost) + .499);
|
|
||||||
}
|
|
||||||
|
|
||||||
void mm_update_dp_max(int qlen, int n_regs, mm_reg1_t *regs, float frac, int a, int b)
|
|
||||||
{
|
|
||||||
int32_t max = -1, max2 = -1, i, max_i = -1;
|
|
||||||
double div, b2;
|
|
||||||
if (n_regs < 2) return;
|
|
||||||
for (i = 0; i < n_regs; ++i) {
|
|
||||||
mm_reg1_t *r = ®s[i];
|
|
||||||
if (r->p == 0) continue;
|
|
||||||
if (r->p->dp_max > max) max2 = max, max = r->p->dp_max, max_i = i;
|
|
||||||
else if (r->p->dp_max > max2) max2 = r->p->dp_max;
|
|
||||||
}
|
|
||||||
if (max_i < 0 || max < 0 || max2 < 0) return;
|
|
||||||
if (regs[max_i].qe - regs[max_i].qs < (double)qlen * frac) return;
|
|
||||||
if (max2 < (double)max * frac) return;
|
|
||||||
div = 1. - mm_event_identity(®s[max_i]);
|
|
||||||
if (div < 0.02) div = 0.02;
|
|
||||||
b2 = 0.5 / div; // max value: 25
|
|
||||||
if (b2 * a < b) b2 = (double)a / b;
|
|
||||||
for (i = 0; i < n_regs; ++i) {
|
|
||||||
mm_reg1_t *r = ®s[i];
|
|
||||||
if (r->p == 0) continue;
|
|
||||||
r->p->dp_max = mm_recal_max_dp(r, b2, a);
|
|
||||||
if (r->p->dp_max < 0) r->p->dp_max = 0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a)
|
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a)
|
||||||
{
|
{
|
||||||
extern unsigned char seq_nt4_table[256];
|
extern unsigned char seq_nt4_table[256];
|
||||||
@@ -1002,7 +953,7 @@ mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
|||||||
regs[i].p->trans_strand = opt->flag&MM_F_SPLICE_FOR? 1 : 2;
|
regs[i].p->trans_strand = opt->flag&MM_F_SPLICE_FOR? 1 : 2;
|
||||||
}
|
}
|
||||||
if (r2.cnt > 0) regs = mm_insert_reg(&r2, i, &n_regs, regs);
|
if (r2.cnt > 0) regs = mm_insert_reg(&r2, i, &n_regs, regs);
|
||||||
if (i > 0 && regs[i].split_inv && !(opt->flag & MM_F_NO_INV)) {
|
if (i > 0 && regs[i].split_inv) {
|
||||||
if (mm_align1_inv(km, opt, mi, qlen, qseq0, ®s[i-1], ®s[i], &r2, &ez)) {
|
if (mm_align1_inv(km, opt, mi, qlen, qseq0, ®s[i-1], ®s[i], &r2, &ez)) {
|
||||||
regs = mm_insert_reg(&r2, i, &n_regs, regs);
|
regs = mm_insert_reg(&r2, i, &n_regs, regs);
|
||||||
++i; // skip the inserted INV alignment
|
++i; // skip the inserted INV alignment
|
||||||
@@ -1013,10 +964,6 @@ mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
|||||||
kfree(km, qseq0[0]);
|
kfree(km, qseq0[0]);
|
||||||
kfree(km, ez.cigar);
|
kfree(km, ez.cigar);
|
||||||
mm_filter_regs(opt, qlen, n_regs_, regs);
|
mm_filter_regs(opt, qlen, n_regs_, regs);
|
||||||
if (!(opt->flag&MM_F_SR) && !opt->split_prefix && qlen >= opt->rank_min_len) {
|
|
||||||
mm_update_dp_max(qlen, *n_regs_, regs, opt->rank_frac, opt->a, opt->b);
|
|
||||||
mm_filter_regs(opt, qlen, n_regs_, regs);
|
|
||||||
}
|
|
||||||
mm_hit_sort(km, n_regs_, regs, opt->alt_drop);
|
mm_hit_sort(km, n_regs_, regs, opt->alt_drop);
|
||||||
return regs;
|
return regs;
|
||||||
}
|
}
|
||||||
|
|||||||
Executable
+16
@@ -0,0 +1,16 @@
|
|||||||
|
ref_data=$1
|
||||||
|
preset=$2
|
||||||
|
|
||||||
|
make clean && make no_opt=1
|
||||||
|
touch temp_read.fastq
|
||||||
|
./minimap2 -ax $2 $1 temp_read.fastq -Z 1 >/dev/null
|
||||||
|
|
||||||
|
kv_file=$1"_"$2"_minimizers_key_value_sorted"
|
||||||
|
|
||||||
|
full_path=`readlink -f $kv_file`
|
||||||
|
|
||||||
|
cd ./ext/TAL
|
||||||
|
make lisa_hash
|
||||||
|
./build-lisa-hash-index $full_path
|
||||||
|
|
||||||
|
rm ../../temp_read.fastq
|
||||||
@@ -0,0 +1,265 @@
|
|||||||
|
/* The MIT License
|
||||||
|
|
||||||
|
Copyright (c) 2018- Dana-Farber Cancer Institute
|
||||||
|
2017-2018 Broad Institute, Inc.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
|
Modified Copyright (C) 2021 Intel Corporation
|
||||||
|
Contacts: Saurabh Kalikar <saurabh.kalikar@intel.com>;
|
||||||
|
Vasimuddin Md <vasimuddin.md@intel.com>; Sanchit Misra <sanchit.misra@intel.com>;
|
||||||
|
Chirag Jain <chirag@iisc.ac.in>; Heng Li <hli@jimmy.harvard.edu>
|
||||||
|
*/
|
||||||
|
#include <stdint.h>
|
||||||
|
#include <string.h>
|
||||||
|
#include <stdio.h>
|
||||||
|
#include "minimap.h"
|
||||||
|
#include "mmpriv.h"
|
||||||
|
#include "kalloc.h"
|
||||||
|
|
||||||
|
#if defined(VECTORIZED_CHAINING) && defined(__AVX512BW__)
|
||||||
|
#include "parallel_chaining_32_bit.h"
|
||||||
|
#endif
|
||||||
|
|
||||||
|
static const char LogTable256[256] = {
|
||||||
|
#define LT(n) n, n, n, n, n, n, n, n, n, n, n, n, n, n, n, n
|
||||||
|
-1, 0, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 3,
|
||||||
|
LT(4), LT(5), LT(5), LT(6), LT(6), LT(6), LT(6),
|
||||||
|
LT(7), LT(7), LT(7), LT(7), LT(7), LT(7), LT(7), LT(7)
|
||||||
|
};
|
||||||
|
|
||||||
|
static inline int ilog2_32(uint32_t v)
|
||||||
|
{
|
||||||
|
uint32_t t, tt;
|
||||||
|
if ((tt = v>>16)) return (t = tt>>8) ? 24 + LogTable256[t] : 16 + LogTable256[tt];
|
||||||
|
return (t = v>>8) ? 8 + LogTable256[t] : LogTable256[v];
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
mm128_t *mm_chain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float gap_scale, int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||||
|
{ // TODO: make sure this works when n has more than 32 bits
|
||||||
|
int32_t k, *p, *t, *v, n_u, n_v;
|
||||||
|
uint32_t *f;
|
||||||
|
int64_t i, j;
|
||||||
|
uint64_t *u, *u2;
|
||||||
|
mm128_t *b, *w;
|
||||||
|
|
||||||
|
if (_u) *_u = 0, *n_u_ = 0;
|
||||||
|
if (n == 0 || a == 0) {
|
||||||
|
kfree(km, a);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
f = (uint32_t*)kmalloc(km, n * 4);
|
||||||
|
p = (int32_t*)kmalloc(km, n * 4);
|
||||||
|
t = (int32_t*)kmalloc(km, n * 4);
|
||||||
|
v = (int32_t*)kmalloc(km, n * 4);
|
||||||
|
memset(t, 0, n * 4);
|
||||||
|
#if defined(VECTORIZED_CHAINING) && defined(__AVX512BW__)
|
||||||
|
/* Allocation for debugging
|
||||||
|
f_avx = (uint32_t*)kmalloc(km, n * 4);
|
||||||
|
p_avx = (int32_t*)kmalloc(km, n * 4);
|
||||||
|
*/
|
||||||
|
anchor_t* anchors = (anchor_t*)malloc(n* sizeof(anchor_t));
|
||||||
|
for (i = 0; i < n; ++i) {
|
||||||
|
uint64_t ri = a[i].x;
|
||||||
|
int32_t qi = (int32_t)a[i].y, q_span = a[i].y>>32&0xff; // NB: only 8 bits of span is used!!!
|
||||||
|
anchors[i].r = ri;
|
||||||
|
anchors[i].q = qi;
|
||||||
|
anchors[i].l = q_span;
|
||||||
|
}
|
||||||
|
num_bits_t *anchor_r, *anchor_q, *anchor_l;
|
||||||
|
create_SoA_Anchors_32_bit(anchors, n, anchor_r, anchor_q, anchor_l);
|
||||||
|
|
||||||
|
dp_chain obj(max_dist_x, max_dist_y, bw, max_skip, max_iter, gap_scale, is_cdna, n_segs);
|
||||||
|
obj.mm_dp_vectorized(n, &anchors[0], anchor_r, anchor_q, anchor_l, f, p, v, max_dist_x, max_dist_y, NULL, NULL);
|
||||||
|
|
||||||
|
// -16 is due to extra padding at the start of arrays
|
||||||
|
anchor_r -= 16; anchor_q -= 16; anchor_l -= 16;
|
||||||
|
free(anchor_r);
|
||||||
|
free(anchor_q);
|
||||||
|
free(anchor_l);
|
||||||
|
free(anchors);
|
||||||
|
#else
|
||||||
|
int64_t st = 0;
|
||||||
|
uint64_t sum_qspan = 0;
|
||||||
|
float avg_qspan;
|
||||||
|
for (i = 0; i < n; ++i) sum_qspan += a[i].y>>32&0xff;
|
||||||
|
avg_qspan = (float)sum_qspan / n;
|
||||||
|
|
||||||
|
// fill the score and backtrack arrays
|
||||||
|
for (i = 0; i < n; ++i) {
|
||||||
|
uint64_t ri = a[i].x;
|
||||||
|
int64_t max_j = -1;
|
||||||
|
int32_t qi = (int32_t)a[i].y, q_span = a[i].y>>32&0xff; // NB: only 8 bits of span is used!!!
|
||||||
|
int32_t max_f = q_span, n_skip = 0, min_d;
|
||||||
|
int32_t sidi = (a[i].y & MM_SEED_SEG_MASK) >> MM_SEED_SEG_SHIFT;
|
||||||
|
while (st < i && ri > a[st].x + max_dist_x) ++st;
|
||||||
|
if (i - st > max_iter) st = i - max_iter;
|
||||||
|
for (j = i - 1; j >= st; --j) {
|
||||||
|
int64_t dr = ri - a[j].x;
|
||||||
|
int32_t dq = qi - (int32_t)a[j].y, dd, sc, log_dd, gap_cost;
|
||||||
|
int32_t sidj = (a[j].y & MM_SEED_SEG_MASK) >> MM_SEED_SEG_SHIFT;
|
||||||
|
if ((sidi == sidj && dr == 0) || dq <= 0) continue; // don't skip if an anchor is used by multiple segments; see below
|
||||||
|
if ((sidi == sidj && dq > max_dist_y) || dq > max_dist_x) continue;
|
||||||
|
dd = dr > dq? dr - dq : dq - dr;
|
||||||
|
if (sidi == sidj && dd > bw) continue;
|
||||||
|
if (n_segs > 1 && !is_cdna && sidi == sidj && dr > max_dist_y) continue;
|
||||||
|
min_d = dq < dr? dq : dr;
|
||||||
|
sc = min_d > q_span? q_span : dq < dr? dq : dr;
|
||||||
|
log_dd = dd? ilog2_32(dd) : 0;
|
||||||
|
gap_cost = 0;
|
||||||
|
if (is_cdna || sidi != sidj) {
|
||||||
|
int c_log, c_lin;
|
||||||
|
c_lin = (int)(dd * .01 * avg_qspan);
|
||||||
|
c_log = log_dd;
|
||||||
|
if (sidi != sidj && dr == 0) ++sc; // possibly due to overlapping paired ends; give a minor bonus
|
||||||
|
else if (dr > dq || sidi != sidj) gap_cost = c_lin < c_log? c_lin : c_log;
|
||||||
|
else gap_cost = c_lin + (c_log>>1);
|
||||||
|
} else gap_cost = (int)(dd * .01 * avg_qspan) + (log_dd>>1);
|
||||||
|
sc -= (int)((double)gap_cost * gap_scale + .499);
|
||||||
|
sc += f[j];
|
||||||
|
if (sc > max_f) {
|
||||||
|
max_f = sc, max_j = j;
|
||||||
|
if (n_skip > 0) --n_skip;
|
||||||
|
} else if (t[j] == i) {
|
||||||
|
if (++n_skip > max_skip)
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (p[j] >= 0) t[p[j]] = i;
|
||||||
|
}
|
||||||
|
f[i] = max_f, p[i] = max_j;
|
||||||
|
v[i] = max_j >= 0 && v[max_j] > max_f? v[max_j] : max_f; // v[] keeps the peak score up to i; f[] is the score ending at i, not always the peak
|
||||||
|
}
|
||||||
|
|
||||||
|
#if 0
|
||||||
|
for (i = 0; i < n; ++i) {
|
||||||
|
assert(f[i] == f_avx[i] && p[i] == p_avx[i]);
|
||||||
|
|
||||||
|
//if(! (f[i] == f_avx[i] && p[i] == p_avx[i]))
|
||||||
|
{
|
||||||
|
#if 0
|
||||||
|
fprintf(stderr, "mm2-score:\n");
|
||||||
|
for (int itt = 0; itt < n; ++itt) {
|
||||||
|
fprintf(stderr, "%ld %ld \n", f[itt], p[itt]);
|
||||||
|
}
|
||||||
|
fprintf(stderr, "mm2-simd-score:\n");
|
||||||
|
for (int itt = 0; itt < n; ++itt) {
|
||||||
|
fprintf(stderr, "%ld %ld \n", f_avx[itt], p_avx[itt]);
|
||||||
|
}
|
||||||
|
fprintf(stderr, "anchors:\n");
|
||||||
|
fprintf(stderr, "%lld\n", n);
|
||||||
|
for (int itt = 0; itt < n; ++itt) {
|
||||||
|
uint64_t ri = a[itt].x;
|
||||||
|
int32_t qi = (int32_t)a[itt].y, q_span = a[itt].y>>32&0xff; // NB: only 8 bits of span is used!!!
|
||||||
|
fprintf(stderr, "%llu %ld %ld\n", ri, qi, q_span);
|
||||||
|
}
|
||||||
|
//exit(0);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
|
|
||||||
|
}
|
||||||
|
#if 0
|
||||||
|
fprintf(stderr, "%llu\n", n);
|
||||||
|
for (int itt = 0; itt < n; ++itt) {
|
||||||
|
uint64_t ri = a[itt].x;
|
||||||
|
int32_t qi = (int32_t)a[itt].y, q_span = a[itt].y>>32&0xff; // NB: only 8 bits of span is used!!!
|
||||||
|
fprintf(stderr, "%llu %ld %ld\n", ri, qi, q_span);
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif
|
||||||
|
kfree(km, f_avx); kfree(km, p_avx);
|
||||||
|
#endif
|
||||||
|
#endif
|
||||||
|
// find the ending positions of chains
|
||||||
|
memset(t, 0, n * 4);
|
||||||
|
for (i = 0; i < n; ++i)
|
||||||
|
if (p[i] >= 0) t[p[i]] = 1;
|
||||||
|
for (i = n_u = 0; i < n; ++i)
|
||||||
|
if (t[i] == 0 && v[i] >= min_sc)
|
||||||
|
++n_u;
|
||||||
|
if (n_u == 0) {
|
||||||
|
kfree(km, a); kfree(km, f); kfree(km, p); kfree(km, t); kfree(km, v);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
u = (uint64_t*)kmalloc(km, n_u * 8);
|
||||||
|
for (i = n_u = 0; i < n; ++i) {
|
||||||
|
if (t[i] == 0 && v[i] >= min_sc) {
|
||||||
|
j = i;
|
||||||
|
while (j >= 0 && f[j] < v[j]) j = p[j]; // find the peak that maximizes f[]
|
||||||
|
if (j < 0) j = i; // TODO: this should really be assert(j>=0)
|
||||||
|
u[n_u++] = (uint64_t)f[j] << 32 | j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
radix_sort_64(u, u + n_u);
|
||||||
|
for (i = 0; i < n_u>>1; ++i) { // reverse, s.t. the highest scoring chain is the first
|
||||||
|
uint64_t t = u[i];
|
||||||
|
u[i] = u[n_u - i - 1], u[n_u - i - 1] = t;
|
||||||
|
}
|
||||||
|
|
||||||
|
// backtrack
|
||||||
|
memset(t, 0, n * 4);
|
||||||
|
for (i = n_v = k = 0; i < n_u; ++i) { // starting from the highest score
|
||||||
|
int32_t n_v0 = n_v, k0 = k;
|
||||||
|
j = (int32_t)u[i];
|
||||||
|
do {
|
||||||
|
v[n_v++] = j;
|
||||||
|
t[j] = 1;
|
||||||
|
j = p[j];
|
||||||
|
} while (j >= 0 && t[j] == 0);
|
||||||
|
if (j < 0) {
|
||||||
|
if (n_v - n_v0 >= min_cnt) u[k++] = u[i]>>32<<32 | (n_v - n_v0);
|
||||||
|
} else if ((int32_t)(u[i]>>32) - f[j] >= min_sc) {
|
||||||
|
if (n_v - n_v0 >= min_cnt) u[k++] = ((u[i]>>32) - f[j]) << 32 | (n_v - n_v0);
|
||||||
|
}
|
||||||
|
if (k0 == k) n_v = n_v0; // no new chain added, reset
|
||||||
|
}
|
||||||
|
*n_u_ = n_u = k, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||||
|
|
||||||
|
// free temporary arrays
|
||||||
|
kfree(km, f); kfree(km, p); kfree(km, t);
|
||||||
|
|
||||||
|
// write the result to b[]
|
||||||
|
b = (mm128_t*)kmalloc(km, n_v * sizeof(mm128_t));
|
||||||
|
for (i = 0, k = 0; i < n_u; ++i) {
|
||||||
|
int32_t k0 = k, ni = (int32_t)u[i];
|
||||||
|
for (j = 0; j < ni; ++j)
|
||||||
|
b[k] = a[v[k0 + (ni - j - 1)]], ++k;
|
||||||
|
}
|
||||||
|
kfree(km, v);
|
||||||
|
|
||||||
|
// sort u[] and a[] by a[].x, such that adjacent chains may be joined (required by mm_join_long)
|
||||||
|
w = (mm128_t*)kmalloc(km, n_u * sizeof(mm128_t));
|
||||||
|
for (i = k = 0; i < n_u; ++i) {
|
||||||
|
w[i].x = b[k].x, w[i].y = (uint64_t)k<<32|i;
|
||||||
|
k += (int32_t)u[i];
|
||||||
|
}
|
||||||
|
radix_sort_128x(w, w + n_u);
|
||||||
|
u2 = (uint64_t*)kmalloc(km, n_u * 8);
|
||||||
|
for (i = k = 0; i < n_u; ++i) {
|
||||||
|
int32_t j = (int32_t)w[i].y, n = (int32_t)u[j];
|
||||||
|
u2[i] = u[j];
|
||||||
|
memcpy(&a[k], &b[w[i].y>>32], n * sizeof(mm128_t));
|
||||||
|
k += n;
|
||||||
|
}
|
||||||
|
if (n_u) memcpy(u, u2, n_u * 8);
|
||||||
|
if (k) memcpy(b, a, k * sizeof(mm128_t)); // write _a_ to _b_ and deallocate _a_ because _a_ is oversized, sometimes a lot
|
||||||
|
kfree(km, a); kfree(km, w); kfree(km, u2);
|
||||||
|
return b;
|
||||||
|
}
|
||||||
@@ -1,30 +0,0 @@
|
|||||||
## Contributor Code of Conduct
|
|
||||||
|
|
||||||
As contributors and maintainers of this project, we pledge to respect all
|
|
||||||
people who contribute through reporting issues, posting feature requests,
|
|
||||||
updating documentation, submitting pull requests or patches, and other
|
|
||||||
activities.
|
|
||||||
|
|
||||||
We are committed to making participation in this project a harassment-free
|
|
||||||
experience for everyone, regardless of level of experience, gender, gender
|
|
||||||
identity and expression, sexual orientation, disability, personal appearance,
|
|
||||||
body size, race, age, or religion.
|
|
||||||
|
|
||||||
Examples of unacceptable behavior by participants include the use of sexual
|
|
||||||
language or imagery, derogatory comments or personal attacks, trolling, public
|
|
||||||
or private harassment, insults, or other unprofessional conduct.
|
|
||||||
|
|
||||||
Project maintainers have the right and responsibility to remove, edit, or
|
|
||||||
reject comments, commits, code, wiki edits, issues, and other contributions
|
|
||||||
that are not aligned to this Code of Conduct. Project maintainers or
|
|
||||||
contributors who do not follow the Code of Conduct may be removed from the
|
|
||||||
project team.
|
|
||||||
|
|
||||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
|
||||||
reported by opening an issue or contacting the maintainer via email.
|
|
||||||
|
|
||||||
This Code of Conduct is adapted from the [Contributor Covenant][cc], [version
|
|
||||||
1.0.0][v1].
|
|
||||||
|
|
||||||
[cc]: http://contributor-covenant.org/
|
|
||||||
[v1]: http://contributor-covenant.org/version/1/0/0/
|
|
||||||
+6
-6
@@ -31,8 +31,8 @@ To acquire the data used in this cookbook and to install minimap2 and paftools,
|
|||||||
please follow the command lines below:
|
please follow the command lines below:
|
||||||
```sh
|
```sh
|
||||||
# install minimap2 executables
|
# install minimap2 executables
|
||||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.25/minimap2-2.25_x64-linux.tar.bz2 | tar jxf -
|
curl -L https://github.com/lh3/minimap2/releases/download/v2.18/minimap2-2.18_x64-linux.tar.bz2 | tar jxf -
|
||||||
cp minimap2-2.25_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
cp minimap2-2.18_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
||||||
export PATH="$PATH:"`pwd` # put the current directory on PATH
|
export PATH="$PATH:"`pwd` # put the current directory on PATH
|
||||||
# download example datasets
|
# download example datasets
|
||||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.10/cookbook-data.tgz | tar zxf -
|
curl -L https://github.com/lh3/minimap2/releases/download/v2.10/cookbook-data.tgz | tar zxf -
|
||||||
@@ -80,12 +80,12 @@ where a `U`-line gives the number of unmapped reads (for SAM input only); a
|
|||||||
5. Accumulative number of mappings
|
5. Accumulative number of mappings
|
||||||
|
|
||||||
For `paftools.js mapeval` to work, you need to encode the true read positions
|
For `paftools.js mapeval` to work, you need to encode the true read positions
|
||||||
in read names in the right format. For [pbsim2][pbsim] and [mason2][mason2], we
|
in read names in the right format. For [PBSIM][pbsim] and [mason2][mason2], we
|
||||||
provide scripts to generate the right format. Simulated reads in this cookbook
|
provide scripts to generate the right format. Simulated reads in this cookbook
|
||||||
were created with the following command lines:
|
were created with the following command lines:
|
||||||
```sh
|
```sh
|
||||||
# in the pbsim2 source code directory:
|
# in PBSIM source code directory:
|
||||||
src/pbsim --depth 1 --length-min 5000 --length-mean 20000 --accuracy-mean 0.95 --hmm_model data/R94.model ../ecoli_ref.fa
|
src/pbsim ../ecoli_ref.fa --depth 1 --sample-fastq sample/sample.fastq
|
||||||
paftools.js pbsim2fq ../ecoli_ref.fa.fai sd_0001.maf > ../ecoli_pbsim.fa
|
paftools.js pbsim2fq ../ecoli_ref.fa.fai sd_0001.maf > ../ecoli_pbsim.fa
|
||||||
|
|
||||||
# mason2 simulation
|
# mason2 simulation
|
||||||
@@ -237,7 +237,7 @@ with `-x ava-pb` (99% vs 93% with `-x ava-ont`).
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
[pbsim]: https://github.com/yukiteruono/pbsim2
|
[pbsim]: https://github.com/pfaucon/PBSIM-PacBio-Simulator
|
||||||
[mason2]: https://github.com/seqan/seqan/tree/master/apps/mason2
|
[mason2]: https://github.com/seqan/seqan/tree/master/apps/mason2
|
||||||
[paf]: https://github.com/lh3/miniasm/blob/master/PAF.md
|
[paf]: https://github.com/lh3/miniasm/blob/master/PAF.md
|
||||||
[v2.10]: https://github.com/lh3/minimap2/releases/tag/v2.10
|
[v2.10]: https://github.com/lh3/minimap2/releases/tag/v2.10
|
||||||
|
|||||||
@@ -47,7 +47,7 @@ int main(int argc, char *argv[])
|
|||||||
printf("%s\t%d\t%d\t%d\t%c\t", ks->name.s, ks->seq.l, r->qs, r->qe, "+-"[r->rev]);
|
printf("%s\t%d\t%d\t%d\t%c\t", ks->name.s, ks->seq.l, r->qs, r->qe, "+-"[r->rev]);
|
||||||
printf("%s\t%d\t%d\t%d\t%d\t%d\t%d\tcg:Z:", mi->seq[r->rid].name, mi->seq[r->rid].len, r->rs, r->re, r->mlen, r->blen, r->mapq);
|
printf("%s\t%d\t%d\t%d\t%d\t%d\t%d\tcg:Z:", mi->seq[r->rid].name, mi->seq[r->rid].len, r->rs, r->re, r->mlen, r->blen, r->mapq);
|
||||||
for (i = 0; i < r->p->n_cigar; ++i) // IMPORTANT: this gives the CIGAR in the aligned regions. NO soft/hard clippings!
|
for (i = 0; i < r->p->n_cigar; ++i) // IMPORTANT: this gives the CIGAR in the aligned regions. NO soft/hard clippings!
|
||||||
printf("%d%c", r->p->cigar[i]>>4, MM_CIGAR_STR[r->p->cigar[i]&0xf]);
|
printf("%d%c", r->p->cigar[i]>>4, "MIDNSH"[r->p->cigar[i]&0xf]);
|
||||||
putchar('\n');
|
putchar('\n');
|
||||||
free(r->p);
|
free(r->p);
|
||||||
}
|
}
|
||||||
|
|||||||
Submodule
+1
Submodule ext/TAL added at 6f82aa4c6a
@@ -119,7 +119,6 @@ int mm_write_sam_hdr(const mm_idx_t *idx, const char *rg, const char *ver, int a
|
|||||||
{
|
{
|
||||||
kstring_t str = {0,0,0};
|
kstring_t str = {0,0,0};
|
||||||
int ret = 0;
|
int ret = 0;
|
||||||
mm_sprintf_lite(&str, "@HD\tVN:1.6\tSO:unsorted\tGO:query\n");
|
|
||||||
if (idx) {
|
if (idx) {
|
||||||
uint32_t i;
|
uint32_t i;
|
||||||
for (i = 0; i < idx->n_seq; ++i)
|
for (i = 0; i < idx->n_seq; ++i)
|
||||||
@@ -145,8 +144,8 @@ static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
|||||||
if (write_tag) mm_sprintf_lite(s, "\tcs:Z:");
|
if (write_tag) mm_sprintf_lite(s, "\tcs:Z:");
|
||||||
for (i = q_off = t_off = 0; i < (int)r->p->n_cigar; ++i) {
|
for (i = q_off = t_off = 0; i < (int)r->p->n_cigar; ++i) {
|
||||||
int j, op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
int j, op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||||
assert((op >= MM_CIGAR_MATCH && op <= MM_CIGAR_N_SKIP) || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH);
|
assert((op >= 0 && op <= 3) || op == 7 || op == 8);
|
||||||
if (op == MM_CIGAR_MATCH || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH) {
|
if (op == 0 || op == 7 || op == 8) { // match
|
||||||
int l_tmp = 0;
|
int l_tmp = 0;
|
||||||
for (j = 0; j < len; ++j) {
|
for (j = 0; j < len; ++j) {
|
||||||
if (qseq[q_off + j] != tseq[t_off + j]) {
|
if (qseq[q_off + j] != tseq[t_off + j]) {
|
||||||
@@ -167,12 +166,12 @@ static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
|||||||
} else mm_sprintf_lite(s, ":%d", l_tmp);
|
} else mm_sprintf_lite(s, ":%d", l_tmp);
|
||||||
}
|
}
|
||||||
q_off += len, t_off += len;
|
q_off += len, t_off += len;
|
||||||
} else if (op == MM_CIGAR_INS) {
|
} else if (op == 1) { // insertion to ref
|
||||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||||
tmp[j] = "acgtn"[qseq[q_off + j]];
|
tmp[j] = "acgtn"[qseq[q_off + j]];
|
||||||
mm_sprintf_lite(s, "+%s", tmp);
|
mm_sprintf_lite(s, "+%s", tmp);
|
||||||
q_off += len;
|
q_off += len;
|
||||||
} else if (op == MM_CIGAR_DEL) {
|
} else if (op == 2) { // deletion from ref
|
||||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||||
tmp[j] = "acgtn"[tseq[t_off + j]];
|
tmp[j] = "acgtn"[tseq[t_off + j]];
|
||||||
mm_sprintf_lite(s, "-%s", tmp);
|
mm_sprintf_lite(s, "-%s", tmp);
|
||||||
@@ -193,8 +192,8 @@ static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
|||||||
if (write_tag) mm_sprintf_lite(s, "\tMD:Z:");
|
if (write_tag) mm_sprintf_lite(s, "\tMD:Z:");
|
||||||
for (i = q_off = t_off = 0; i < (int)r->p->n_cigar; ++i) {
|
for (i = q_off = t_off = 0; i < (int)r->p->n_cigar; ++i) {
|
||||||
int j, op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
int j, op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||||
assert((op >= MM_CIGAR_MATCH && op <= MM_CIGAR_N_SKIP) || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH);
|
assert((op >= 0 && op <= 3) || op == 7 || op == 8);
|
||||||
if (op == MM_CIGAR_MATCH || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH) {
|
if (op == 0 || op == 7 || op == 8) { // match
|
||||||
for (j = 0; j < len; ++j) {
|
for (j = 0; j < len; ++j) {
|
||||||
if (qseq[q_off + j] != tseq[t_off + j]) {
|
if (qseq[q_off + j] != tseq[t_off + j]) {
|
||||||
mm_sprintf_lite(s, "%d%c", l_MD, "ACGTN"[tseq[t_off + j]]);
|
mm_sprintf_lite(s, "%d%c", l_MD, "ACGTN"[tseq[t_off + j]]);
|
||||||
@@ -202,15 +201,15 @@ static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
|||||||
} else ++l_MD;
|
} else ++l_MD;
|
||||||
}
|
}
|
||||||
q_off += len, t_off += len;
|
q_off += len, t_off += len;
|
||||||
} else if (op == MM_CIGAR_INS) {
|
} else if (op == 1) { // insertion to ref
|
||||||
q_off += len;
|
q_off += len;
|
||||||
} else if (op == MM_CIGAR_DEL) {
|
} else if (op == 2) { // deletion from ref
|
||||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||||
tmp[j] = "ACGTN"[tseq[t_off + j]];
|
tmp[j] = "ACGTN"[tseq[t_off + j]];
|
||||||
mm_sprintf_lite(s, "%d^%s", l_MD, tmp);
|
mm_sprintf_lite(s, "%d^%s", l_MD, tmp);
|
||||||
l_MD = 0;
|
l_MD = 0;
|
||||||
t_off += len;
|
t_off += len;
|
||||||
} else if (op == MM_CIGAR_N_SKIP) {
|
} else if (op == 3) { // reference skip
|
||||||
t_off += len;
|
t_off += len;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -218,7 +217,7 @@ static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
|||||||
assert(t_off == r->re - r->rs && q_off == r->qe - r->qs);
|
assert(t_off == r->re - r->rs && q_off == r->qe - r->qs);
|
||||||
}
|
}
|
||||||
|
|
||||||
static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int no_iden, int is_MD, int write_tag, int is_qstrand)
|
static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int no_iden, int is_MD, int write_tag)
|
||||||
{
|
{
|
||||||
extern unsigned char seq_nt4_table[256];
|
extern unsigned char seq_nt4_table[256];
|
||||||
int i;
|
int i;
|
||||||
@@ -228,20 +227,14 @@ static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_
|
|||||||
qseq = (uint8_t*)kmalloc(km, r->qe - r->qs);
|
qseq = (uint8_t*)kmalloc(km, r->qe - r->qs);
|
||||||
tseq = (uint8_t*)kmalloc(km, r->re - r->rs);
|
tseq = (uint8_t*)kmalloc(km, r->re - r->rs);
|
||||||
tmp = (char*)kmalloc(km, r->re - r->rs > r->qe - r->qs? r->re - r->rs + 1 : r->qe - r->qs + 1);
|
tmp = (char*)kmalloc(km, r->re - r->rs > r->qe - r->qs? r->re - r->rs + 1 : r->qe - r->qs + 1);
|
||||||
if (is_qstrand) {
|
mm_idx_getseq(mi, r->rid, r->rs, r->re, tseq);
|
||||||
mm_idx_getseq2(mi, r->rev, r->rid, r->rs, r->re, tseq);
|
if (!r->rev) {
|
||||||
for (i = r->qs; i < r->qe; ++i)
|
for (i = r->qs; i < r->qe; ++i)
|
||||||
qseq[i - r->qs] = seq_nt4_table[(uint8_t)t->seq[i]];
|
qseq[i - r->qs] = seq_nt4_table[(uint8_t)t->seq[i]];
|
||||||
} else {
|
} else {
|
||||||
mm_idx_getseq(mi, r->rid, r->rs, r->re, tseq);
|
for (i = r->qs; i < r->qe; ++i) {
|
||||||
if (!r->rev) {
|
uint8_t c = seq_nt4_table[(uint8_t)t->seq[i]];
|
||||||
for (i = r->qs; i < r->qe; ++i)
|
qseq[r->qe - i - 1] = c >= 4? 4 : 3 - c;
|
||||||
qseq[i - r->qs] = seq_nt4_table[(uint8_t)t->seq[i]];
|
|
||||||
} else {
|
|
||||||
for (i = r->qs; i < r->qe; ++i) {
|
|
||||||
uint8_t c = seq_nt4_table[(uint8_t)t->seq[i]];
|
|
||||||
qseq[r->qe - i - 1] = c >= 4? 4 : 3 - c;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (is_MD) write_MD_core(s, tseq, qseq, r, tmp, write_tag);
|
if (is_MD) write_MD_core(s, tseq, qseq, r, tmp, write_tag);
|
||||||
@@ -249,14 +242,14 @@ static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_
|
|||||||
kfree(km, qseq); kfree(km, tseq); kfree(km, tmp);
|
kfree(km, qseq); kfree(km, tseq); kfree(km, tmp);
|
||||||
}
|
}
|
||||||
|
|
||||||
int mm_gen_cs_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int is_MD, int no_iden, int is_qstrand)
|
int mm_gen_cs_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int is_MD, int no_iden)
|
||||||
{
|
{
|
||||||
mm_bseq1_t t;
|
mm_bseq1_t t;
|
||||||
kstring_t str;
|
kstring_t str;
|
||||||
str.s = *buf, str.l = 0, str.m = *max_len;
|
str.s = *buf, str.l = 0, str.m = *max_len;
|
||||||
t.l_seq = strlen(seq);
|
t.l_seq = strlen(seq);
|
||||||
t.seq = (char*)seq;
|
t.seq = (char*)seq;
|
||||||
write_cs_or_MD(km, &str, mi, &t, r, no_iden, is_MD, 0, is_qstrand);
|
write_cs_or_MD(km, &str, mi, &t, r, no_iden, is_MD, 0);
|
||||||
*max_len = str.m;
|
*max_len = str.m;
|
||||||
*buf = str.s;
|
*buf = str.s;
|
||||||
return str.l;
|
return str.l;
|
||||||
@@ -264,12 +257,24 @@ int mm_gen_cs_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, cons
|
|||||||
|
|
||||||
int mm_gen_cs(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden)
|
int mm_gen_cs(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden)
|
||||||
{
|
{
|
||||||
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 0, no_iden, 0);
|
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 0, no_iden);
|
||||||
}
|
}
|
||||||
|
|
||||||
int mm_gen_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq)
|
int mm_gen_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq)
|
||||||
{
|
{
|
||||||
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 1, 0, 0);
|
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 1, 0);
|
||||||
|
}
|
||||||
|
|
||||||
|
double mm_event_identity(const mm_reg1_t *r)
|
||||||
|
{
|
||||||
|
int32_t i, n_gapo = 0, n_gap = 0;
|
||||||
|
if (r->p == 0) return -1.0f;
|
||||||
|
for (i = 0; i < r->p->n_cigar; ++i) {
|
||||||
|
int32_t op = r->p->cigar[i] & 0xf, len = r->p->cigar[i] >> 4;
|
||||||
|
if (op == 1 || op == 2)
|
||||||
|
++n_gapo, n_gap += len;
|
||||||
|
}
|
||||||
|
return (double)r->mlen / (r->blen + r->p->n_ambi - n_gap + n_gapo);
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline void write_tags(kstring_t *s, const mm_reg1_t *r)
|
static inline void write_tags(kstring_t *s, const mm_reg1_t *r)
|
||||||
@@ -300,7 +305,7 @@ static inline void write_tags(kstring_t *s, const mm_reg1_t *r)
|
|||||||
if (r->split) mm_sprintf_lite(s, "\tzd:i:%d", r->split);
|
if (r->split) mm_sprintf_lite(s, "\tzd:i:%d", r->split);
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len)
|
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int opt_flag, int rep_len)
|
||||||
{
|
{
|
||||||
s->l = 0;
|
s->l = 0;
|
||||||
if (r == 0) {
|
if (r == 0) {
|
||||||
@@ -311,11 +316,7 @@ void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const
|
|||||||
mm_sprintf_lite(s, "%s\t%d\t%d\t%d\t%c\t", t->name, t->l_seq, r->qs, r->qe, "+-"[r->rev]);
|
mm_sprintf_lite(s, "%s\t%d\t%d\t%d\t%c\t", t->name, t->l_seq, r->qs, r->qe, "+-"[r->rev]);
|
||||||
if (mi->seq[r->rid].name) mm_sprintf_lite(s, "%s", mi->seq[r->rid].name);
|
if (mi->seq[r->rid].name) mm_sprintf_lite(s, "%s", mi->seq[r->rid].name);
|
||||||
else mm_sprintf_lite(s, "%d", r->rid);
|
else mm_sprintf_lite(s, "%d", r->rid);
|
||||||
mm_sprintf_lite(s, "\t%d", mi->seq[r->rid].len);
|
mm_sprintf_lite(s, "\t%d\t%d\t%d", mi->seq[r->rid].len, r->rs, r->re);
|
||||||
if ((opt_flag & MM_F_QSTRAND) && r->rev)
|
|
||||||
mm_sprintf_lite(s, "\t%d\t%d", mi->seq[r->rid].len - r->re, mi->seq[r->rid].len - r->rs);
|
|
||||||
else
|
|
||||||
mm_sprintf_lite(s, "\t%d\t%d", r->rs, r->re);
|
|
||||||
mm_sprintf_lite(s, "\t%d\t%d", r->mlen, r->blen);
|
mm_sprintf_lite(s, "\t%d\t%d", r->mlen, r->blen);
|
||||||
mm_sprintf_lite(s, "\t%d", r->mapq);
|
mm_sprintf_lite(s, "\t%d", r->mapq);
|
||||||
write_tags(s, r);
|
write_tags(s, r);
|
||||||
@@ -324,15 +325,15 @@ void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const
|
|||||||
uint32_t k;
|
uint32_t k;
|
||||||
mm_sprintf_lite(s, "\tcg:Z:");
|
mm_sprintf_lite(s, "\tcg:Z:");
|
||||||
for (k = 0; k < r->p->n_cigar; ++k)
|
for (k = 0; k < r->p->n_cigar; ++k)
|
||||||
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, MM_CIGAR_STR[r->p->cigar[k]&0xf]);
|
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, "MIDNSHP=XB"[r->p->cigar[k]&0xf]);
|
||||||
}
|
}
|
||||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
||||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1, !!(opt_flag&MM_F_QSTRAND));
|
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1);
|
||||||
if ((opt_flag & MM_F_COPY_COMMENT) && t->comment)
|
if ((opt_flag & MM_F_COPY_COMMENT) && t->comment)
|
||||||
mm_sprintf_lite(s, "\t%s", t->comment);
|
mm_sprintf_lite(s, "\t%s", t->comment);
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag)
|
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int opt_flag)
|
||||||
{
|
{
|
||||||
mm_write_paf3(s, mi, t, r, km, opt_flag, -1);
|
mm_write_paf3(s, mi, t, r, km, opt_flag, -1);
|
||||||
}
|
}
|
||||||
@@ -361,7 +362,7 @@ static inline const mm_reg1_t *get_sam_pri(int n_regs, const mm_reg1_t *regs)
|
|||||||
return NULL;
|
return NULL;
|
||||||
}
|
}
|
||||||
|
|
||||||
static void write_sam_cigar(kstring_t *s, int sam_flag, int in_tag, int qlen, const mm_reg1_t *r, int64_t opt_flag)
|
static void write_sam_cigar(kstring_t *s, int sam_flag, int in_tag, int qlen, const mm_reg1_t *r, int opt_flag)
|
||||||
{
|
{
|
||||||
if (r->p == 0) {
|
if (r->p == 0) {
|
||||||
mm_sprintf_lite(s, "*");
|
mm_sprintf_lite(s, "*");
|
||||||
@@ -370,26 +371,24 @@ static void write_sam_cigar(kstring_t *s, int sam_flag, int in_tag, int qlen, co
|
|||||||
clip_len[0] = r->rev? qlen - r->qe : r->qs;
|
clip_len[0] = r->rev? qlen - r->qe : r->qs;
|
||||||
clip_len[1] = r->rev? r->qs : qlen - r->qe;
|
clip_len[1] = r->rev? r->qs : qlen - r->qe;
|
||||||
if (in_tag) {
|
if (in_tag) {
|
||||||
int clip_char = (((sam_flag&0x800) || ((sam_flag&0x100) && (opt_flag&MM_F_SECONDARY_SEQ))) &&
|
int clip_char = (sam_flag&0x800) && !(opt_flag&MM_F_SOFTCLIP)? 5 : 4;
|
||||||
!(opt_flag&MM_F_SOFTCLIP)) ? 5 : 4;
|
|
||||||
mm_sprintf_lite(s, "\tCG:B:I");
|
mm_sprintf_lite(s, "\tCG:B:I");
|
||||||
if (clip_len[0]) mm_sprintf_lite(s, ",%u", clip_len[0]<<4|clip_char);
|
if (clip_len[0]) mm_sprintf_lite(s, ",%u", clip_len[0]<<4|clip_char);
|
||||||
for (k = 0; k < r->p->n_cigar; ++k)
|
for (k = 0; k < r->p->n_cigar; ++k)
|
||||||
mm_sprintf_lite(s, ",%u", r->p->cigar[k]);
|
mm_sprintf_lite(s, ",%u", r->p->cigar[k]);
|
||||||
if (clip_len[1]) mm_sprintf_lite(s, ",%u", clip_len[1]<<4|clip_char);
|
if (clip_len[1]) mm_sprintf_lite(s, ",%u", clip_len[1]<<4|clip_char);
|
||||||
} else {
|
} else {
|
||||||
int clip_char = (((sam_flag&0x800) || ((sam_flag&0x100) && (opt_flag&MM_F_SECONDARY_SEQ))) &&
|
int clip_char = (sam_flag&0x800) && !(opt_flag&MM_F_SOFTCLIP)? 'H' : 'S';
|
||||||
!(opt_flag&MM_F_SOFTCLIP)) ? 'H' : 'S';
|
|
||||||
assert(clip_len[0] < qlen && clip_len[1] < qlen);
|
assert(clip_len[0] < qlen && clip_len[1] < qlen);
|
||||||
if (clip_len[0]) mm_sprintf_lite(s, "%d%c", clip_len[0], clip_char);
|
if (clip_len[0]) mm_sprintf_lite(s, "%d%c", clip_len[0], clip_char);
|
||||||
for (k = 0; k < r->p->n_cigar; ++k)
|
for (k = 0; k < r->p->n_cigar; ++k)
|
||||||
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, MM_CIGAR_STR[r->p->cigar[k]&0xf]);
|
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, "MIDNSHP=XB"[r->p->cigar[k]&0xf]);
|
||||||
if (clip_len[1]) mm_sprintf_lite(s, "%d%c", clip_len[1], clip_char);
|
if (clip_len[1]) mm_sprintf_lite(s, "%d%c", clip_len[1], clip_char);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int64_t opt_flag, int rep_len)
|
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int opt_flag, int rep_len)
|
||||||
{
|
{
|
||||||
const int max_bam_cigar_op = 65535;
|
const int max_bam_cigar_op = 65535;
|
||||||
int flag, n_regs = n_regss[seg_idx], cigar_in_tag = 0;
|
int flag, n_regs = n_regss[seg_idx], cigar_in_tag = 0;
|
||||||
@@ -454,7 +453,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
|||||||
if (cigar_in_tag) {
|
if (cigar_in_tag) {
|
||||||
int slen;
|
int slen;
|
||||||
if ((flag & 0x900) == 0 || (opt_flag & MM_F_SOFTCLIP)) slen = t->l_seq;
|
if ((flag & 0x900) == 0 || (opt_flag & MM_F_SOFTCLIP)) slen = t->l_seq;
|
||||||
else if ((flag & 0x100) && !(opt_flag & MM_F_SECONDARY_SEQ)) slen = 0;
|
else if (flag & 0x100) slen = 0;
|
||||||
else slen = r->qe - r->qs;
|
else slen = r->qe - r->qs;
|
||||||
mm_sprintf_lite(s, "%dS%dN", slen, r->re - r->rs);
|
mm_sprintf_lite(s, "%dS%dN", slen, r->re - r->rs);
|
||||||
} else write_sam_cigar(s, flag, 0, t->l_seq, r, opt_flag);
|
} else write_sam_cigar(s, flag, 0, t->l_seq, r, opt_flag);
|
||||||
@@ -495,7 +494,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
|||||||
mm_sprintf_lite(s, "\t");
|
mm_sprintf_lite(s, "\t");
|
||||||
if (t->qual) sam_write_sq(s, t->qual, t->l_seq, r->rev, 0);
|
if (t->qual) sam_write_sq(s, t->qual, t->l_seq, r->rev, 0);
|
||||||
else mm_sprintf_lite(s, "*");
|
else mm_sprintf_lite(s, "*");
|
||||||
} else if ((flag & 0x100) && !(opt_flag & MM_F_SECONDARY_SEQ)){
|
} else if (flag & 0x100) {
|
||||||
mm_sprintf_lite(s, "*\t*");
|
mm_sprintf_lite(s, "*\t*");
|
||||||
} else {
|
} else {
|
||||||
sam_write_sq(s, t->seq + r->qs, r->qe - r->qs, r->rev, r->rev);
|
sam_write_sq(s, t->seq + r->qs, r->qe - r->qs, r->rev, r->rev);
|
||||||
@@ -536,7 +535,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
||||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1, 0);
|
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1);
|
||||||
if (cigar_in_tag)
|
if (cigar_in_tag)
|
||||||
write_sam_cigar(s, flag, 1, t->l_seq, r, opt_flag);
|
write_sam_cigar(s, flag, 1, t->l_seq, r, opt_flag);
|
||||||
}
|
}
|
||||||
@@ -548,7 +547,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
|||||||
s->s[s->l] = 0; // we always have room for an extra byte (see str_enlarge)
|
s->s[s->l] = 0; // we always have room for an extra byte (see str_enlarge)
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int64_t opt_flag)
|
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int opt_flag)
|
||||||
{
|
{
|
||||||
mm_write_sam3(s, mi, t, seg_idx, reg_idx, n_seg, n_regss, regss, km, opt_flag, -1);
|
mm_write_sam3(s, mi, t, seg_idx, reg_idx, n_seg, n_regss, regss, km, opt_flag, -1);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -20,14 +20,14 @@ static inline void mm_cal_fuzzy_len(mm_reg1_t *r, const mm128_t *a)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline void mm_reg_set_coor(mm_reg1_t *r, int32_t qlen, const mm128_t *a, int is_qstrand)
|
static inline void mm_reg_set_coor(mm_reg1_t *r, int32_t qlen, const mm128_t *a)
|
||||||
{ // NB: r->as and r->cnt MUST BE set correctly for this function to work
|
{ // NB: r->as and r->cnt MUST BE set correctly for this function to work
|
||||||
int32_t k = r->as, q_span = (int32_t)(a[k].y>>32&0xff);
|
int32_t k = r->as, q_span = (int32_t)(a[k].y>>32&0xff);
|
||||||
r->rev = a[k].x>>63;
|
r->rev = a[k].x>>63;
|
||||||
r->rid = a[k].x<<1>>33;
|
r->rid = a[k].x<<1>>33;
|
||||||
r->rs = (int32_t)a[k].x + 1 > q_span? (int32_t)a[k].x + 1 - q_span : 0; // NB: target span may be shorter, so this test is necessary
|
r->rs = (int32_t)a[k].x + 1 > q_span? (int32_t)a[k].x + 1 - q_span : 0; // NB: target span may be shorter, so this test is necessary
|
||||||
r->re = (int32_t)a[k + r->cnt - 1].x + 1;
|
r->re = (int32_t)a[k + r->cnt - 1].x + 1;
|
||||||
if (!r->rev || is_qstrand) {
|
if (!r->rev) {
|
||||||
r->qs = (int32_t)a[k].y + 1 - q_span;
|
r->qs = (int32_t)a[k].y + 1 - q_span;
|
||||||
r->qe = (int32_t)a[k + r->cnt - 1].y + 1;
|
r->qe = (int32_t)a[k + r->cnt - 1].y + 1;
|
||||||
} else {
|
} else {
|
||||||
@@ -49,7 +49,7 @@ static inline uint64_t hash64(uint64_t key)
|
|||||||
return key;
|
return key;
|
||||||
}
|
}
|
||||||
|
|
||||||
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a, int is_qstrand) // convert chains to hits
|
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a) // convert chains to hits
|
||||||
{
|
{
|
||||||
mm128_t *z, tmp;
|
mm128_t *z, tmp;
|
||||||
mm_reg1_t *r;
|
mm_reg1_t *r;
|
||||||
@@ -81,7 +81,7 @@ mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u,
|
|||||||
ri->cnt = (int32_t)z[i].y;
|
ri->cnt = (int32_t)z[i].y;
|
||||||
ri->as = z[i].y >> 32;
|
ri->as = z[i].y >> 32;
|
||||||
ri->div = -1.0f;
|
ri->div = -1.0f;
|
||||||
mm_reg_set_coor(ri, qlen, a, is_qstrand);
|
mm_reg_set_coor(ri, qlen, a);
|
||||||
}
|
}
|
||||||
kfree(km, z);
|
kfree(km, z);
|
||||||
return r;
|
return r;
|
||||||
@@ -103,7 +103,7 @@ static inline int mm_alt_score(int score, float alt_diff_frac)
|
|||||||
return score > 0? score : 1;
|
return score > 0? score : 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a, int is_qstrand)
|
void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a)
|
||||||
{
|
{
|
||||||
if (n <= 0 || n >= r->cnt) return;
|
if (n <= 0 || n >= r->cnt) return;
|
||||||
*r2 = *r;
|
*r2 = *r;
|
||||||
@@ -115,10 +115,10 @@ void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a, int
|
|||||||
r2->score = (int32_t)(r->score * ((float)r2->cnt / r->cnt) + .499);
|
r2->score = (int32_t)(r->score * ((float)r2->cnt / r->cnt) + .499);
|
||||||
r2->as = r->as + n;
|
r2->as = r->as + n;
|
||||||
if (r->parent == r->id) r2->parent = MM_PARENT_TMP_PRI;
|
if (r->parent == r->id) r2->parent = MM_PARENT_TMP_PRI;
|
||||||
mm_reg_set_coor(r2, qlen, a, is_qstrand);
|
mm_reg_set_coor(r2, qlen, a);
|
||||||
r->cnt -= r2->cnt;
|
r->cnt -= r2->cnt;
|
||||||
r->score -= r2->score;
|
r->score -= r2->score;
|
||||||
mm_reg_set_coor(r, qlen, a, is_qstrand);
|
mm_reg_set_coor(r, qlen, a);
|
||||||
r->split |= 1, r2->split |= 2;
|
r->split |= 1, r2->split |= 2;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -252,7 +252,7 @@ void mm_sync_regs(void *km, int n_regs, mm_reg1_t *regs) // keep mm_reg1_t::{id,
|
|||||||
mm_set_sam_pri(n_regs, regs);
|
mm_set_sam_pri(n_regs, regs);
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int check_strand, int min_strand_sc, int *n_, mm_reg1_t *r)
|
void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int *n_, mm_reg1_t *r)
|
||||||
{
|
{
|
||||||
if (pri_ratio > 0.0f && *n_ > 0) {
|
if (pri_ratio > 0.0f && *n_ > 0) {
|
||||||
int i, k, n = *n_, n_2nd = 0;
|
int i, k, n = *n_, n_2nd = 0;
|
||||||
@@ -264,9 +264,6 @@ void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int chec
|
|||||||
if (!(r[i].qs == r[p].qs && r[i].qe == r[p].qe && r[i].rid == r[p].rid && r[i].rs == r[p].rs && r[i].re == r[p].re)) // not identical hits
|
if (!(r[i].qs == r[p].qs && r[i].qe == r[p].qe && r[i].rid == r[p].rid && r[i].rs == r[p].rs && r[i].re == r[p].re)) // not identical hits
|
||||||
r[k++] = r[i], ++n_2nd;
|
r[k++] = r[i], ++n_2nd;
|
||||||
else if (r[i].p) free(r[i].p);
|
else if (r[i].p) free(r[i].p);
|
||||||
} else if (check_strand && n_2nd < best_n && r[i].score > min_strand_sc && r[i].rev != r[p].rev) {
|
|
||||||
r[i].strand_retained = 1;
|
|
||||||
r[k++] = r[i], ++n_2nd;
|
|
||||||
} else if (r[i].p) free(r[i].p);
|
} else if (r[i].p) free(r[i].p);
|
||||||
}
|
}
|
||||||
if (k != n) mm_sync_regs(km, k, r); // removing hits requires sync()
|
if (k != n) mm_sync_regs(km, k, r); // removing hits requires sync()
|
||||||
@@ -274,19 +271,6 @@ void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int chec
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
int mm_filter_strand_retained(int n_regs, mm_reg1_t *r)
|
|
||||||
{
|
|
||||||
int i, k;
|
|
||||||
for (i = k = 0; i < n_regs; ++i) {
|
|
||||||
int p = r[i].parent;
|
|
||||||
if (!r[i].strand_retained || r[i].div < r[p].div * 5.0f || r[i].div < 0.01f) {
|
|
||||||
if (k < i) r[k++] = r[i];
|
|
||||||
else ++k;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return k;
|
|
||||||
}
|
|
||||||
|
|
||||||
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs)
|
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs)
|
||||||
{ // NB: after this call, mm_reg1_t::parent can be -1 if its parent filtered out
|
{ // NB: after this call, mm_reg1_t::parent can be -1 if its parent filtered out
|
||||||
int i, k;
|
int i, k;
|
||||||
@@ -328,6 +312,64 @@ int mm_squeeze_a(void *km, int n_regs, mm_reg1_t *regs, mm128_t *a)
|
|||||||
return as;
|
return as;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void mm_join_long(void *km, const mm_mapopt_t *opt, int qlen, int *n_regs_, mm_reg1_t *regs, mm128_t *a)
|
||||||
|
{
|
||||||
|
int i, n_aux, n_regs = *n_regs_, n_drop = 0;
|
||||||
|
uint64_t *aux;
|
||||||
|
|
||||||
|
if (n_regs < 2) return; // nothing to join
|
||||||
|
mm_squeeze_a(km, n_regs, regs, a);
|
||||||
|
|
||||||
|
aux = (uint64_t*)kmalloc(km, n_regs * 8);
|
||||||
|
for (i = n_aux = 0; i < n_regs; ++i)
|
||||||
|
if (regs[i].parent == i || regs[i].parent < 0)
|
||||||
|
aux[n_aux++] = (uint64_t)regs[i].as << 32 | i;
|
||||||
|
radix_sort_64(aux, aux + n_aux);
|
||||||
|
|
||||||
|
for (i = n_aux - 1; i >= 1; --i) {
|
||||||
|
mm_reg1_t *r0 = ®s[(int32_t)aux[i-1]], *r1 = ®s[(int32_t)aux[i]];
|
||||||
|
mm128_t *a0e, *a1s;
|
||||||
|
int max_gap, min_gap, sc_thres, min_flank_len;
|
||||||
|
|
||||||
|
// test
|
||||||
|
if (r0->as + r0->cnt != r1->as) continue; // not adjacent in a[]
|
||||||
|
if (r0->rid != r1->rid || r0->rev != r1->rev) continue; // make sure on the same target and strand
|
||||||
|
a0e = &a[r0->as + r0->cnt - 1];
|
||||||
|
a1s = &a[r1->as];
|
||||||
|
if (a1s->x <= a0e->x || (int32_t)a1s->y <= (int32_t)a0e->y) continue; // keep colinearity
|
||||||
|
max_gap = min_gap = (int32_t)a1s->y - (int32_t)a0e->y;
|
||||||
|
max_gap = a0e->x + max_gap > a1s->x? max_gap : a1s->x - a0e->x;
|
||||||
|
min_gap = a0e->x + min_gap < a1s->x? min_gap : a1s->x - a0e->x;
|
||||||
|
if (max_gap > opt->max_join_long || min_gap > opt->max_join_short) continue;
|
||||||
|
sc_thres = (int)((float)opt->min_join_flank_sc / opt->max_join_long * max_gap + .499);
|
||||||
|
if (r0->score < sc_thres || r1->score < sc_thres) continue; // require good flanking chains
|
||||||
|
min_flank_len = (int)(max_gap * opt->min_join_flank_ratio);
|
||||||
|
if (r0->re - r0->rs < min_flank_len || r0->qe - r0->qs < min_flank_len) continue; // require enough flanking length
|
||||||
|
if (r1->re - r1->rs < min_flank_len || r1->qe - r1->qs < min_flank_len) continue;
|
||||||
|
|
||||||
|
// all conditions satisfied; join
|
||||||
|
a[r1->as].y |= MM_SEED_LONG_JOIN;
|
||||||
|
r0->cnt += r1->cnt, r0->score += r1->score;
|
||||||
|
mm_reg_set_coor(r0, qlen, a);
|
||||||
|
r1->cnt = 0;
|
||||||
|
r1->parent = r0->id;
|
||||||
|
++n_drop;
|
||||||
|
}
|
||||||
|
kfree(km, aux);
|
||||||
|
|
||||||
|
if (n_drop > 0) { // then fix the hits hierarchy
|
||||||
|
for (i = 0; i < n_regs; ++i) { // adjust the mm_reg1_t::parent
|
||||||
|
mm_reg1_t *r = ®s[i];
|
||||||
|
if (r->parent >= 0 && r->id != r->parent) { // fix for secondary hits only
|
||||||
|
if (regs[r->parent].parent >= 0 && regs[r->parent].parent != r->parent)
|
||||||
|
r->parent = regs[r->parent].parent;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
mm_filter_regs(opt, qlen, n_regs_, regs);
|
||||||
|
mm_sync_regs(km, *n_regs_, regs);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
mm_seg_t *mm_seg_gen(void *km, uint32_t hash, int n_segs, const int *qlens, int n_regs0, const mm_reg1_t *regs0, int *n_regs, mm_reg1_t **regs, const mm128_t *a)
|
mm_seg_t *mm_seg_gen(void *km, uint32_t hash, int n_segs, const int *qlens, int n_regs0, const mm_reg1_t *regs0, int *n_regs, mm_reg1_t **regs, const mm128_t *a)
|
||||||
{
|
{
|
||||||
int s, i, j, acc_qlen[MM_MAX_SEG+1], qlen_sum = 0;
|
int s, i, j, acc_qlen[MM_MAX_SEG+1], qlen_sum = 0;
|
||||||
@@ -374,7 +416,7 @@ mm_seg_t *mm_seg_gen(void *km, uint32_t hash, int n_segs, const int *qlens, int
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
for (s = 0; s < n_segs; ++s) {
|
for (s = 0; s < n_segs; ++s) {
|
||||||
regs[s] = mm_gen_regs(km, hash, qlens[s], seg[s].n_u, seg[s].u, seg[s].a, 0);
|
regs[s] = mm_gen_regs(km, hash, qlens[s], seg[s].n_u, seg[s].u, seg[s].a);
|
||||||
n_regs[s] = seg[s].n_u;
|
n_regs[s] = seg[s].n_u;
|
||||||
for (i = 0; i < n_regs[s]; ++i) {
|
for (i = 0; i < n_regs[s]; ++i) {
|
||||||
regs[s][i].seg_split = 1;
|
regs[s][i].seg_split = 1;
|
||||||
|
|||||||
@@ -1,4 +1,37 @@
|
|||||||
|
/* The MIT License
|
||||||
|
|
||||||
|
Copyright (c) 2018- Dana-Farber Cancer Institute
|
||||||
|
2017-2018 Broad Institute, Inc.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
|
Modified Copyright (C) 2021 Intel Corporation
|
||||||
|
Contacts: Saurabh Kalikar <saurabh.kalikar@intel.com>;
|
||||||
|
Vasimuddin Md <vasimuddin.md@intel.com>; Sanchit Misra <sanchit.misra@intel.com>;
|
||||||
|
Chirag Jain <chirag@iisc.ac.in>; Heng Li <hli@jimmy.harvard.edu>
|
||||||
|
*/
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
|
#include<map>
|
||||||
|
#include <vector>
|
||||||
|
#include <fstream>
|
||||||
|
using namespace std;
|
||||||
#include <assert.h>
|
#include <assert.h>
|
||||||
#if defined(WIN32) || defined(_WIN32)
|
#if defined(WIN32) || defined(_WIN32)
|
||||||
#include <io.h> // for open(2)
|
#include <io.h> // for open(2)
|
||||||
@@ -53,6 +86,37 @@ mm_idx_t *mm_idx_init(int w, int k, int b, int flag)
|
|||||||
return mi;
|
return mi;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
void mm_idx_destroy_mm_hash(mm_idx_t *mi)
|
||||||
|
{
|
||||||
|
uint32_t i;
|
||||||
|
if (mi == 0) return;
|
||||||
|
if (mi->h) kh_destroy(str, (khash_t(str)*)mi->h);
|
||||||
|
if (mi->B) {
|
||||||
|
for (i = 0; i < 1U<<mi->b; ++i) {
|
||||||
|
free(mi->B[i].p);
|
||||||
|
free(mi->B[i].a.a);
|
||||||
|
kh_destroy(idx, (idxhash_t*)mi->B[i].h);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
void mm_idx_destroy_seq(mm_idx_t *mi)
|
||||||
|
{
|
||||||
|
uint32_t i;
|
||||||
|
if (mi->I) {
|
||||||
|
for (i = 0; i < mi->n_seq; ++i)
|
||||||
|
free(mi->I[i].a);
|
||||||
|
free(mi->I);
|
||||||
|
}
|
||||||
|
if (!mi->km) {
|
||||||
|
for (i = 0; i < mi->n_seq; ++i)
|
||||||
|
free(mi->seq[i].name);
|
||||||
|
free(mi->seq);
|
||||||
|
} else km_destroy(mi->km);
|
||||||
|
free(mi->B); free(mi->S); free(mi);
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
void mm_idx_destroy(mm_idx_t *mi)
|
void mm_idx_destroy(mm_idx_t *mi)
|
||||||
{
|
{
|
||||||
uint32_t i;
|
uint32_t i;
|
||||||
@@ -97,6 +161,81 @@ const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
//Output minimap2's hash table entries
|
||||||
|
void mm_idx_dump_hash(const char* f_name, const mm_idx_t *mi)
|
||||||
|
{
|
||||||
|
std::map<uint64_t, vector<uint64_t>> m;
|
||||||
|
|
||||||
|
ofstream f(f_name);
|
||||||
|
fprintf(stderr, "Building sorted key-val map\n");
|
||||||
|
|
||||||
|
uint32_t i,j;
|
||||||
|
uint64_t num_values = 0;
|
||||||
|
for (i = 0; i < 1U<<mi->b; ++i) {
|
||||||
|
|
||||||
|
|
||||||
|
//fprintf(stderr, "BucketID %lu \n", i);
|
||||||
|
idxhash_t *h = (idxhash_t*)mi->B[i].h;
|
||||||
|
khint_t k;
|
||||||
|
if (h == 0) continue;
|
||||||
|
for (k = 0; k < kh_end(h); ++k){
|
||||||
|
if (kh_exist(h, k)) {
|
||||||
|
uint64_t key = kh_key(h, k), bucket_id = i;
|
||||||
|
key = key>>1;
|
||||||
|
|
||||||
|
key = key<<mi->b | bucket_id;
|
||||||
|
|
||||||
|
if(kh_key(h, k)&1)
|
||||||
|
{
|
||||||
|
//print key value
|
||||||
|
//fprintf(stderr, "%llu %llu %llu\n", key, kh_val(h, k), 0);
|
||||||
|
m[key].push_back(kh_val(h, k));
|
||||||
|
}
|
||||||
|
else
|
||||||
|
{ // print key
|
||||||
|
uint32_t n = (uint32_t)kh_val(h, k);
|
||||||
|
//fprintf(stderr, "%llu %llu %llu ", key, kh_val(h, k), n);
|
||||||
|
// for 0 to lsb 32 val
|
||||||
|
// print b->p[msb 32 of val]
|
||||||
|
for(j = 0; j < n; j++)
|
||||||
|
{
|
||||||
|
//fprintf(stderr, "%llu ", mi->B[i].p[(kh_val(h, k)>>32) + j]);
|
||||||
|
m[key].push_back(mi->B[i].p[(kh_val(h, k)>>32) + j]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
fprintf(stderr, "Storing hash to %s \n", f_name);
|
||||||
|
vector<uint64_t> key_list;
|
||||||
|
key_list.push_back(m.size());
|
||||||
|
for(auto k : m){
|
||||||
|
key_list.push_back(k.first);
|
||||||
|
f<<k.first << " "<<k.second.size()<<endl;
|
||||||
|
for(int j = 0; j < k.second.size(); j++){
|
||||||
|
f<<k.second[j]<<" ";
|
||||||
|
num_values++;
|
||||||
|
}
|
||||||
|
f<<endl;
|
||||||
|
}
|
||||||
|
f.close();
|
||||||
|
string size_file_name = (string) f_name + "_size";
|
||||||
|
ofstream size_f(size_file_name);
|
||||||
|
size_f<<m.size()<<" "<<num_values;
|
||||||
|
size_f.close();
|
||||||
|
|
||||||
|
string prefix = (string)f_name + "_keys";
|
||||||
|
string keys_bin_file_name = prefix + ".uint64";
|
||||||
|
ofstream wf(keys_bin_file_name, ios::out | ios::binary);
|
||||||
|
wf.write((char*)&key_list[0], (key_list.size())*sizeof(uint64_t));
|
||||||
|
wf.close();
|
||||||
|
|
||||||
|
key_list.clear();
|
||||||
|
m.clear();
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
void mm_idx_stat(const mm_idx_t *mi)
|
void mm_idx_stat(const mm_idx_t *mi)
|
||||||
{
|
{
|
||||||
int n = 0, n1 = 0;
|
int n = 0, n1 = 0;
|
||||||
@@ -161,28 +300,6 @@ int mm_idx_getseq(const mm_idx_t *mi, uint32_t rid, uint32_t st, uint32_t en, ui
|
|||||||
return en - st;
|
return en - st;
|
||||||
}
|
}
|
||||||
|
|
||||||
int mm_idx_getseq_rev(const mm_idx_t *mi, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq)
|
|
||||||
{
|
|
||||||
uint64_t i, st1, en1;
|
|
||||||
const mm_idx_seq_t *s;
|
|
||||||
if (rid >= mi->n_seq || st >= mi->seq[rid].len) return -1;
|
|
||||||
s = &mi->seq[rid];
|
|
||||||
if (en > s->len) en = s->len;
|
|
||||||
st1 = s->offset + (s->len - en);
|
|
||||||
en1 = s->offset + (s->len - st);
|
|
||||||
for (i = st1; i < en1; ++i) {
|
|
||||||
uint8_t c = mm_seq4_get(mi->S, i);
|
|
||||||
seq[en1 - i - 1] = c < 4? 3 - c : c;
|
|
||||||
}
|
|
||||||
return en - st;
|
|
||||||
}
|
|
||||||
|
|
||||||
int mm_idx_getseq2(const mm_idx_t *mi, int is_rev, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq)
|
|
||||||
{
|
|
||||||
if (is_rev) return mm_idx_getseq_rev(mi, rid, st, en, seq);
|
|
||||||
else return mm_idx_getseq(mi, rid, st, en, seq);
|
|
||||||
}
|
|
||||||
|
|
||||||
int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f)
|
int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f)
|
||||||
{
|
{
|
||||||
int i;
|
int i;
|
||||||
|
|||||||
@@ -40,8 +40,7 @@ void *km_init2(void *km_par, size_t min_core_size)
|
|||||||
kmem_t *km;
|
kmem_t *km;
|
||||||
km = (kmem_t*)kcalloc(km_par, 1, sizeof(kmem_t));
|
km = (kmem_t*)kcalloc(km_par, 1, sizeof(kmem_t));
|
||||||
km->par = km_par;
|
km->par = km_par;
|
||||||
if (km_par) km->min_core_size = min_core_size > 0? min_core_size : ((kmem_t*)km_par)->min_core_size - 2;
|
km->min_core_size = min_core_size > 0? min_core_size : 0x80000;
|
||||||
else km->min_core_size = min_core_size > 0? min_core_size : 0x80000;
|
|
||||||
return (void*)km;
|
return (void*)km;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -184,16 +183,6 @@ void *krealloc(void *_km, void *ap, size_t n_bytes) // TODO: this can be made mo
|
|||||||
return q;
|
return q;
|
||||||
}
|
}
|
||||||
|
|
||||||
void *krelocate(void *km, void *ap, size_t n_bytes)
|
|
||||||
{
|
|
||||||
void *p;
|
|
||||||
if (km == 0 || ap == 0) return ap;
|
|
||||||
p = kmalloc(km, n_bytes);
|
|
||||||
memcpy(p, ap, n_bytes);
|
|
||||||
kfree(km, ap);
|
|
||||||
return p;
|
|
||||||
}
|
|
||||||
|
|
||||||
void km_stat(const void *_km, km_stat_t *s)
|
void km_stat(const void *_km, km_stat_t *s)
|
||||||
{
|
{
|
||||||
kmem_t *km = (kmem_t*)_km;
|
kmem_t *km = (kmem_t*)_km;
|
||||||
@@ -214,11 +203,3 @@ void km_stat(const void *_km, km_stat_t *s)
|
|||||||
s->largest = s->largest > size? s->largest : size;
|
s->largest = s->largest > size? s->largest : size;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void km_stat_print(const void *km)
|
|
||||||
{
|
|
||||||
km_stat_t st;
|
|
||||||
km_stat(km, &st);
|
|
||||||
fprintf(stderr, "[km_stat] cap=%ld, avail=%ld, largest=%ld, n_core=%ld, n_block=%ld\n",
|
|
||||||
st.capacity, st.available, st.largest, st.n_blocks, st.n_cores);
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -13,7 +13,6 @@ typedef struct {
|
|||||||
|
|
||||||
void *kmalloc(void *km, size_t size);
|
void *kmalloc(void *km, size_t size);
|
||||||
void *krealloc(void *km, void *ptr, size_t size);
|
void *krealloc(void *km, void *ptr, size_t size);
|
||||||
void *krelocate(void *km, void *ap, size_t n_bytes);
|
|
||||||
void *kcalloc(void *km, size_t count, size_t size);
|
void *kcalloc(void *km, size_t count, size_t size);
|
||||||
void kfree(void *km, void *ptr);
|
void kfree(void *km, void *ptr);
|
||||||
|
|
||||||
@@ -21,21 +20,11 @@ void *km_init(void);
|
|||||||
void *km_init2(void *km_par, size_t min_core_size);
|
void *km_init2(void *km_par, size_t min_core_size);
|
||||||
void km_destroy(void *km);
|
void km_destroy(void *km);
|
||||||
void km_stat(const void *_km, km_stat_t *s);
|
void km_stat(const void *_km, km_stat_t *s);
|
||||||
void km_stat_print(const void *km);
|
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#define Kmalloc(km, type, cnt) ((type*)kmalloc((km), (cnt) * sizeof(type)))
|
|
||||||
#define Kcalloc(km, type, cnt) ((type*)kcalloc((km), (cnt), sizeof(type)))
|
|
||||||
#define Krealloc(km, type, ptr, cnt) ((type*)krealloc((km), (ptr), (cnt) * sizeof(type)))
|
|
||||||
|
|
||||||
#define Kexpand(km, type, a, m) do { \
|
|
||||||
(m) = (m) >= 4? (m) + ((m)>>1) : 16; \
|
|
||||||
(a) = Krealloc(km, type, (a), (m)); \
|
|
||||||
} while (0)
|
|
||||||
|
|
||||||
#define KMALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kmalloc((km), (len) * sizeof(*(ptr))))
|
#define KMALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kmalloc((km), (len) * sizeof(*(ptr))))
|
||||||
#define KCALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kcalloc((km), (len), sizeof(*(ptr))))
|
#define KCALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kcalloc((km), (len), sizeof(*(ptr))))
|
||||||
#define KREALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))krealloc((km), (ptr), (len) * sizeof(*(ptr))))
|
#define KREALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))krealloc((km), (ptr), (len) * sizeof(*(ptr))))
|
||||||
@@ -45,43 +34,4 @@ void km_stat_print(const void *km);
|
|||||||
KREALLOC((km), (a), (m)); \
|
KREALLOC((km), (a), (m)); \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#ifndef klib_unused
|
|
||||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
|
||||||
#define klib_unused __attribute__ ((__unused__))
|
|
||||||
#else
|
|
||||||
#define klib_unused
|
|
||||||
#endif
|
|
||||||
#endif /* klib_unused */
|
|
||||||
|
|
||||||
#define KALLOC_POOL_INIT2(SCOPE, name, kmptype_t) \
|
|
||||||
typedef struct { \
|
|
||||||
size_t cnt, n, max; \
|
|
||||||
kmptype_t **buf; \
|
|
||||||
void *km; \
|
|
||||||
} kmp_##name##_t; \
|
|
||||||
SCOPE kmp_##name##_t *kmp_init_##name(void *km) { \
|
|
||||||
kmp_##name##_t *mp; \
|
|
||||||
mp = Kcalloc(km, kmp_##name##_t, 1); \
|
|
||||||
mp->km = km; \
|
|
||||||
return mp; \
|
|
||||||
} \
|
|
||||||
SCOPE void kmp_destroy_##name(kmp_##name##_t *mp) { \
|
|
||||||
size_t k; \
|
|
||||||
for (k = 0; k < mp->n; ++k) kfree(mp->km, mp->buf[k]); \
|
|
||||||
kfree(mp->km, mp->buf); kfree(mp->km, mp); \
|
|
||||||
} \
|
|
||||||
SCOPE kmptype_t *kmp_alloc_##name(kmp_##name##_t *mp) { \
|
|
||||||
++mp->cnt; \
|
|
||||||
if (mp->n == 0) return (kmptype_t*)kcalloc(mp->km, 1, sizeof(kmptype_t)); \
|
|
||||||
return mp->buf[--mp->n]; \
|
|
||||||
} \
|
|
||||||
SCOPE void kmp_free_##name(kmp_##name##_t *mp, kmptype_t *p) { \
|
|
||||||
--mp->cnt; \
|
|
||||||
if (mp->n == mp->max) Kexpand(mp->km, kmptype_t*, mp->buf, mp->max); \
|
|
||||||
mp->buf[mp->n++] = p; \
|
|
||||||
}
|
|
||||||
|
|
||||||
#define KALLOC_POOL_INIT(name, kmptype_t) \
|
|
||||||
KALLOC_POOL_INIT2(static inline klib_unused, name, kmptype_t)
|
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,474 +0,0 @@
|
|||||||
/* The MIT License
|
|
||||||
|
|
||||||
Copyright (c) 2019 by Attractive Chaos <attractor@live.co.uk>
|
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining
|
|
||||||
a copy of this software and associated documentation files (the
|
|
||||||
"Software"), to deal in the Software without restriction, including
|
|
||||||
without limitation the rights to use, copy, modify, merge, publish,
|
|
||||||
distribute, sublicense, and/or sell copies of the Software, and to
|
|
||||||
permit persons to whom the Software is furnished to do so, subject to
|
|
||||||
the following conditions:
|
|
||||||
|
|
||||||
The above copyright notice and this permission notice shall be
|
|
||||||
included in all copies or substantial portions of the Software.
|
|
||||||
|
|
||||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
|
||||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
|
||||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
|
||||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
|
||||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
|
||||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
|
||||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
||||||
SOFTWARE.
|
|
||||||
*/
|
|
||||||
|
|
||||||
/* An example:
|
|
||||||
|
|
||||||
#include <stdio.h>
|
|
||||||
#include <string.h>
|
|
||||||
#include <stdlib.h>
|
|
||||||
#include "krmq.h"
|
|
||||||
|
|
||||||
struct my_node {
|
|
||||||
char key;
|
|
||||||
KRMQ_HEAD(struct my_node) head;
|
|
||||||
};
|
|
||||||
#define my_cmp(p, q) (((q)->key < (p)->key) - ((p)->key < (q)->key))
|
|
||||||
KRMQ_INIT(my, struct my_node, head, my_cmp)
|
|
||||||
|
|
||||||
int main(void) {
|
|
||||||
const char *str = "MNOLKQOPHIA"; // from wiki, except a duplicate
|
|
||||||
struct my_node *root = 0;
|
|
||||||
int i, l = strlen(str);
|
|
||||||
for (i = 0; i < l; ++i) { // insert in the input order
|
|
||||||
struct my_node *q, *p = malloc(sizeof(*p));
|
|
||||||
p->key = str[i];
|
|
||||||
q = krmq_insert(my, &root, p, 0);
|
|
||||||
if (p != q) free(p); // if already present, free
|
|
||||||
}
|
|
||||||
krmq_itr_t(my) itr;
|
|
||||||
krmq_itr_first(my, root, &itr); // place at first
|
|
||||||
do { // traverse
|
|
||||||
const struct my_node *p = krmq_at(&itr);
|
|
||||||
putchar(p->key);
|
|
||||||
free((void*)p); // free node
|
|
||||||
} while (krmq_itr_next(my, &itr));
|
|
||||||
putchar('\n');
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
*/
|
|
||||||
|
|
||||||
#ifndef KRMQ_H
|
|
||||||
#define KRMQ_H
|
|
||||||
|
|
||||||
#ifdef __STRICT_ANSI__
|
|
||||||
#define inline __inline__
|
|
||||||
#endif
|
|
||||||
|
|
||||||
#define KRMQ_MAX_DEPTH 64
|
|
||||||
|
|
||||||
#define krmq_size(head, p) ((p)? (p)->head.size : 0)
|
|
||||||
#define krmq_size_child(head, q, i) ((q)->head.p[(i)]? (q)->head.p[(i)]->head.size : 0)
|
|
||||||
|
|
||||||
#define KRMQ_HEAD(__type) \
|
|
||||||
struct { \
|
|
||||||
__type *p[2], *s; \
|
|
||||||
signed char balance; /* balance factor */ \
|
|
||||||
unsigned size; /* #elements in subtree */ \
|
|
||||||
}
|
|
||||||
|
|
||||||
#define __KRMQ_FIND(suf, __scope, __type, __head, __cmp) \
|
|
||||||
__scope __type *krmq_find_##suf(const __type *root, const __type *x, unsigned *cnt_) { \
|
|
||||||
const __type *p = root; \
|
|
||||||
unsigned cnt = 0; \
|
|
||||||
while (p != 0) { \
|
|
||||||
int cmp; \
|
|
||||||
cmp = __cmp(x, p); \
|
|
||||||
if (cmp >= 0) cnt += krmq_size_child(__head, p, 0) + 1; \
|
|
||||||
if (cmp < 0) p = p->__head.p[0]; \
|
|
||||||
else if (cmp > 0) p = p->__head.p[1]; \
|
|
||||||
else break; \
|
|
||||||
} \
|
|
||||||
if (cnt_) *cnt_ = cnt; \
|
|
||||||
return (__type*)p; \
|
|
||||||
} \
|
|
||||||
__scope __type *krmq_interval_##suf(const __type *root, const __type *x, __type **lower, __type **upper) { \
|
|
||||||
const __type *p = root, *l = 0, *u = 0; \
|
|
||||||
while (p != 0) { \
|
|
||||||
int cmp; \
|
|
||||||
cmp = __cmp(x, p); \
|
|
||||||
if (cmp < 0) u = p, p = p->__head.p[0]; \
|
|
||||||
else if (cmp > 0) l = p, p = p->__head.p[1]; \
|
|
||||||
else { l = u = p; break; } \
|
|
||||||
} \
|
|
||||||
if (lower) *lower = (__type*)l; \
|
|
||||||
if (upper) *upper = (__type*)u; \
|
|
||||||
return (__type*)p; \
|
|
||||||
}
|
|
||||||
|
|
||||||
#define __KRMQ_RMQ(suf, __scope, __type, __head, __cmp, __lt2) \
|
|
||||||
__scope __type *krmq_rmq_##suf(const __type *root, const __type *lo, const __type *up) { /* CLOSED interval */ \
|
|
||||||
const __type *p = root, *path[2][KRMQ_MAX_DEPTH], *min; \
|
|
||||||
int plen[2] = {0, 0}, pcmp[2][KRMQ_MAX_DEPTH], i, cmp, lca; \
|
|
||||||
if (root == 0) return 0; \
|
|
||||||
while (p) { \
|
|
||||||
cmp = __cmp(lo, p); \
|
|
||||||
path[0][plen[0]] = p, pcmp[0][plen[0]++] = cmp; \
|
|
||||||
if (cmp < 0) p = p->__head.p[0]; \
|
|
||||||
else if (cmp > 0) p = p->__head.p[1]; \
|
|
||||||
else break; \
|
|
||||||
} \
|
|
||||||
p = root; \
|
|
||||||
while (p) { \
|
|
||||||
cmp = __cmp(up, p); \
|
|
||||||
path[1][plen[1]] = p, pcmp[1][plen[1]++] = cmp; \
|
|
||||||
if (cmp < 0) p = p->__head.p[0]; \
|
|
||||||
else if (cmp > 0) p = p->__head.p[1]; \
|
|
||||||
else break; \
|
|
||||||
} \
|
|
||||||
for (i = 0; i < plen[0] && i < plen[1]; ++i) /* find the LCA */ \
|
|
||||||
if (path[0][i] == path[1][i] && pcmp[0][i] <= 0 && pcmp[1][i] >= 0) \
|
|
||||||
break; \
|
|
||||||
if (i == plen[0] || i == plen[1]) return 0; /* no elements in the closed interval */ \
|
|
||||||
lca = i, min = path[0][lca]; \
|
|
||||||
for (i = lca + 1; i < plen[0]; ++i) { \
|
|
||||||
if (pcmp[0][i] <= 0) { \
|
|
||||||
if (__lt2(path[0][i], min)) min = path[0][i]; \
|
|
||||||
if (path[0][i]->__head.p[1] && __lt2(path[0][i]->__head.p[1]->__head.s, min)) \
|
|
||||||
min = path[0][i]->__head.p[1]->__head.s; \
|
|
||||||
} \
|
|
||||||
} \
|
|
||||||
for (i = lca + 1; i < plen[1]; ++i) { \
|
|
||||||
if (pcmp[1][i] >= 0) { \
|
|
||||||
if (__lt2(path[1][i], min)) min = path[1][i]; \
|
|
||||||
if (path[1][i]->__head.p[0] && __lt2(path[1][i]->__head.p[0]->__head.s, min)) \
|
|
||||||
min = path[1][i]->__head.p[0]->__head.s; \
|
|
||||||
} \
|
|
||||||
} \
|
|
||||||
return (__type*)min; \
|
|
||||||
}
|
|
||||||
|
|
||||||
#define __KRMQ_ROTATE(suf, __type, __head, __lt2) \
|
|
||||||
/* */ \
|
|
||||||
static inline void krmq_update_min_##suf(__type *p, const __type *q, const __type *r) { \
|
|
||||||
p->__head.s = !q || __lt2(p, q->__head.s)? p : q->__head.s; \
|
|
||||||
p->__head.s = !r || __lt2(p->__head.s, r->__head.s)? p->__head.s : r->__head.s; \
|
|
||||||
} \
|
|
||||||
/* one rotation: (a,(b,c)q)p => ((a,b)p,c)q */ \
|
|
||||||
static inline __type *krmq_rotate1_##suf(__type *p, int dir) { /* dir=0 to left; dir=1 to right */ \
|
|
||||||
int opp = 1 - dir; /* opposite direction */ \
|
|
||||||
__type *q = p->__head.p[opp], *s = p->__head.s; \
|
|
||||||
unsigned size_p = p->__head.size; \
|
|
||||||
p->__head.size -= q->__head.size - krmq_size_child(__head, q, dir); \
|
|
||||||
q->__head.size = size_p; \
|
|
||||||
krmq_update_min_##suf(p, p->__head.p[dir], q->__head.p[dir]); \
|
|
||||||
q->__head.s = s; \
|
|
||||||
p->__head.p[opp] = q->__head.p[dir]; \
|
|
||||||
q->__head.p[dir] = p; \
|
|
||||||
return q; \
|
|
||||||
} \
|
|
||||||
/* two consecutive rotations: (a,((b,c)r,d)q)p => ((a,b)p,(c,d)q)r */ \
|
|
||||||
static inline __type *krmq_rotate2_##suf(__type *p, int dir) { \
|
|
||||||
int b1, opp = 1 - dir; \
|
|
||||||
__type *q = p->__head.p[opp], *r = q->__head.p[dir], *s = p->__head.s; \
|
|
||||||
unsigned size_x_dir = krmq_size_child(__head, r, dir); \
|
|
||||||
r->__head.size = p->__head.size; \
|
|
||||||
p->__head.size -= q->__head.size - size_x_dir; \
|
|
||||||
q->__head.size -= size_x_dir + 1; \
|
|
||||||
krmq_update_min_##suf(p, p->__head.p[dir], r->__head.p[dir]); \
|
|
||||||
krmq_update_min_##suf(q, q->__head.p[opp], r->__head.p[opp]); \
|
|
||||||
r->__head.s = s; \
|
|
||||||
p->__head.p[opp] = r->__head.p[dir]; \
|
|
||||||
r->__head.p[dir] = p; \
|
|
||||||
q->__head.p[dir] = r->__head.p[opp]; \
|
|
||||||
r->__head.p[opp] = q; \
|
|
||||||
b1 = dir == 0? +1 : -1; \
|
|
||||||
if (r->__head.balance == b1) q->__head.balance = 0, p->__head.balance = -b1; \
|
|
||||||
else if (r->__head.balance == 0) q->__head.balance = p->__head.balance = 0; \
|
|
||||||
else q->__head.balance = b1, p->__head.balance = 0; \
|
|
||||||
r->__head.balance = 0; \
|
|
||||||
return r; \
|
|
||||||
}
|
|
||||||
|
|
||||||
#define __KRMQ_INSERT(suf, __scope, __type, __head, __cmp, __lt2) \
|
|
||||||
__scope __type *krmq_insert_##suf(__type **root_, __type *x, unsigned *cnt_) { \
|
|
||||||
unsigned char stack[KRMQ_MAX_DEPTH]; \
|
|
||||||
__type *path[KRMQ_MAX_DEPTH]; \
|
|
||||||
__type *bp, *bq; \
|
|
||||||
__type *p, *q, *r = 0; /* _r_ is potentially the new root */ \
|
|
||||||
int i, which = 0, top, b1, path_len; \
|
|
||||||
unsigned cnt = 0; \
|
|
||||||
bp = *root_, bq = 0; \
|
|
||||||
/* find the insertion location */ \
|
|
||||||
for (p = bp, q = bq, top = path_len = 0; p; q = p, p = p->__head.p[which]) { \
|
|
||||||
int cmp; \
|
|
||||||
cmp = __cmp(x, p); \
|
|
||||||
if (cmp >= 0) cnt += krmq_size_child(__head, p, 0) + 1; \
|
|
||||||
if (cmp == 0) { \
|
|
||||||
if (cnt_) *cnt_ = cnt; \
|
|
||||||
return p; \
|
|
||||||
} \
|
|
||||||
if (p->__head.balance != 0) \
|
|
||||||
bq = q, bp = p, top = 0; \
|
|
||||||
stack[top++] = which = (cmp > 0); \
|
|
||||||
path[path_len++] = p; \
|
|
||||||
} \
|
|
||||||
if (cnt_) *cnt_ = cnt; \
|
|
||||||
x->__head.balance = 0, x->__head.size = 1, x->__head.p[0] = x->__head.p[1] = 0, x->__head.s = x; \
|
|
||||||
if (q == 0) *root_ = x; \
|
|
||||||
else q->__head.p[which] = x; \
|
|
||||||
if (bp == 0) return x; \
|
|
||||||
for (i = 0; i < path_len; ++i) ++path[i]->__head.size; \
|
|
||||||
for (i = path_len - 1; i >= 0; --i) { \
|
|
||||||
krmq_update_min_##suf(path[i], path[i]->__head.p[0], path[i]->__head.p[1]); \
|
|
||||||
if (path[i]->__head.s != x) break; \
|
|
||||||
} \
|
|
||||||
for (p = bp, top = 0; p != x; p = p->__head.p[stack[top]], ++top) /* update balance factors */ \
|
|
||||||
if (stack[top] == 0) --p->__head.balance; \
|
|
||||||
else ++p->__head.balance; \
|
|
||||||
if (bp->__head.balance > -2 && bp->__head.balance < 2) return x; /* no re-balance needed */ \
|
|
||||||
/* re-balance */ \
|
|
||||||
which = (bp->__head.balance < 0); \
|
|
||||||
b1 = which == 0? +1 : -1; \
|
|
||||||
q = bp->__head.p[1 - which]; \
|
|
||||||
if (q->__head.balance == b1) { \
|
|
||||||
r = krmq_rotate1_##suf(bp, which); \
|
|
||||||
q->__head.balance = bp->__head.balance = 0; \
|
|
||||||
} else r = krmq_rotate2_##suf(bp, which); \
|
|
||||||
if (bq == 0) *root_ = r; \
|
|
||||||
else bq->__head.p[bp != bq->__head.p[0]] = r; \
|
|
||||||
return x; \
|
|
||||||
}
|
|
||||||
|
|
||||||
#define __KRMQ_ERASE(suf, __scope, __type, __head, __cmp, __lt2) \
|
|
||||||
__scope __type *krmq_erase_##suf(__type **root_, const __type *x, unsigned *cnt_) { \
|
|
||||||
__type *p, *path[KRMQ_MAX_DEPTH], fake; \
|
|
||||||
unsigned char dir[KRMQ_MAX_DEPTH]; \
|
|
||||||
int i, d = 0, cmp; \
|
|
||||||
unsigned cnt = 0; \
|
|
||||||
fake = **root_, fake.__head.p[0] = *root_, fake.__head.p[1] = 0; \
|
|
||||||
if (cnt_) *cnt_ = 0; \
|
|
||||||
if (x) { \
|
|
||||||
for (cmp = -1, p = &fake; cmp; cmp = __cmp(x, p)) { \
|
|
||||||
int which = (cmp > 0); \
|
|
||||||
if (cmp > 0) cnt += krmq_size_child(__head, p, 0) + 1; \
|
|
||||||
dir[d] = which; \
|
|
||||||
path[d++] = p; \
|
|
||||||
p = p->__head.p[which]; \
|
|
||||||
if (p == 0) { \
|
|
||||||
if (cnt_) *cnt_ = 0; \
|
|
||||||
return 0; \
|
|
||||||
} \
|
|
||||||
} \
|
|
||||||
cnt += krmq_size_child(__head, p, 0) + 1; /* because p==x is not counted */ \
|
|
||||||
} else { \
|
|
||||||
for (p = &fake, cnt = 1; p; p = p->__head.p[0]) \
|
|
||||||
dir[d] = 0, path[d++] = p; \
|
|
||||||
p = path[--d]; \
|
|
||||||
} \
|
|
||||||
if (cnt_) *cnt_ = cnt; \
|
|
||||||
for (i = 1; i < d; ++i) --path[i]->__head.size; \
|
|
||||||
if (p->__head.p[1] == 0) { /* ((1,.)2,3)4 => (1,3)4; p=2 */ \
|
|
||||||
path[d-1]->__head.p[dir[d-1]] = p->__head.p[0]; \
|
|
||||||
} else { \
|
|
||||||
__type *q = p->__head.p[1]; \
|
|
||||||
if (q->__head.p[0] == 0) { /* ((1,2)3,4)5 => ((1)2,4)5; p=3,q=2 */ \
|
|
||||||
q->__head.p[0] = p->__head.p[0]; \
|
|
||||||
q->__head.balance = p->__head.balance; \
|
|
||||||
path[d-1]->__head.p[dir[d-1]] = q; \
|
|
||||||
path[d] = q, dir[d++] = 1; \
|
|
||||||
q->__head.size = p->__head.size - 1; \
|
|
||||||
} else { /* ((1,((.,2)3,4)5)6,7)8 => ((1,(2,4)5)3,7)8; p=6 */ \
|
|
||||||
__type *r; \
|
|
||||||
int e = d++; /* backup _d_ */\
|
|
||||||
for (;;) { \
|
|
||||||
dir[d] = 0; \
|
|
||||||
path[d++] = q; \
|
|
||||||
r = q->__head.p[0]; \
|
|
||||||
if (r->__head.p[0] == 0) break; \
|
|
||||||
q = r; \
|
|
||||||
} \
|
|
||||||
r->__head.p[0] = p->__head.p[0]; \
|
|
||||||
q->__head.p[0] = r->__head.p[1]; \
|
|
||||||
r->__head.p[1] = p->__head.p[1]; \
|
|
||||||
r->__head.balance = p->__head.balance; \
|
|
||||||
path[e-1]->__head.p[dir[e-1]] = r; \
|
|
||||||
path[e] = r, dir[e] = 1; \
|
|
||||||
for (i = e + 1; i < d; ++i) --path[i]->__head.size; \
|
|
||||||
r->__head.size = p->__head.size - 1; \
|
|
||||||
} \
|
|
||||||
} \
|
|
||||||
for (i = d - 1; i >= 0; --i) /* not sure why adding condition "path[i]->__head.s==p" doesn't work */ \
|
|
||||||
krmq_update_min_##suf(path[i], path[i]->__head.p[0], path[i]->__head.p[1]); \
|
|
||||||
while (--d > 0) { \
|
|
||||||
__type *q = path[d]; \
|
|
||||||
int which, other, b1 = 1, b2 = 2; \
|
|
||||||
which = dir[d], other = 1 - which; \
|
|
||||||
if (which) b1 = -b1, b2 = -b2; \
|
|
||||||
q->__head.balance += b1; \
|
|
||||||
if (q->__head.balance == b1) break; \
|
|
||||||
else if (q->__head.balance == b2) { \
|
|
||||||
__type *r = q->__head.p[other]; \
|
|
||||||
if (r->__head.balance == -b1) { \
|
|
||||||
path[d-1]->__head.p[dir[d-1]] = krmq_rotate2_##suf(q, which); \
|
|
||||||
} else { \
|
|
||||||
path[d-1]->__head.p[dir[d-1]] = krmq_rotate1_##suf(q, which); \
|
|
||||||
if (r->__head.balance == 0) { \
|
|
||||||
r->__head.balance = -b1; \
|
|
||||||
q->__head.balance = b1; \
|
|
||||||
break; \
|
|
||||||
} else r->__head.balance = q->__head.balance = 0; \
|
|
||||||
} \
|
|
||||||
} \
|
|
||||||
} \
|
|
||||||
*root_ = fake.__head.p[0]; \
|
|
||||||
return p; \
|
|
||||||
}
|
|
||||||
|
|
||||||
#define krmq_free(__type, __head, __root, __free) do { \
|
|
||||||
__type *_p, *_q; \
|
|
||||||
for (_p = __root; _p; _p = _q) { \
|
|
||||||
if (_p->__head.p[0] == 0) { \
|
|
||||||
_q = _p->__head.p[1]; \
|
|
||||||
__free(_p); \
|
|
||||||
} else { \
|
|
||||||
_q = _p->__head.p[0]; \
|
|
||||||
_p->__head.p[0] = _q->__head.p[1]; \
|
|
||||||
_q->__head.p[1] = _p; \
|
|
||||||
} \
|
|
||||||
} \
|
|
||||||
} while (0)
|
|
||||||
|
|
||||||
#define __KRMQ_ITR(suf, __scope, __type, __head, __cmp) \
|
|
||||||
struct krmq_itr_##suf { \
|
|
||||||
const __type *stack[KRMQ_MAX_DEPTH], **top; \
|
|
||||||
}; \
|
|
||||||
__scope void krmq_itr_first_##suf(const __type *root, struct krmq_itr_##suf *itr) { \
|
|
||||||
const __type *p; \
|
|
||||||
for (itr->top = itr->stack - 1, p = root; p; p = p->__head.p[0]) \
|
|
||||||
*++itr->top = p; \
|
|
||||||
} \
|
|
||||||
__scope int krmq_itr_find_##suf(const __type *root, const __type *x, struct krmq_itr_##suf *itr) { \
|
|
||||||
const __type *p = root; \
|
|
||||||
itr->top = itr->stack - 1; \
|
|
||||||
while (p != 0) { \
|
|
||||||
int cmp; \
|
|
||||||
*++itr->top = p; \
|
|
||||||
cmp = __cmp(x, p); \
|
|
||||||
if (cmp < 0) p = p->__head.p[0]; \
|
|
||||||
else if (cmp > 0) p = p->__head.p[1]; \
|
|
||||||
else break; \
|
|
||||||
} \
|
|
||||||
return p? 1 : 0; \
|
|
||||||
} \
|
|
||||||
__scope int krmq_itr_next_bidir_##suf(struct krmq_itr_##suf *itr, int dir) { \
|
|
||||||
const __type *p; \
|
|
||||||
if (itr->top < itr->stack) return 0; \
|
|
||||||
dir = !!dir; \
|
|
||||||
p = (*itr->top)->__head.p[dir]; \
|
|
||||||
if (p) { /* go down */ \
|
|
||||||
for (; p; p = p->__head.p[!dir]) \
|
|
||||||
*++itr->top = p; \
|
|
||||||
return 1; \
|
|
||||||
} else { /* go up */ \
|
|
||||||
do { \
|
|
||||||
p = *itr->top--; \
|
|
||||||
} while (itr->top >= itr->stack && p == (*itr->top)->__head.p[dir]); \
|
|
||||||
return itr->top < itr->stack? 0 : 1; \
|
|
||||||
} \
|
|
||||||
} \
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Insert a node to the tree
|
|
||||||
*
|
|
||||||
* @param suf name suffix used in KRMQ_INIT()
|
|
||||||
* @param proot pointer to the root of the tree (in/out: root may change)
|
|
||||||
* @param x node to insert (in)
|
|
||||||
* @param cnt number of nodes smaller than or equal to _x_; can be NULL (out)
|
|
||||||
*
|
|
||||||
* @return _x_ if not present in the tree, or the node equal to x.
|
|
||||||
*/
|
|
||||||
#define krmq_insert(suf, proot, x, cnt) krmq_insert_##suf(proot, x, cnt)
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Find a node in the tree
|
|
||||||
*
|
|
||||||
* @param suf name suffix used in KRMQ_INIT()
|
|
||||||
* @param root root of the tree
|
|
||||||
* @param x node value to find (in)
|
|
||||||
* @param cnt number of nodes smaller than or equal to _x_; can be NULL (out)
|
|
||||||
*
|
|
||||||
* @return node equal to _x_ if present, or NULL if absent
|
|
||||||
*/
|
|
||||||
#define krmq_find(suf, root, x, cnt) krmq_find_##suf(root, x, cnt)
|
|
||||||
#define krmq_interval(suf, root, x, lower, upper) krmq_interval_##suf(root, x, lower, upper)
|
|
||||||
#define krmq_rmq(suf, root, lo, up) krmq_rmq_##suf(root, lo, up)
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Delete a node from the tree
|
|
||||||
*
|
|
||||||
* @param suf name suffix used in KRMQ_INIT()
|
|
||||||
* @param proot pointer to the root of the tree (in/out: root may change)
|
|
||||||
* @param x node value to delete; if NULL, delete the first node (in)
|
|
||||||
*
|
|
||||||
* @return node removed from the tree if present, or NULL if absent
|
|
||||||
*/
|
|
||||||
#define krmq_erase(suf, proot, x, cnt) krmq_erase_##suf(proot, x, cnt)
|
|
||||||
#define krmq_erase_first(suf, proot) krmq_erase_##suf(proot, 0, 0)
|
|
||||||
|
|
||||||
#define krmq_itr_t(suf) struct krmq_itr_##suf
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Place the iterator at the smallest object
|
|
||||||
*
|
|
||||||
* @param suf name suffix used in KRMQ_INIT()
|
|
||||||
* @param root root of the tree
|
|
||||||
* @param itr iterator
|
|
||||||
*/
|
|
||||||
#define krmq_itr_first(suf, root, itr) krmq_itr_first_##suf(root, itr)
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Place the iterator at the object equal to or greater than the query
|
|
||||||
*
|
|
||||||
* @param suf name suffix used in KRMQ_INIT()
|
|
||||||
* @param root root of the tree
|
|
||||||
* @param x query (in)
|
|
||||||
* @param itr iterator (out)
|
|
||||||
*
|
|
||||||
* @return 1 if find; 0 otherwise. krmq_at(itr) is NULL if and only if query is
|
|
||||||
* larger than all objects in the tree
|
|
||||||
*/
|
|
||||||
#define krmq_itr_find(suf, root, x, itr) krmq_itr_find_##suf(root, x, itr)
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Move to the next object in order
|
|
||||||
*
|
|
||||||
* @param itr iterator (modified)
|
|
||||||
*
|
|
||||||
* @return 1 if there is a next object; 0 otherwise
|
|
||||||
*/
|
|
||||||
#define krmq_itr_next(suf, itr) krmq_itr_next_bidir_##suf(itr, 1)
|
|
||||||
#define krmq_itr_prev(suf, itr) krmq_itr_next_bidir_##suf(itr, 0)
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Return the pointer at the iterator
|
|
||||||
*
|
|
||||||
* @param itr iterator
|
|
||||||
*
|
|
||||||
* @return pointer if present; NULL otherwise
|
|
||||||
*/
|
|
||||||
#define krmq_at(itr) ((itr)->top < (itr)->stack? 0 : *(itr)->top)
|
|
||||||
|
|
||||||
#define KRMQ_INIT2(suf, __scope, __type, __head, __cmp, __lt2) \
|
|
||||||
__KRMQ_FIND(suf, __scope, __type, __head, __cmp) \
|
|
||||||
__KRMQ_RMQ(suf, __scope, __type, __head, __cmp, __lt2) \
|
|
||||||
__KRMQ_ROTATE(suf, __type, __head, __lt2) \
|
|
||||||
__KRMQ_INSERT(suf, __scope, __type, __head, __cmp, __lt2) \
|
|
||||||
__KRMQ_ERASE(suf, __scope, __type, __head, __cmp, __lt2) \
|
|
||||||
__KRMQ_ITR(suf, __scope, __type, __head, __cmp)
|
|
||||||
|
|
||||||
#define KRMQ_INIT(suf, __type, __head, __cmp, __lt2) \
|
|
||||||
KRMQ_INIT2(suf,, __type, __head, __cmp, __lt2)
|
|
||||||
|
|
||||||
#endif
|
|
||||||
@@ -15,14 +15,6 @@
|
|||||||
#define KSW_EZ_SPLICE_FOR 0x100
|
#define KSW_EZ_SPLICE_FOR 0x100
|
||||||
#define KSW_EZ_SPLICE_REV 0x200
|
#define KSW_EZ_SPLICE_REV 0x200
|
||||||
#define KSW_EZ_SPLICE_FLANK 0x400
|
#define KSW_EZ_SPLICE_FLANK 0x400
|
||||||
#define KSW_EZ_SPLICE_CMPLX 0x800
|
|
||||||
|
|
||||||
// The subset of CIGAR operators used by ksw code.
|
|
||||||
// Use MM_CIGAR_* from minimap.h if you need the full list.
|
|
||||||
#define KSW_CIGAR_MATCH 0
|
|
||||||
#define KSW_CIGAR_INS 1
|
|
||||||
#define KSW_CIGAR_DEL 2
|
|
||||||
#define KSW_CIGAR_N_SKIP 3
|
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
extern "C" {
|
extern "C" {
|
||||||
@@ -145,13 +137,13 @@ static inline void ksw_backtrack(void *km, int is_rot, int is_rev, int min_intro
|
|||||||
else if (!(tmp >> (state + 2) & 1)) state = 0; // if requesting other states, _state_ stays the same if it is a continuation; otherwise, set to H
|
else if (!(tmp >> (state + 2) & 1)) state = 0; // if requesting other states, _state_ stays the same if it is a continuation; otherwise, set to H
|
||||||
if (state == 0) state = tmp & 7; // TODO: probably this line can be merged into the "else if" line right above; not 100% sure
|
if (state == 0) state = tmp & 7; // TODO: probably this line can be merged into the "else if" line right above; not 100% sure
|
||||||
if (force_state >= 0) state = force_state;
|
if (force_state >= 0) state = force_state;
|
||||||
if (state == 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_MATCH, 1), --i, --j;
|
if (state == 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 0, 1), --i, --j; // match
|
||||||
else if (state == 1 || (state == 3 && min_intron_len <= 0)) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_DEL, 1), --i;
|
else if (state == 1 || (state == 3 && min_intron_len <= 0)) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 2, 1), --i; // deletion
|
||||||
else if (state == 3 && min_intron_len > 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_N_SKIP, 1), --i;
|
else if (state == 3 && min_intron_len > 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 3, 1), --i; // intron
|
||||||
else cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_INS, 1), --j;
|
else cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, 1), --j; // insertion
|
||||||
}
|
}
|
||||||
if (i >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, min_intron_len > 0 && i >= min_intron_len? KSW_CIGAR_N_SKIP : KSW_CIGAR_DEL, i + 1); // first deletion
|
if (i >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, min_intron_len > 0 && i >= min_intron_len? 3 : 2, i + 1); // first deletion
|
||||||
if (j >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_INS, j + 1); // first insertion
|
if (j >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, j + 1); // first insertion
|
||||||
if (!is_rev)
|
if (!is_rev)
|
||||||
for (i = 0; i < n_cigar>>1; ++i) // reverse CIGAR
|
for (i = 0; i < n_cigar>>1; ++i) // reverse CIGAR
|
||||||
tmp = cigar[i], cigar[i] = cigar[n_cigar-1-i], cigar[n_cigar-1-i] = tmp;
|
tmp = cigar[i], cigar[i] = cigar[n_cigar-1-i], cigar[n_cigar-1-i] = tmp;
|
||||||
|
|||||||
+1340
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,42 @@
|
|||||||
|
/* The MIT License
|
||||||
|
|
||||||
|
Copyright (c) 2018- Dana-Farber Cancer Institute
|
||||||
|
2017-2018 Broad Institute, Inc.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
|
Modified Copyright (C) 2021 Intel Corporation
|
||||||
|
Contacts: Saurabh Kalikar <saurabh.kalikar@intel.com>;
|
||||||
|
Vasimuddin Md <vasimuddin.md@intel.com>; Sanchit Misra <sanchit.misra@intel.com>;
|
||||||
|
Chirag Jain <chirag@iisc.ac.in>; Heng Li <hli@jimmy.harvard.edu>
|
||||||
|
*/
|
||||||
|
#include <string.h>
|
||||||
|
#include <stdio.h>
|
||||||
|
#include <assert.h>
|
||||||
|
#include "ksw2.h"
|
||||||
|
#include <immintrin.h>
|
||||||
|
#include <x86intrin.h>
|
||||||
|
#include <smmintrin.h>
|
||||||
|
#include <emmintrin.h>
|
||||||
|
void ksw_extd2_avx512(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
|
int8_t q, int8_t e, int8_t q2, int8_t e2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||||
|
|
||||||
|
void ksw_extd2_avx2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
|
int8_t q, int8_t e, int8_t q2, int8_t e2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||||
+1
-1
@@ -358,7 +358,7 @@ void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||||
// update ez
|
// update ez
|
||||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||||
ez->mte = H[en0], ez->mte_q = r - en0;
|
ez->mte = H[en0], ez->mte_q = r - en;
|
||||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e2)) break;
|
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e2)) break;
|
||||||
|
|||||||
+40
-79
@@ -71,7 +71,6 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
|
|
||||||
ksw_reset_extz(ez);
|
ksw_reset_extz(ez);
|
||||||
if (m <= 1 || qlen <= 0 || tlen <= 0 || q2 <= q + e) return;
|
if (m <= 1 || qlen <= 0 || tlen <= 0 || q2 <= q + e) return;
|
||||||
assert((flag & KSW_EZ_SPLICE_FOR) == 0 || (flag & KSW_EZ_SPLICE_REV) == 0); // can't be both set
|
|
||||||
|
|
||||||
zero_ = _mm_set1_epi8(0);
|
zero_ = _mm_set1_epi8(0);
|
||||||
q_ = _mm_set1_epi8(q);
|
q_ = _mm_set1_epi8(q);
|
||||||
@@ -119,93 +118,55 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
|
|
||||||
// set the donor and acceptor arrays. TODO: this assumes 0/1/2/3 encoding!
|
// set the donor and acceptor arrays. TODO: this assumes 0/1/2/3 encoding!
|
||||||
if (flag & (KSW_EZ_SPLICE_FOR|KSW_EZ_SPLICE_REV)) {
|
if (flag & (KSW_EZ_SPLICE_FOR|KSW_EZ_SPLICE_REV)) {
|
||||||
const int sp0[4] = { 8, 15, 21, 30 };
|
int semi_cost = flag&KSW_EZ_SPLICE_FLANK? -noncan/2 : 0; // GTr or yAG is worth 0.5 bit; see PMID:18688272
|
||||||
int sp[4];
|
memset(donor, -noncan, tlen_ * 16);
|
||||||
if (flag & KSW_EZ_SPLICE_CMPLX) {
|
memset(acceptor, -noncan, tlen_ * 16);
|
||||||
for (t = 0; t < 4; ++t)
|
|
||||||
sp[t] = (int)((double)sp0[t] / 3. + .499);
|
|
||||||
} else {
|
|
||||||
sp[0] = flag&KSW_EZ_SPLICE_FLANK? noncan / 2 : 0;
|
|
||||||
sp[1] = sp[2] = sp[3] = noncan;
|
|
||||||
}
|
|
||||||
memset(donor, -sp[3], tlen_ * 16);
|
|
||||||
memset(acceptor, -sp[3], tlen_ * 16);
|
|
||||||
if (!(flag & KSW_EZ_REV_CIGAR)) {
|
if (!(flag & KSW_EZ_REV_CIGAR)) {
|
||||||
for (t = 0; t < tlen - 4; ++t) {
|
for (t = 0; t < tlen - 4; ++t) {
|
||||||
int z = 3;
|
int can_type = 0; // type of canonical site: 0=none, 1=GT/AG only, 2=GTr/yAG
|
||||||
if (flag & KSW_EZ_SPLICE_FOR) {
|
if ((flag & KSW_EZ_SPLICE_FOR) && target[t+1] == 2 && target[t+2] == 3) can_type = 1; // GTr...
|
||||||
if (target[t+1] == 2 && target[t+2] == 3) // |GT.
|
if ((flag & KSW_EZ_SPLICE_REV) && target[t+1] == 1 && target[t+2] == 3) can_type = 1; // CTr...
|
||||||
z = target[t+3] == 0 || target[t+3] == 2? -1 : 0; // |GTr or not
|
if (can_type && (target[t+3] == 0 || target[t+3] == 2)) can_type = 2;
|
||||||
else if (target[t+1] == 2 && target[t+2] == 1) z = 1; // |GC.
|
if (can_type) ((int8_t*)donor)[t] = can_type == 2? 0 : semi_cost;
|
||||||
else if (target[t+1] == 0 && target[t+2] == 3) z = 2; // |AT.
|
|
||||||
} else if (flag & KSW_EZ_SPLICE_REV) {
|
|
||||||
if (target[t+1] == 1 && target[t+2] == 3) // |CT. (revcomp of .AG|)
|
|
||||||
z = target[t+3] == 0 || target[t+3] == 2? -1 : 0;
|
|
||||||
else if (target[t+1] == 2 && target[t+2] == 3) z = 2; // |GT. (revcomp of .AC|)
|
|
||||||
}
|
|
||||||
((int8_t*)donor)[t] = z < 0? 0 : -sp[z];
|
|
||||||
}
|
}
|
||||||
|
if (junc)
|
||||||
|
for (t = 0; t < tlen - 1; ++t)
|
||||||
|
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&8)))
|
||||||
|
((int8_t*)donor)[t] += junc_bonus;
|
||||||
for (t = 2; t < tlen; ++t) {
|
for (t = 2; t < tlen; ++t) {
|
||||||
int z = 3;
|
int can_type = 0;
|
||||||
if (flag & KSW_EZ_SPLICE_FOR) {
|
if ((flag & KSW_EZ_SPLICE_FOR) && target[t-1] == 0 && target[t] == 2) can_type = 1; // ...yAG
|
||||||
if (target[t-1] == 0 && target[t] == 2) // .AG|
|
if ((flag & KSW_EZ_SPLICE_REV) && target[t-1] == 0 && target[t] == 1) can_type = 1; // ...yAC
|
||||||
z = target[t-2] == 1 || target[t-2] == 3? -1 : 0; // yAG| or not
|
if (can_type && (target[t-2] == 1 || target[t-2] == 3)) can_type = 2;
|
||||||
else if (target[t-1] == 0 && target[t] == 1) z = 2; // .AC|
|
if (can_type) ((int8_t*)acceptor)[t] = can_type == 2? 0 : semi_cost;
|
||||||
} else if (flag & KSW_EZ_SPLICE_REV) {
|
|
||||||
if (target[t-1] == 0 && target[t] == 1) // .AC| (revcomp of |GT.)
|
|
||||||
z = target[t-2] == 1 || target[t-2] == 3? -1 : 0; // yAC| or not
|
|
||||||
else if (target[t-1] == 2 && target[t] == 1) z = 1; // .GC| (revcomp of |GC.)
|
|
||||||
else if (target[t-1] == 0 && target[t] == 3) z = 2; // .AT| (revcomp of |AT.)
|
|
||||||
}
|
|
||||||
((int8_t*)acceptor)[t] = z < 0? 0 : -sp[z];
|
|
||||||
}
|
}
|
||||||
|
if (junc)
|
||||||
|
for (t = 0; t < tlen; ++t)
|
||||||
|
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&4)))
|
||||||
|
((int8_t*)acceptor)[t] += junc_bonus;
|
||||||
} else {
|
} else {
|
||||||
for (t = 0; t < tlen - 4; ++t) {
|
for (t = 0; t < tlen - 4; ++t) {
|
||||||
int z = 3;
|
int can_type = 0; // type of canonical site: 0=none, 1=GT/AG only, 2=GTr/yAG
|
||||||
if (flag & KSW_EZ_SPLICE_FOR) {
|
if ((flag & KSW_EZ_SPLICE_FOR) && target[t+1] == 2 && target[t+2] == 0) can_type = 1; // GAy...
|
||||||
if (target[t+1] == 2 && target[t+2] == 0) // |GA. (rev of .AG|)
|
if ((flag & KSW_EZ_SPLICE_REV) && target[t+1] == 1 && target[t+2] == 0) can_type = 1; // CAy...
|
||||||
z = target[t+3] == 1 || target[t+3] == 3? -1 : 0;
|
if (can_type && (target[t+3] == 1 || target[t+3] == 3)) can_type = 2;
|
||||||
else if (target[t+1] == 1 && target[t+2] == 0) z = 2; // |CA. (rev of .AC|)
|
if (can_type) ((int8_t*)donor)[t] = can_type == 2? 0 : semi_cost;
|
||||||
} else if (flag & KSW_EZ_SPLICE_REV) {
|
|
||||||
if (target[t+1] == 1 && target[t+2] == 0) // |CA. (comp of |GT.)
|
|
||||||
z = target[t+3] == 1 || target[t+3] == 3? -1 : 0;
|
|
||||||
else if (target[t+1] == 1 && target[t+2] == 2) z = 1; // |CG. (comp of |GC.)
|
|
||||||
else if (target[t+1] == 3 && target[t+2] == 0) z = 2; // |TA. (comp of |AT.)
|
|
||||||
}
|
|
||||||
((int8_t*)donor)[t] = z < 0? 0 : -sp[z];
|
|
||||||
}
|
}
|
||||||
|
if (junc)
|
||||||
|
for (t = 0; t < tlen - 1; ++t)
|
||||||
|
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&4)))
|
||||||
|
((int8_t*)donor)[t] += junc_bonus;
|
||||||
for (t = 2; t < tlen; ++t) {
|
for (t = 2; t < tlen; ++t) {
|
||||||
int z = 3;
|
int can_type = 0;
|
||||||
if (flag & KSW_EZ_SPLICE_FOR) {
|
if ((flag & KSW_EZ_SPLICE_FOR) && target[t-1] == 3 && target[t] == 2) can_type = 1; // ...rTG
|
||||||
if (target[t-1] == 3 && target[t] == 2) // .TG| (rev of |GT.)
|
if ((flag & KSW_EZ_SPLICE_REV) && target[t-1] == 3 && target[t] == 1) can_type = 1; // ...rTC
|
||||||
z = target[t-2] == 0 || target[t-2] == 2? -1 : 0;
|
if (can_type && (target[t-2] == 0 || target[t-2] == 2)) can_type = 2;
|
||||||
else if (target[t-1] == 1 && target[t] == 2) z = 1; // .CG| (rev of |GC.)
|
if (can_type) ((int8_t*)acceptor)[t] = can_type == 2? 0 : semi_cost;
|
||||||
else if (target[t-1] == 3 && target[t] == 0) z = 2; // .TA| (rev of |AT.)
|
|
||||||
} else if (flag & KSW_EZ_SPLICE_REV) {
|
|
||||||
if (target[t-1] == 3 && target[t] == 1) // .TC| (comp of .AG|)
|
|
||||||
z = target[t-2] == 0 || target[t-2] == 2? -1 : 0;
|
|
||||||
else if (target[t-1] == 3 && target[t] == 2) z = 2; // .TG| (comp of .AC|)
|
|
||||||
}
|
|
||||||
((int8_t*)acceptor)[t] = z < 0? 0 : -sp[z];
|
|
||||||
}
|
}
|
||||||
}
|
if (junc)
|
||||||
}
|
for (t = 0; t < tlen; ++t)
|
||||||
|
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&8)))
|
||||||
if (junc) {
|
((int8_t*)acceptor)[t] += junc_bonus;
|
||||||
if (!(flag & KSW_EZ_REV_CIGAR)) {
|
|
||||||
for (t = 0; t < tlen - 1; ++t)
|
|
||||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&8)))
|
|
||||||
((int8_t*)donor)[t] += junc_bonus;
|
|
||||||
for (t = 0; t < tlen; ++t)
|
|
||||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&4)))
|
|
||||||
((int8_t*)acceptor)[t] += junc_bonus;
|
|
||||||
} else {
|
|
||||||
for (t = 0; t < tlen - 1; ++t)
|
|
||||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&4)))
|
|
||||||
((int8_t*)donor)[t] += junc_bonus;
|
|
||||||
for (t = 0; t < tlen; ++t)
|
|
||||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&8)))
|
|
||||||
((int8_t*)acceptor)[t] += junc_bonus;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -415,7 +376,7 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||||
// update ez
|
// update ez
|
||||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||||
ez->mte = H[en0], ez->mte_q = r - en0;
|
ez->mte = H[en0], ez->mte_q = r - en;
|
||||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, 0)) break;
|
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, 0)) break;
|
||||||
|
|||||||
+1
-1
@@ -269,7 +269,7 @@ void ksw_extz2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
} else H[0] = v8[0] - qe - qe, max_H = H[0], max_t = 0; // special casing r==0
|
} else H[0] = v8[0] - qe - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||||
// update ez
|
// update ez
|
||||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||||
ez->mte = H[en0], ez->mte_q = r - en0;
|
ez->mte = H[en0], ez->mte_q = r - en;
|
||||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e)) break;
|
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e)) break;
|
||||||
|
|||||||
@@ -1,368 +0,0 @@
|
|||||||
#include <stdint.h>
|
|
||||||
#include <string.h>
|
|
||||||
#include <stdio.h>
|
|
||||||
#include <assert.h>
|
|
||||||
#include "mmpriv.h"
|
|
||||||
#include "kalloc.h"
|
|
||||||
#include "krmq.h"
|
|
||||||
|
|
||||||
static int64_t mg_chain_bk_end(int32_t max_drop, const mm128_t *z, const int32_t *f, const int64_t *p, int32_t *t, int64_t k)
|
|
||||||
{
|
|
||||||
int64_t i = z[k].y, end_i = -1, max_i = i;
|
|
||||||
int32_t max_s = 0;
|
|
||||||
if (i < 0 || t[i] != 0) return i;
|
|
||||||
do {
|
|
||||||
int32_t s;
|
|
||||||
t[i] = 2;
|
|
||||||
end_i = i = p[i];
|
|
||||||
s = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
|
||||||
if (s > max_s) max_s = s, max_i = i;
|
|
||||||
else if (max_s - s > max_drop) break;
|
|
||||||
} while (i >= 0 && t[i] == 0);
|
|
||||||
for (i = z[k].y; i >= 0 && i != end_i; i = p[i]) // reset modified t[]
|
|
||||||
t[i] = 0;
|
|
||||||
return max_i;
|
|
||||||
}
|
|
||||||
|
|
||||||
uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_t *p, int32_t *v, int32_t *t, int32_t min_cnt, int32_t min_sc, int32_t max_drop, int32_t *n_u_, int32_t *n_v_)
|
|
||||||
{
|
|
||||||
mm128_t *z;
|
|
||||||
uint64_t *u;
|
|
||||||
int64_t i, k, n_z, n_v;
|
|
||||||
int32_t n_u;
|
|
||||||
|
|
||||||
*n_u_ = *n_v_ = 0;
|
|
||||||
for (i = 0, n_z = 0; i < n; ++i) // precompute n_z
|
|
||||||
if (f[i] >= min_sc) ++n_z;
|
|
||||||
if (n_z == 0) return 0;
|
|
||||||
z = Kmalloc(km, mm128_t, n_z);
|
|
||||||
for (i = 0, k = 0; i < n; ++i) // populate z[]
|
|
||||||
if (f[i] >= min_sc) z[k].x = f[i], z[k++].y = i;
|
|
||||||
radix_sort_128x(z, z + n_z);
|
|
||||||
|
|
||||||
memset(t, 0, n * 4);
|
|
||||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // precompute n_u
|
|
||||||
if (t[z[k].y] == 0) {
|
|
||||||
int64_t n_v0 = n_v, end_i;
|
|
||||||
int32_t sc;
|
|
||||||
end_i = mg_chain_bk_end(max_drop, z, f, p, t, k);
|
|
||||||
for (i = z[k].y; i != end_i; i = p[i])
|
|
||||||
++n_v, t[i] = 1;
|
|
||||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
|
||||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
|
||||||
++n_u;
|
|
||||||
else n_v = n_v0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
u = Kmalloc(km, uint64_t, n_u);
|
|
||||||
memset(t, 0, n * 4);
|
|
||||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // populate u[]
|
|
||||||
if (t[z[k].y] == 0) {
|
|
||||||
int64_t n_v0 = n_v, end_i;
|
|
||||||
int32_t sc;
|
|
||||||
end_i = mg_chain_bk_end(max_drop, z, f, p, t, k);
|
|
||||||
for (i = z[k].y; i != end_i; i = p[i])
|
|
||||||
v[n_v++] = i, t[i] = 1;
|
|
||||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
|
||||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
|
||||||
u[n_u++] = (uint64_t)sc << 32 | (n_v - n_v0);
|
|
||||||
else n_v = n_v0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
kfree(km, z);
|
|
||||||
assert(n_v < INT32_MAX);
|
|
||||||
*n_u_ = n_u, *n_v_ = n_v;
|
|
||||||
return u;
|
|
||||||
}
|
|
||||||
|
|
||||||
static mm128_t *compact_a(void *km, int32_t n_u, uint64_t *u, int32_t n_v, int32_t *v, mm128_t *a)
|
|
||||||
{
|
|
||||||
mm128_t *b, *w;
|
|
||||||
uint64_t *u2;
|
|
||||||
int64_t i, j, k;
|
|
||||||
|
|
||||||
// write the result to b[]
|
|
||||||
b = Kmalloc(km, mm128_t, n_v);
|
|
||||||
for (i = 0, k = 0; i < n_u; ++i) {
|
|
||||||
int32_t k0 = k, ni = (int32_t)u[i];
|
|
||||||
for (j = 0; j < ni; ++j)
|
|
||||||
b[k++] = a[v[k0 + (ni - j - 1)]];
|
|
||||||
}
|
|
||||||
kfree(km, v);
|
|
||||||
|
|
||||||
// sort u[] and a[] by the target position, such that adjacent chains may be joined
|
|
||||||
w = Kmalloc(km, mm128_t, n_u);
|
|
||||||
for (i = k = 0; i < n_u; ++i) {
|
|
||||||
w[i].x = b[k].x, w[i].y = (uint64_t)k<<32|i;
|
|
||||||
k += (int32_t)u[i];
|
|
||||||
}
|
|
||||||
radix_sort_128x(w, w + n_u);
|
|
||||||
u2 = Kmalloc(km, uint64_t, n_u);
|
|
||||||
for (i = k = 0; i < n_u; ++i) {
|
|
||||||
int32_t j = (int32_t)w[i].y, n = (int32_t)u[j];
|
|
||||||
u2[i] = u[j];
|
|
||||||
memcpy(&a[k], &b[w[i].y>>32], n * sizeof(mm128_t));
|
|
||||||
k += n;
|
|
||||||
}
|
|
||||||
memcpy(u, u2, n_u * 8);
|
|
||||||
memcpy(b, a, k * sizeof(mm128_t)); // write _a_ to _b_ and deallocate _a_ because _a_ is oversized, sometimes a lot
|
|
||||||
kfree(km, a); kfree(km, w); kfree(km, u2);
|
|
||||||
return b;
|
|
||||||
}
|
|
||||||
|
|
||||||
static inline int32_t comput_sc(const mm128_t *ai, const mm128_t *aj, int32_t max_dist_x, int32_t max_dist_y, int32_t bw, float chn_pen_gap, float chn_pen_skip, int is_cdna, int n_seg)
|
|
||||||
{
|
|
||||||
int32_t dq = (int32_t)ai->y - (int32_t)aj->y, dr, dd, dg, q_span, sc;
|
|
||||||
int32_t sidi = (ai->y & MM_SEED_SEG_MASK) >> MM_SEED_SEG_SHIFT;
|
|
||||||
int32_t sidj = (aj->y & MM_SEED_SEG_MASK) >> MM_SEED_SEG_SHIFT;
|
|
||||||
if (dq <= 0 || dq > max_dist_x) return INT32_MIN;
|
|
||||||
dr = (int32_t)(ai->x - aj->x);
|
|
||||||
if (sidi == sidj && (dr == 0 || dq > max_dist_y)) return INT32_MIN;
|
|
||||||
dd = dr > dq? dr - dq : dq - dr;
|
|
||||||
if (sidi == sidj && dd > bw) return INT32_MIN;
|
|
||||||
if (n_seg > 1 && !is_cdna && sidi == sidj && dr > max_dist_y) return INT32_MIN;
|
|
||||||
dg = dr < dq? dr : dq;
|
|
||||||
q_span = aj->y>>32&0xff;
|
|
||||||
sc = q_span < dg? q_span : dg;
|
|
||||||
if (dd || dg > q_span) {
|
|
||||||
float lin_pen, log_pen;
|
|
||||||
lin_pen = chn_pen_gap * (float)dd + chn_pen_skip * (float)dg;
|
|
||||||
log_pen = dd >= 1? mg_log2(dd + 1) : 0.0f; // mg_log2() only works for dd>=2
|
|
||||||
if (is_cdna || sidi != sidj) {
|
|
||||||
if (sidi != sidj && dr == 0) ++sc; // possibly due to overlapping paired ends; give a minor bonus
|
|
||||||
else if (dr > dq || sidi != sidj) sc -= (int)(lin_pen < log_pen? lin_pen : log_pen); // deletion or jump between paired ends
|
|
||||||
else sc -= (int)(lin_pen + .5f * log_pen);
|
|
||||||
} else sc -= (int)(lin_pen + .5f * log_pen);
|
|
||||||
}
|
|
||||||
return sc;
|
|
||||||
}
|
|
||||||
|
|
||||||
/* Input:
|
|
||||||
* a[].x: rev<<63 | tid<<32 | tpos
|
|
||||||
* a[].y: flags<<40 | q_span<<32 | q_pos
|
|
||||||
* Output:
|
|
||||||
* n_u: #chains
|
|
||||||
* u[]: score<<32 | #anchors (sum of lower 32 bits of u[] is the returned length of a[])
|
|
||||||
* input a[] is deallocated on return
|
|
||||||
*/
|
|
||||||
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
|
||||||
int is_cdna, int n_seg, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
|
||||||
{ // TODO: make sure this works when n has more than 32 bits
|
|
||||||
int32_t *f, *t, *v, n_u, n_v, mmax_f = 0, max_drop = bw;
|
|
||||||
int64_t *p, i, j, max_ii, st = 0, n_iter = 0;
|
|
||||||
uint64_t *u;
|
|
||||||
|
|
||||||
if (_u) *_u = 0, *n_u_ = 0;
|
|
||||||
if (n == 0 || a == 0) {
|
|
||||||
kfree(km, a);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
if (max_dist_x < bw) max_dist_x = bw;
|
|
||||||
if (max_dist_y < bw && !is_cdna) max_dist_y = bw;
|
|
||||||
if (is_cdna) max_drop = INT32_MAX;
|
|
||||||
p = Kmalloc(km, int64_t, n);
|
|
||||||
f = Kmalloc(km, int32_t, n);
|
|
||||||
v = Kmalloc(km, int32_t, n);
|
|
||||||
t = Kcalloc(km, int32_t, n);
|
|
||||||
|
|
||||||
// fill the score and backtrack arrays
|
|
||||||
for (i = 0, max_ii = -1; i < n; ++i) {
|
|
||||||
int64_t max_j = -1, end_j;
|
|
||||||
int32_t max_f = a[i].y>>32&0xff, n_skip = 0;
|
|
||||||
while (st < i && (a[i].x>>32 != a[st].x>>32 || a[i].x > a[st].x + max_dist_x)) ++st;
|
|
||||||
if (i - st > max_iter) st = i - max_iter;
|
|
||||||
for (j = i - 1; j >= st; --j) {
|
|
||||||
int32_t sc;
|
|
||||||
sc = comput_sc(&a[i], &a[j], max_dist_x, max_dist_y, bw, chn_pen_gap, chn_pen_skip, is_cdna, n_seg);
|
|
||||||
++n_iter;
|
|
||||||
if (sc == INT32_MIN) continue;
|
|
||||||
sc += f[j];
|
|
||||||
if (sc > max_f) {
|
|
||||||
max_f = sc, max_j = j;
|
|
||||||
if (n_skip > 0) --n_skip;
|
|
||||||
} else if (t[j] == (int32_t)i) {
|
|
||||||
if (++n_skip > max_skip)
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
if (p[j] >= 0) t[p[j]] = i;
|
|
||||||
}
|
|
||||||
end_j = j;
|
|
||||||
if (max_ii < 0 || a[i].x - a[max_ii].x > (int64_t)max_dist_x) {
|
|
||||||
int32_t max = INT32_MIN;
|
|
||||||
max_ii = -1;
|
|
||||||
for (j = i - 1; j >= st; --j)
|
|
||||||
if (max < f[j]) max = f[j], max_ii = j;
|
|
||||||
}
|
|
||||||
if (max_ii >= 0 && max_ii < end_j) {
|
|
||||||
int32_t tmp;
|
|
||||||
tmp = comput_sc(&a[i], &a[max_ii], max_dist_x, max_dist_y, bw, chn_pen_gap, chn_pen_skip, is_cdna, n_seg);
|
|
||||||
if (tmp != INT32_MIN && max_f < tmp + f[max_ii])
|
|
||||||
max_f = tmp + f[max_ii], max_j = max_ii;
|
|
||||||
}
|
|
||||||
f[i] = max_f, p[i] = max_j;
|
|
||||||
v[i] = max_j >= 0 && v[max_j] > max_f? v[max_j] : max_f; // v[] keeps the peak score up to i; f[] is the score ending at i, not always the peak
|
|
||||||
if (max_ii < 0 || (a[i].x - a[max_ii].x <= (int64_t)max_dist_x && f[max_ii] < f[i]))
|
|
||||||
max_ii = i;
|
|
||||||
if (mmax_f < max_f) mmax_f = max_f;
|
|
||||||
}
|
|
||||||
|
|
||||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, max_drop, &n_u, &n_v);
|
|
||||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
|
||||||
kfree(km, p); kfree(km, f); kfree(km, t);
|
|
||||||
if (n_u == 0) {
|
|
||||||
kfree(km, a); kfree(km, v);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
return compact_a(km, n_u, u, n_v, v, a);
|
|
||||||
}
|
|
||||||
|
|
||||||
typedef struct lc_elem_s {
|
|
||||||
int32_t y;
|
|
||||||
int64_t i;
|
|
||||||
double pri;
|
|
||||||
KRMQ_HEAD(struct lc_elem_s) head;
|
|
||||||
} lc_elem_t;
|
|
||||||
|
|
||||||
#define lc_elem_cmp(a, b) ((a)->y < (b)->y? -1 : (a)->y > (b)->y? 1 : ((a)->i > (b)->i) - ((a)->i < (b)->i))
|
|
||||||
#define lc_elem_lt2(a, b) ((a)->pri < (b)->pri)
|
|
||||||
KRMQ_INIT(lc_elem, lc_elem_t, head, lc_elem_cmp, lc_elem_lt2)
|
|
||||||
|
|
||||||
KALLOC_POOL_INIT(rmq, lc_elem_t)
|
|
||||||
|
|
||||||
static inline int32_t comput_sc_simple(const mm128_t *ai, const mm128_t *aj, float chn_pen_gap, float chn_pen_skip, int32_t *exact, int32_t *width)
|
|
||||||
{
|
|
||||||
int32_t dq = (int32_t)ai->y - (int32_t)aj->y, dr, dd, dg, q_span, sc;
|
|
||||||
dr = (int32_t)(ai->x - aj->x);
|
|
||||||
*width = dd = dr > dq? dr - dq : dq - dr;
|
|
||||||
dg = dr < dq? dr : dq;
|
|
||||||
q_span = aj->y>>32&0xff;
|
|
||||||
sc = q_span < dg? q_span : dg;
|
|
||||||
if (exact) *exact = (dd == 0 && dg <= q_span);
|
|
||||||
if (dd || dq > q_span) {
|
|
||||||
float lin_pen, log_pen;
|
|
||||||
lin_pen = chn_pen_gap * (float)dd + chn_pen_skip * (float)dg;
|
|
||||||
log_pen = dd >= 1? mg_log2(dd + 1) : 0.0f; // mg_log2() only works for dd>=2
|
|
||||||
sc -= (int)(lin_pen + .5f * log_pen);
|
|
||||||
}
|
|
||||||
return sc;
|
|
||||||
}
|
|
||||||
|
|
||||||
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
|
||||||
int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
|
||||||
{
|
|
||||||
int32_t *f,*t, *v, n_u, n_v, mmax_f = 0, max_rmq_size = 0, max_drop = bw;
|
|
||||||
int64_t *p, i, i0, st = 0, st_inner = 0;
|
|
||||||
uint64_t *u;
|
|
||||||
lc_elem_t *root = 0, *root_inner = 0;
|
|
||||||
void *mem_mp = 0;
|
|
||||||
kmp_rmq_t *mp;
|
|
||||||
|
|
||||||
if (_u) *_u = 0, *n_u_ = 0;
|
|
||||||
if (n == 0 || a == 0) {
|
|
||||||
kfree(km, a);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
if (max_dist < bw) max_dist = bw;
|
|
||||||
if (max_dist_inner <= 0 || max_dist_inner >= max_dist) max_dist_inner = 0;
|
|
||||||
p = Kmalloc(km, int64_t, n);
|
|
||||||
f = Kmalloc(km, int32_t, n);
|
|
||||||
t = Kcalloc(km, int32_t, n);
|
|
||||||
v = Kmalloc(km, int32_t, n);
|
|
||||||
mem_mp = km_init2(km, 0x10000);
|
|
||||||
mp = kmp_init_rmq(mem_mp);
|
|
||||||
|
|
||||||
// fill the score and backtrack arrays
|
|
||||||
for (i = i0 = 0; i < n; ++i) {
|
|
||||||
int64_t max_j = -1;
|
|
||||||
int32_t q_span = a[i].y>>32&0xff, max_f = q_span;
|
|
||||||
lc_elem_t s, *q, *r, lo, hi;
|
|
||||||
// add in-range anchors
|
|
||||||
if (i0 < i && a[i0].x != a[i].x) {
|
|
||||||
int64_t j;
|
|
||||||
for (j = i0; j < i; ++j) {
|
|
||||||
q = kmp_alloc_rmq(mp);
|
|
||||||
q->y = (int32_t)a[j].y, q->i = j, q->pri = -(f[j] + 0.5 * chn_pen_gap * ((int32_t)a[j].x + (int32_t)a[j].y));
|
|
||||||
krmq_insert(lc_elem, &root, q, 0);
|
|
||||||
if (max_dist_inner > 0) {
|
|
||||||
r = kmp_alloc_rmq(mp);
|
|
||||||
*r = *q;
|
|
||||||
krmq_insert(lc_elem, &root_inner, r, 0);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
i0 = i;
|
|
||||||
}
|
|
||||||
// get rid of active chains out of range
|
|
||||||
while (st < i && (a[i].x>>32 != a[st].x>>32 || a[i].x > a[st].x + max_dist || krmq_size(head, root) > cap_rmq_size)) {
|
|
||||||
s.y = (int32_t)a[st].y, s.i = st;
|
|
||||||
if ((q = krmq_find(lc_elem, root, &s, 0)) != 0) {
|
|
||||||
q = krmq_erase(lc_elem, &root, q, 0);
|
|
||||||
kmp_free_rmq(mp, q);
|
|
||||||
}
|
|
||||||
++st;
|
|
||||||
}
|
|
||||||
if (max_dist_inner > 0) { // similar to the block above, but applied to the inner tree
|
|
||||||
while (st_inner < i && (a[i].x>>32 != a[st_inner].x>>32 || a[i].x > a[st_inner].x + max_dist_inner || krmq_size(head, root_inner) > cap_rmq_size)) {
|
|
||||||
s.y = (int32_t)a[st_inner].y, s.i = st_inner;
|
|
||||||
if ((q = krmq_find(lc_elem, root_inner, &s, 0)) != 0) {
|
|
||||||
q = krmq_erase(lc_elem, &root_inner, q, 0);
|
|
||||||
kmp_free_rmq(mp, q);
|
|
||||||
}
|
|
||||||
++st_inner;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// RMQ
|
|
||||||
lo.i = INT32_MAX, lo.y = (int32_t)a[i].y - max_dist;
|
|
||||||
hi.i = 0, hi.y = (int32_t)a[i].y;
|
|
||||||
if ((q = krmq_rmq(lc_elem, root, &lo, &hi)) != 0) {
|
|
||||||
int32_t sc, exact, width, n_skip = 0;
|
|
||||||
int64_t j = q->i;
|
|
||||||
assert(q->y >= lo.y && q->y <= hi.y);
|
|
||||||
sc = f[j] + comput_sc_simple(&a[i], &a[j], chn_pen_gap, chn_pen_skip, &exact, &width);
|
|
||||||
if (width <= bw && sc > max_f) max_f = sc, max_j = j;
|
|
||||||
if (!exact && root_inner && (int32_t)a[i].y > 0) {
|
|
||||||
lc_elem_t *lo, *hi;
|
|
||||||
s.y = (int32_t)a[i].y - 1, s.i = n;
|
|
||||||
krmq_interval(lc_elem, root_inner, &s, &lo, &hi);
|
|
||||||
if (lo) {
|
|
||||||
const lc_elem_t *q;
|
|
||||||
int32_t width, n_rmq_iter = 0;
|
|
||||||
krmq_itr_t(lc_elem) itr;
|
|
||||||
krmq_itr_find(lc_elem, root_inner, lo, &itr);
|
|
||||||
while ((q = krmq_at(&itr)) != 0) {
|
|
||||||
if (q->y < (int32_t)a[i].y - max_dist_inner) break;
|
|
||||||
++n_rmq_iter;
|
|
||||||
j = q->i;
|
|
||||||
sc = f[j] + comput_sc_simple(&a[i], &a[j], chn_pen_gap, chn_pen_skip, 0, &width);
|
|
||||||
if (width <= bw) {
|
|
||||||
if (sc > max_f) {
|
|
||||||
max_f = sc, max_j = j;
|
|
||||||
if (n_skip > 0) --n_skip;
|
|
||||||
} else if (t[j] == (int32_t)i) {
|
|
||||||
if (++n_skip > max_chn_skip)
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
if (p[j] >= 0) t[p[j]] = i;
|
|
||||||
}
|
|
||||||
if (!krmq_itr_prev(lc_elem, &itr)) break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// set max
|
|
||||||
assert(max_j < 0 || (a[max_j].x < a[i].x && (int32_t)a[max_j].y < (int32_t)a[i].y));
|
|
||||||
f[i] = max_f, p[i] = max_j;
|
|
||||||
v[i] = max_j >= 0 && v[max_j] > max_f? v[max_j] : max_f; // v[] keeps the peak score up to i; f[] is the score ending at i, not always the peak
|
|
||||||
if (mmax_f < max_f) mmax_f = max_f;
|
|
||||||
if (max_rmq_size < krmq_size(head, root)) max_rmq_size = krmq_size(head, root);
|
|
||||||
}
|
|
||||||
km_destroy(mem_mp);
|
|
||||||
|
|
||||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, max_drop, &n_u, &n_v);
|
|
||||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
|
||||||
kfree(km, p); kfree(km, f); kfree(km, t);
|
|
||||||
if (n_u == 0) {
|
|
||||||
kfree(km, a); kfree(km, v);
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
return compact_a(km, n_u, u, n_v, v, a);
|
|
||||||
}
|
|
||||||
@@ -1,11 +1,54 @@
|
|||||||
|
/* The MIT License
|
||||||
|
|
||||||
|
Copyright (c) 2018- Dana-Farber Cancer Institute
|
||||||
|
2017-2018 Broad Institute, Inc.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
|
Modified Copyright (C) 2021 Intel Corporation
|
||||||
|
Contacts: Saurabh Kalikar <saurabh.kalikar@intel.com>;
|
||||||
|
Vasimuddin Md <vasimuddin.md@intel.com>; Sanchit Misra <sanchit.misra@intel.com>;
|
||||||
|
Chirag Jain <chirag@iisc.ac.in>; Heng Li <hli@jimmy.harvard.edu>
|
||||||
|
*/
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
|
#include <string>
|
||||||
#include <errno.h>
|
#include <errno.h>
|
||||||
#include "bseq.h"
|
#include "bseq.h"
|
||||||
#include "minimap.h"
|
#include "minimap.h"
|
||||||
#include "mmpriv.h"
|
#include "mmpriv.h"
|
||||||
#include "ketopt.h"
|
#include "ketopt.h"
|
||||||
|
#include <x86intrin.h>
|
||||||
|
|
||||||
|
#define MM_VERSION "2.18-r1015"
|
||||||
|
using namespace std;
|
||||||
|
|
||||||
|
#ifdef MANUAL_PROFILING
|
||||||
|
uint64_t num_reads = 0, minimizer_hit_time = 0, dp_chaining_time = 0, alignment_time = 0;
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#ifdef LISA_HASH
|
||||||
|
#include "lisa_hash.h"
|
||||||
|
lisa_hash<uint64_t, uint64_t> *lh;
|
||||||
|
#endif
|
||||||
|
|
||||||
#ifdef __linux__
|
#ifdef __linux__
|
||||||
#include <sys/resource.h>
|
#include <sys/resource.h>
|
||||||
@@ -69,14 +112,6 @@ static ko_longopt_t long_options[] = {
|
|||||||
{ "alt", ko_required_argument, 344 },
|
{ "alt", ko_required_argument, 344 },
|
||||||
{ "alt-drop", ko_required_argument, 345 },
|
{ "alt-drop", ko_required_argument, 345 },
|
||||||
{ "mask-len", ko_required_argument, 346 },
|
{ "mask-len", ko_required_argument, 346 },
|
||||||
{ "rmq", ko_optional_argument, 347 },
|
|
||||||
{ "qstrand", ko_no_argument, 348 },
|
|
||||||
{ "cap-kalloc", ko_required_argument, 349 },
|
|
||||||
{ "q-occ-frac", ko_required_argument, 350 },
|
|
||||||
{ "chain-skip-scale",ko_required_argument,351 },
|
|
||||||
{ "print-chains", ko_no_argument, 352 },
|
|
||||||
{ "no-hash-name", ko_no_argument, 353 },
|
|
||||||
{ "secondary-seq", ko_no_argument, 354 },
|
|
||||||
{ "help", ko_no_argument, 'h' },
|
{ "help", ko_no_argument, 'h' },
|
||||||
{ "max-intron-len", ko_required_argument, 'G' },
|
{ "max-intron-len", ko_required_argument, 'G' },
|
||||||
{ "version", ko_no_argument, 'V' },
|
{ "version", ko_no_argument, 'V' },
|
||||||
@@ -88,24 +123,18 @@ static ko_longopt_t long_options[] = {
|
|||||||
{ 0, 0, 0 }
|
{ 0, 0, 0 }
|
||||||
};
|
};
|
||||||
|
|
||||||
static inline int64_t mm_parse_num2(const char *str, char **q)
|
static inline int64_t mm_parse_num(const char *str)
|
||||||
{
|
{
|
||||||
double x;
|
double x;
|
||||||
char *p;
|
char *p;
|
||||||
x = strtod(str, &p);
|
x = strtod(str, &p);
|
||||||
if (*p == 'G' || *p == 'g') x *= 1e9, ++p;
|
if (*p == 'G' || *p == 'g') x *= 1e9;
|
||||||
else if (*p == 'M' || *p == 'm') x *= 1e6, ++p;
|
else if (*p == 'M' || *p == 'm') x *= 1e6;
|
||||||
else if (*p == 'K' || *p == 'k') x *= 1e3, ++p;
|
else if (*p == 'K' || *p == 'k') x *= 1e3;
|
||||||
if (q) *q = p;
|
|
||||||
return (int64_t)(x + .499);
|
return (int64_t)(x + .499);
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline int64_t mm_parse_num(const char *str)
|
static inline void yes_or_no(mm_mapopt_t *opt, int flag, int long_idx, const char *arg, int yes_to_set)
|
||||||
{
|
|
||||||
return mm_parse_num2(str, 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
static inline void yes_or_no(mm_mapopt_t *opt, int64_t flag, int long_idx, const char *arg, int yes_to_set)
|
|
||||||
{
|
{
|
||||||
if (yes_to_set) {
|
if (yes_to_set) {
|
||||||
if (strcmp(arg, "yes") == 0 || strcmp(arg, "y") == 0) opt->flag |= flag;
|
if (strcmp(arg, "yes") == 0 || strcmp(arg, "y") == 0) opt->flag |= flag;
|
||||||
@@ -120,11 +149,39 @@ static inline void yes_or_no(mm_mapopt_t *opt, int64_t flag, int long_idx, const
|
|||||||
|
|
||||||
int main(int argc, char *argv[])
|
int main(int argc, char *argv[])
|
||||||
{
|
{
|
||||||
const char *opt_str = "2aSDw:k:K:t:r:f:Vv:g:G:I:d:XT:s:x:Hcp:M:n:z:A:B:O:E:m:N:Qu:R:hF:LC:yYPo:e:U:J:";
|
#ifdef LISA_HASH
|
||||||
|
#if VECTORIZE && __AVX512BW__
|
||||||
|
fprintf(stderr, "Using LISA hash with AVX512-vectorized last-mile search.\n");
|
||||||
|
#else
|
||||||
|
fprintf(stderr, "Using LISA hash with sequential last-mile search.\n");
|
||||||
|
#endif
|
||||||
|
#else
|
||||||
|
fprintf(stderr, "Using default hash lookup.\n");
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#if defined(VECTORIZED_CHAINING) && defined(__AVX512BW__)
|
||||||
|
fprintf(stderr, "Using AVX512-vectorized chaining.\n");
|
||||||
|
#else
|
||||||
|
fprintf(stderr, "Using default chaining.\n");
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
|
#if defined (ALIGN_AVX) && (defined(__AVX512BW__) || (defined(__AVX2__) && defined(APPLY_AVX2)))
|
||||||
|
#ifdef __AVX512BW__
|
||||||
|
fprintf(stderr, "Using AVX512-vectorized alignment.\n");
|
||||||
|
#elif __AVX2__
|
||||||
|
fprintf(stderr, "Using AVX2-vectorized alignment.\n");
|
||||||
|
#endif
|
||||||
|
#else
|
||||||
|
fprintf(stderr, "Using default SSE-vectorized alignment.\n");
|
||||||
|
#endif
|
||||||
|
|
||||||
|
const char *opt_str = "2aSDw:k:K:t:r:f:Vv:g:G:I:d:XT:s:x:Hcp:M:n:z:A:B:O:E:m:N:Qu:R:hF:LC:yYPo:Z:";
|
||||||
ketopt_t o = KETOPT_INIT;
|
ketopt_t o = KETOPT_INIT;
|
||||||
mm_mapopt_t opt;
|
mm_mapopt_t opt;
|
||||||
mm_idxopt_t ipt;
|
mm_idxopt_t ipt;
|
||||||
int i, c, n_threads = 3, n_parts, old_best_n = -1;
|
int i, c, n_threads = 3, n_parts, old_best_n = -1;
|
||||||
|
uint64_t total_time = 0;
|
||||||
char *fnw = 0, *rg = 0, *junc_bed = 0, *s, *alt_list = 0;
|
char *fnw = 0, *rg = 0, *junc_bed = 0, *s, *alt_list = 0;
|
||||||
FILE *fp_help = stderr;
|
FILE *fp_help = stderr;
|
||||||
mm_idx_reader_t *idx_rdr;
|
mm_idx_reader_t *idx_rdr;
|
||||||
@@ -133,10 +190,13 @@ int main(int argc, char *argv[])
|
|||||||
mm_verbose = 3;
|
mm_verbose = 3;
|
||||||
liftrlimit();
|
liftrlimit();
|
||||||
mm_realtime0 = realtime();
|
mm_realtime0 = realtime();
|
||||||
|
double mapping_time = realtime();
|
||||||
mm_set_opt(0, &ipt, &opt);
|
mm_set_opt(0, &ipt, &opt);
|
||||||
|
string preset_arg = "";
|
||||||
|
|
||||||
while ((c = ketopt(&o, argc, argv, 1, opt_str, long_options)) >= 0) { // test command line options and apply option -x/preset first
|
while ((c = ketopt(&o, argc, argv, 1, opt_str, long_options)) >= 0) { // test command line options and apply option -x/preset first
|
||||||
if (c == 'x') {
|
if (c == 'x') {
|
||||||
|
preset_arg += (string) o.arg;
|
||||||
if (mm_set_opt(o.arg, &ipt, &opt) < 0) {
|
if (mm_set_opt(o.arg, &ipt, &opt) < 0) {
|
||||||
fprintf(stderr, "[ERROR] unknown preset '%s'\n", o.arg);
|
fprintf(stderr, "[ERROR] unknown preset '%s'\n", o.arg);
|
||||||
return 1;
|
return 1;
|
||||||
@@ -153,9 +213,11 @@ int main(int argc, char *argv[])
|
|||||||
|
|
||||||
while ((c = ketopt(&o, argc, argv, 1, opt_str, long_options)) >= 0) {
|
while ((c = ketopt(&o, argc, argv, 1, opt_str, long_options)) >= 0) {
|
||||||
if (c == 'w') ipt.w = atoi(o.arg);
|
if (c == 'w') ipt.w = atoi(o.arg);
|
||||||
|
else if (c == 'Z') opt.L_hash = atoi(o.arg);
|
||||||
else if (c == 'k') ipt.k = atoi(o.arg);
|
else if (c == 'k') ipt.k = atoi(o.arg);
|
||||||
else if (c == 'H') ipt.flag |= MM_I_HPC;
|
else if (c == 'H') ipt.flag |= MM_I_HPC;
|
||||||
else if (c == 'd') fnw = o.arg; // the above are indexing related options, except -I
|
else if (c == 'd') fnw = o.arg; // the above are indexing related options, except -I
|
||||||
|
else if (c == 'r') opt.bw = (int)mm_parse_num(o.arg);
|
||||||
else if (c == 't') n_threads = atoi(o.arg);
|
else if (c == 't') n_threads = atoi(o.arg);
|
||||||
else if (c == 'v') mm_verbose = atoi(o.arg);
|
else if (c == 'v') mm_verbose = atoi(o.arg);
|
||||||
else if (c == 'g') opt.max_gap = (int)mm_parse_num(o.arg);
|
else if (c == 'g') opt.max_gap = (int)mm_parse_num(o.arg);
|
||||||
@@ -182,16 +244,10 @@ int main(int argc, char *argv[])
|
|||||||
else if (c == 'C') opt.noncan = atoi(o.arg);
|
else if (c == 'C') opt.noncan = atoi(o.arg);
|
||||||
else if (c == 'I') ipt.batch_size = mm_parse_num(o.arg);
|
else if (c == 'I') ipt.batch_size = mm_parse_num(o.arg);
|
||||||
else if (c == 'K') opt.mini_batch_size = mm_parse_num(o.arg);
|
else if (c == 'K') opt.mini_batch_size = mm_parse_num(o.arg);
|
||||||
else if (c == 'e') opt.occ_dist = mm_parse_num(o.arg);
|
|
||||||
else if (c == 'R') rg = o.arg;
|
else if (c == 'R') rg = o.arg;
|
||||||
else if (c == 'h') fp_help = stdout;
|
else if (c == 'h') fp_help = stdout;
|
||||||
else if (c == '2') opt.flag |= MM_F_2_IO_THREADS;
|
else if (c == '2') opt.flag |= MM_F_2_IO_THREADS;
|
||||||
else if (c == 'J') {
|
else if (c == 'o') {
|
||||||
int t;
|
|
||||||
t = atoi(o.arg);
|
|
||||||
if (t == 0) opt.flag |= MM_F_SPLICE_OLD;
|
|
||||||
else if (t == 1) opt.flag &= ~MM_F_SPLICE_OLD;
|
|
||||||
} else if (c == 'o') {
|
|
||||||
if (strcmp(o.arg, "-") != 0) {
|
if (strcmp(o.arg, "-") != 0) {
|
||||||
if (freopen(o.arg, "wb", stdout) == NULL) {
|
if (freopen(o.arg, "wb", stdout) == NULL) {
|
||||||
fprintf(stderr, "[ERROR]\033[1;31m failed to write the output to file '%s'\033[0m: %s\n", o.arg, strerror(errno));
|
fprintf(stderr, "[ERROR]\033[1;31m failed to write the output to file '%s'\033[0m: %s\n", o.arg, strerror(errno));
|
||||||
@@ -220,6 +276,7 @@ int main(int argc, char *argv[])
|
|||||||
else if (c == 327) opt.max_clip_ratio = atof(o.arg); // --max-clip-ratio
|
else if (c == 327) opt.max_clip_ratio = atof(o.arg); // --max-clip-ratio
|
||||||
else if (c == 328) opt.min_mid_occ = atoi(o.arg); // --min-occ-floor
|
else if (c == 328) opt.min_mid_occ = atoi(o.arg); // --min-occ-floor
|
||||||
else if (c == 329) opt.flag |= MM_F_OUT_MD; // --MD
|
else if (c == 329) opt.flag |= MM_F_OUT_MD; // --MD
|
||||||
|
else if (c == 330) opt.min_join_flank_ratio = atof(o.arg); // --lj-min-ratio
|
||||||
else if (c == 331) opt.sc_ambi = atoi(o.arg); // --score-N
|
else if (c == 331) opt.sc_ambi = atoi(o.arg); // --score-N
|
||||||
else if (c == 332) opt.flag |= MM_F_EQX; // --eqx
|
else if (c == 332) opt.flag |= MM_F_EQX; // --eqx
|
||||||
else if (c == 333) opt.flag |= MM_F_PAF_NO_HIT; // --paf-no-hit
|
else if (c == 333) opt.flag |= MM_F_PAF_NO_HIT; // --paf-no-hit
|
||||||
@@ -232,19 +289,10 @@ int main(int argc, char *argv[])
|
|||||||
else if (c == 341) opt.junc_bonus = atoi(o.arg); // --junc-bonus
|
else if (c == 341) opt.junc_bonus = atoi(o.arg); // --junc-bonus
|
||||||
else if (c == 342) opt.flag |= MM_F_SAM_HIT_ONLY; // --sam-hit-only
|
else if (c == 342) opt.flag |= MM_F_SAM_HIT_ONLY; // --sam-hit-only
|
||||||
else if (c == 343) opt.chain_gap_scale = atof(o.arg); // --chain-gap-scale
|
else if (c == 343) opt.chain_gap_scale = atof(o.arg); // --chain-gap-scale
|
||||||
else if (c == 351) opt.chain_skip_scale = atof(o.arg); // --chain-skip-scale
|
|
||||||
else if (c == 344) alt_list = o.arg; // --alt
|
else if (c == 344) alt_list = o.arg; // --alt
|
||||||
else if (c == 345) opt.alt_drop = atof(o.arg); // --alt-drop
|
else if (c == 345) opt.alt_drop = atof(o.arg); // --alt-drop
|
||||||
else if (c == 346) opt.mask_len = mm_parse_num(o.arg); // --mask-len
|
else if (c == 346) opt.mask_len = mm_parse_num(o.arg); // --mask-len
|
||||||
else if (c == 348) opt.flag |= MM_F_QSTRAND | MM_F_NO_INV; // --qstrand
|
else if (c == 314) { // --frag
|
||||||
else if (c == 349) opt.cap_kalloc = mm_parse_num(o.arg); // --cap-kalloc
|
|
||||||
else if (c == 350) opt.q_occ_frac = atof(o.arg); // --q-occ-frac
|
|
||||||
else if (c == 352) mm_dbg_flag |= MM_DBG_PRINT_CHAIN; // --print-chains
|
|
||||||
else if (c == 353) opt.flag |= MM_F_NO_HASH_NAME; // --no-hash-name
|
|
||||||
else if (c == 354) opt.flag |= MM_F_SECONDARY_SEQ; // --secondary-seq
|
|
||||||
else if (c == 330) {
|
|
||||||
fprintf(stderr, "[WARNING] \033[1;31m --lj-min-ratio has been deprecated.\033[0m\n");
|
|
||||||
} else if (c == 314) { // --frag
|
|
||||||
yes_or_no(&opt, MM_F_FRAG_MODE, o.longidx, o.arg, 1);
|
yes_or_no(&opt, MM_F_FRAG_MODE, o.longidx, o.arg, 1);
|
||||||
} else if (c == 315) { // --secondary
|
} else if (c == 315) { // --secondary
|
||||||
yes_or_no(&opt, MM_F_NO_PRINT_2ND, o.longidx, o.arg, 0);
|
yes_or_no(&opt, MM_F_NO_PRINT_2ND, o.longidx, o.arg, 0);
|
||||||
@@ -265,9 +313,6 @@ int main(int argc, char *argv[])
|
|||||||
yes_or_no(&opt, MM_F_HEAP_SORT, o.longidx, o.arg, 1);
|
yes_or_no(&opt, MM_F_HEAP_SORT, o.longidx, o.arg, 1);
|
||||||
} else if (c == 326) { // --dual
|
} else if (c == 326) { // --dual
|
||||||
yes_or_no(&opt, MM_F_NO_DUAL, o.longidx, o.arg, 0);
|
yes_or_no(&opt, MM_F_NO_DUAL, o.longidx, o.arg, 0);
|
||||||
} else if (c == 347) { // --rmq
|
|
||||||
if (o.arg) yes_or_no(&opt, MM_F_RMQ, o.longidx, o.arg, 1);
|
|
||||||
else opt.flag |= MM_F_RMQ;
|
|
||||||
} else if (c == 'S') {
|
} else if (c == 'S') {
|
||||||
opt.flag |= MM_F_OUT_CS | MM_F_CIGAR | MM_F_OUT_CS_LONG;
|
opt.flag |= MM_F_OUT_CS | MM_F_CIGAR | MM_F_OUT_CS_LONG;
|
||||||
if (mm_verbose >= 2)
|
if (mm_verbose >= 2)
|
||||||
@@ -275,12 +320,6 @@ int main(int argc, char *argv[])
|
|||||||
} else if (c == 'V') {
|
} else if (c == 'V') {
|
||||||
puts(MM_VERSION);
|
puts(MM_VERSION);
|
||||||
return 0;
|
return 0;
|
||||||
} else if (c == 'r') {
|
|
||||||
opt.bw = (int)mm_parse_num2(o.arg, &s);
|
|
||||||
if (*s == ',') opt.bw_long = (int)mm_parse_num2(s + 1, &s);
|
|
||||||
} else if (c == 'U') {
|
|
||||||
opt.min_mid_occ = strtol(o.arg, &s, 10);
|
|
||||||
if (*s == ',') opt.max_mid_occ = strtol(s + 1, &s, 10);
|
|
||||||
} else if (c == 'f') {
|
} else if (c == 'f') {
|
||||||
double x;
|
double x;
|
||||||
char *p;
|
char *p;
|
||||||
@@ -328,14 +367,14 @@ int main(int argc, char *argv[])
|
|||||||
fprintf(fp_help, " -H use homopolymer-compressed k-mer (preferrable for PacBio)\n");
|
fprintf(fp_help, " -H use homopolymer-compressed k-mer (preferrable for PacBio)\n");
|
||||||
fprintf(fp_help, " -k INT k-mer size (no larger than 28) [%d]\n", ipt.k);
|
fprintf(fp_help, " -k INT k-mer size (no larger than 28) [%d]\n", ipt.k);
|
||||||
fprintf(fp_help, " -w INT minimizer window size [%d]\n", ipt.w);
|
fprintf(fp_help, " -w INT minimizer window size [%d]\n", ipt.w);
|
||||||
fprintf(fp_help, " -I NUM split index for every ~NUM input bases [8G]\n");
|
fprintf(fp_help, " -I NUM split index for every ~NUM input bases [4G]\n");
|
||||||
fprintf(fp_help, " -d FILE dump index to FILE []\n");
|
fprintf(fp_help, " -d FILE dump index to FILE []\n");
|
||||||
fprintf(fp_help, " Mapping:\n");
|
fprintf(fp_help, " Mapping:\n");
|
||||||
fprintf(fp_help, " -f FLOAT filter out top FLOAT fraction of repetitive minimizers [%g]\n", opt.mid_occ_frac);
|
fprintf(fp_help, " -f FLOAT filter out top FLOAT fraction of repetitive minimizers [%g]\n", opt.mid_occ_frac);
|
||||||
fprintf(fp_help, " -g NUM stop chain enlongation if there are no minimizers in INT-bp [%d]\n", opt.max_gap);
|
fprintf(fp_help, " -g NUM stop chain enlongation if there are no minimizers in INT-bp [%d]\n", opt.max_gap);
|
||||||
fprintf(fp_help, " -G NUM max intron length (effective with -xsplice; changing -r) [200k]\n");
|
fprintf(fp_help, " -G NUM max intron length (effective with -xsplice; changing -r) [200k]\n");
|
||||||
fprintf(fp_help, " -F NUM max fragment length (effective with -xsr or in the fragment mode) [800]\n");
|
fprintf(fp_help, " -F NUM max fragment length (effective with -xsr or in the fragment mode) [800]\n");
|
||||||
fprintf(fp_help, " -r NUM[,NUM] chaining/alignment bandwidth and long-join bandwidth [%d,%d]\n", opt.bw, opt.bw_long);
|
fprintf(fp_help, " -r NUM bandwidth used in chaining and DP-based alignment [%d]\n", opt.bw);
|
||||||
fprintf(fp_help, " -n INT minimal number of minimizers on a chain [%d]\n", opt.min_cnt);
|
fprintf(fp_help, " -n INT minimal number of minimizers on a chain [%d]\n", opt.min_cnt);
|
||||||
fprintf(fp_help, " -m INT minimal chaining score (matching bases minus log gap penalty) [%d]\n", opt.min_chain_score);
|
fprintf(fp_help, " -m INT minimal chaining score (matching bases minus log gap penalty) [%d]\n", opt.min_chain_score);
|
||||||
// fprintf(fp_help, " -T INT SDUST threshold; 0 to disable SDUST [%d]\n", opt.sdust_thres); // TODO: this option is never used; might be buggy
|
// fprintf(fp_help, " -T INT SDUST threshold; 0 to disable SDUST [%d]\n", opt.sdust_thres); // TODO: this option is never used; might be buggy
|
||||||
@@ -344,13 +383,12 @@ int main(int argc, char *argv[])
|
|||||||
fprintf(fp_help, " -N INT retain at most INT secondary alignments [%d]\n", opt.best_n);
|
fprintf(fp_help, " -N INT retain at most INT secondary alignments [%d]\n", opt.best_n);
|
||||||
fprintf(fp_help, " Alignment:\n");
|
fprintf(fp_help, " Alignment:\n");
|
||||||
fprintf(fp_help, " -A INT matching score [%d]\n", opt.a);
|
fprintf(fp_help, " -A INT matching score [%d]\n", opt.a);
|
||||||
fprintf(fp_help, " -B INT mismatch penalty (larger value for lower divergence) [%d]\n", opt.b);
|
fprintf(fp_help, " -B INT mismatch penalty [%d]\n", opt.b);
|
||||||
fprintf(fp_help, " -O INT[,INT] gap open penalty [%d,%d]\n", opt.q, opt.q2);
|
fprintf(fp_help, " -O INT[,INT] gap open penalty [%d,%d]\n", opt.q, opt.q2);
|
||||||
fprintf(fp_help, " -E INT[,INT] gap extension penalty; a k-long gap costs min{O1+k*E1,O2+k*E2} [%d,%d]\n", opt.e, opt.e2);
|
fprintf(fp_help, " -E INT[,INT] gap extension penalty; a k-long gap costs min{O1+k*E1,O2+k*E2} [%d,%d]\n", opt.e, opt.e2);
|
||||||
fprintf(fp_help, " -z INT[,INT] Z-drop score and inversion Z-drop score [%d,%d]\n", opt.zdrop, opt.zdrop_inv);
|
fprintf(fp_help, " -z INT[,INT] Z-drop score and inversion Z-drop score [%d,%d]\n", opt.zdrop, opt.zdrop_inv);
|
||||||
fprintf(fp_help, " -s INT minimal peak DP alignment score [%d]\n", opt.min_dp_max);
|
fprintf(fp_help, " -s INT minimal peak DP alignment score [%d]\n", opt.min_dp_max);
|
||||||
fprintf(fp_help, " -u CHAR how to find GT-AG. f:transcript strand, b:both strands, n:don't match GT-AG [n]\n");
|
fprintf(fp_help, " -u CHAR how to find GT-AG. f:transcript strand, b:both strands, n:don't match GT-AG [n]\n");
|
||||||
fprintf(fp_help, " -J INT splice mode. 0: original minimap2 model; 1: miniprot model [1]\n");
|
|
||||||
fprintf(fp_help, " Input/Output:\n");
|
fprintf(fp_help, " Input/Output:\n");
|
||||||
fprintf(fp_help, " -a output in the SAM format (PAF by default)\n");
|
fprintf(fp_help, " -a output in the SAM format (PAF by default)\n");
|
||||||
fprintf(fp_help, " -o FILE output alignments to FILE [stdout]\n");
|
fprintf(fp_help, " -o FILE output alignments to FILE [stdout]\n");
|
||||||
@@ -367,8 +405,7 @@ int main(int argc, char *argv[])
|
|||||||
fprintf(fp_help, " --version show version number\n");
|
fprintf(fp_help, " --version show version number\n");
|
||||||
fprintf(fp_help, " Preset:\n");
|
fprintf(fp_help, " Preset:\n");
|
||||||
fprintf(fp_help, " -x STR preset (always applied before other options; see minimap2.1 for details) []\n");
|
fprintf(fp_help, " -x STR preset (always applied before other options; see minimap2.1 for details) []\n");
|
||||||
fprintf(fp_help, " - map-pb/map-ont - PacBio CLR/Nanopore vs reference mapping\n");
|
fprintf(fp_help, " - map-pb/map-ont - PacBio/Nanopore vs reference mapping\n");
|
||||||
fprintf(fp_help, " - map-hifi - PacBio HiFi reads vs reference mapping\n");
|
|
||||||
fprintf(fp_help, " - ava-pb/ava-ont - PacBio/Nanopore read overlap\n");
|
fprintf(fp_help, " - ava-pb/ava-ont - PacBio/Nanopore read overlap\n");
|
||||||
fprintf(fp_help, " - asm5/asm10/asm20 - asm-to-ref mapping, for ~0.1/1/5%% sequence divergence\n");
|
fprintf(fp_help, " - asm5/asm10/asm20 - asm-to-ref mapping, for ~0.1/1/5%% sequence divergence\n");
|
||||||
fprintf(fp_help, " - splice/splice:hq - long-read/Pacbio-CCS spliced alignment\n");
|
fprintf(fp_help, " - splice/splice:hq - long-read/Pacbio-CCS spliced alignment\n");
|
||||||
@@ -381,7 +418,12 @@ int main(int argc, char *argv[])
|
|||||||
fprintf(stderr, "[ERROR] incorrect input: in the sr mode, please specify no more than two query files.\n");
|
fprintf(stderr, "[ERROR] incorrect input: in the sr mode, please specify no more than two query files.\n");
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
preset_arg = (string)argv[o.ind] + "_" + preset_arg + "_minimizers_key_value_sorted";
|
||||||
|
|
||||||
|
|
||||||
idx_rdr = mm_idx_reader_open(argv[o.ind], &ipt, fnw);
|
idx_rdr = mm_idx_reader_open(argv[o.ind], &ipt, fnw);
|
||||||
|
total_time = __rdtsc();
|
||||||
if (idx_rdr == 0) {
|
if (idx_rdr == 0) {
|
||||||
fprintf(stderr, "[ERROR] failed to open file '%s': %s\n", argv[o.ind], strerror(errno));
|
fprintf(stderr, "[ERROR] failed to open file '%s': %s\n", argv[o.ind], strerror(errno));
|
||||||
return 1;
|
return 1;
|
||||||
@@ -423,13 +465,25 @@ int main(int argc, char *argv[])
|
|||||||
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), mi->n_seq);
|
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), mi->n_seq);
|
||||||
if (argc != o.ind + 1) mm_mapopt_update(&opt, mi);
|
if (argc != o.ind + 1) mm_mapopt_update(&opt, mi);
|
||||||
if (mm_verbose >= 3) mm_idx_stat(mi);
|
if (mm_verbose >= 3) mm_idx_stat(mi);
|
||||||
|
if(opt.L_hash == 1) {
|
||||||
|
fprintf(stderr, "Generating lisa-hash..\n");
|
||||||
|
mm_idx_dump_hash(preset_arg.c_str(), mi);
|
||||||
|
fprintf(stderr, "Lisa-hash saving done.. \n");
|
||||||
|
exit(0);
|
||||||
|
}
|
||||||
if (junc_bed) mm_idx_bed_read(mi, junc_bed, 1);
|
if (junc_bed) mm_idx_bed_read(mi, junc_bed, 1);
|
||||||
if (alt_list) mm_idx_alt_read(mi, alt_list);
|
if (alt_list) mm_idx_alt_read(mi, alt_list);
|
||||||
if (argc - (o.ind + 1) == 0) {
|
|
||||||
mm_idx_destroy(mi);
|
|
||||||
continue; // no query files
|
|
||||||
}
|
|
||||||
ret = 0;
|
ret = 0;
|
||||||
|
#ifdef LISA_HASH
|
||||||
|
fprintf(stderr, "Using LISA_HASH..\n");
|
||||||
|
mm_idx_destroy_mm_hash(mi);
|
||||||
|
char* prefix;
|
||||||
|
lh = new lisa_hash<uint64_t, uint64_t>(preset_arg, prefix);
|
||||||
|
fprintf(stderr, "Loading done.\n");
|
||||||
|
total_time = __rdtsc();
|
||||||
|
fprintf(stderr, "\nIndexing Real time: %.3f sec;\n", realtime() - mapping_time);
|
||||||
|
#endif
|
||||||
|
mapping_time = realtime();
|
||||||
if (!(opt.flag & MM_F_FRAG_MODE)) {
|
if (!(opt.flag & MM_F_FRAG_MODE)) {
|
||||||
for (i = o.ind + 1; i < argc; ++i) {
|
for (i = o.ind + 1; i < argc; ++i) {
|
||||||
ret = mm_map_file(mi, argv[i], &opt, n_threads);
|
ret = mm_map_file(mi, argv[i], &opt, n_threads);
|
||||||
@@ -438,11 +492,15 @@ int main(int argc, char *argv[])
|
|||||||
} else {
|
} else {
|
||||||
ret = mm_map_file_frag(mi, argc - (o.ind + 1), (const char**)&argv[o.ind + 1], &opt, n_threads);
|
ret = mm_map_file_frag(mi, argc - (o.ind + 1), (const char**)&argv[o.ind + 1], &opt, n_threads);
|
||||||
}
|
}
|
||||||
mm_idx_destroy(mi);
|
|
||||||
if (ret < 0) {
|
if (ret < 0) {
|
||||||
fprintf(stderr, "ERROR: failed to map the query file\n");
|
fprintf(stderr, "ERROR: failed to map the query file\n");
|
||||||
exit(EXIT_FAILURE);
|
exit(EXIT_FAILURE);
|
||||||
}
|
}
|
||||||
|
#ifdef LISA_HASH
|
||||||
|
mm_idx_destroy_seq(mi);
|
||||||
|
#else
|
||||||
|
mm_idx_destroy(mi);
|
||||||
|
#endif
|
||||||
}
|
}
|
||||||
n_parts = idx_rdr->n_parts;
|
n_parts = idx_rdr->n_parts;
|
||||||
mm_idx_reader_close(idx_rdr);
|
mm_idx_reader_close(idx_rdr);
|
||||||
@@ -462,5 +520,14 @@ int main(int argc, char *argv[])
|
|||||||
fprintf(stderr, " %s", argv[i]);
|
fprintf(stderr, " %s", argv[i]);
|
||||||
fprintf(stderr, "\n[M::%s] Real time: %.3f sec; CPU: %.3f sec; Peak RSS: %.3f GB\n", __func__, realtime() - mm_realtime0, cputime(), peakrss() / 1024.0 / 1024.0 / 1024.0);
|
fprintf(stderr, "\n[M::%s] Real time: %.3f sec; CPU: %.3f sec; Peak RSS: %.3f GB\n", __func__, realtime() - mm_realtime0, cputime(), peakrss() / 1024.0 / 1024.0 / 1024.0);
|
||||||
}
|
}
|
||||||
return 0;
|
|
||||||
|
#ifdef MANUAL_PROFILING
|
||||||
|
fprintf(stderr, "\n Number of reads = %lld Minimizer hit time = %lld dp_chaining time = %lld alignment time = %lld total time = %lld \n", num_reads, minimizer_hit_time, dp_chaining_time, alignment_time, __rdtsc() - total_time);
|
||||||
|
#endif
|
||||||
|
fprintf(stderr, "Total ticks: %lld \n",__rdtsc() - total_time);
|
||||||
|
fprintf(stderr, "\nMapping Real time: %.3f sec;\n", realtime() - mapping_time);
|
||||||
|
#ifdef LISA_HASH
|
||||||
|
delete lh;
|
||||||
|
#endif
|
||||||
|
return 0;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,3 +1,32 @@
|
|||||||
|
/* The MIT License
|
||||||
|
|
||||||
|
Copyright (c) 2018- Dana-Farber Cancer Institute
|
||||||
|
2017-2018 Broad Institute, Inc.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
|
Modified Copyright (C) 2021 Intel Corporation
|
||||||
|
Contacts: Saurabh Kalikar <saurabh.kalikar@intel.com>;
|
||||||
|
Vasimuddin Md <vasimuddin.md@intel.com>; Sanchit Misra <sanchit.misra@intel.com>;
|
||||||
|
Chirag Jain <chirag@iisc.ac.in>; Heng Li <hli@jimmy.harvard.edu>
|
||||||
|
*/
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include <assert.h>
|
#include <assert.h>
|
||||||
@@ -9,6 +38,22 @@
|
|||||||
#include "mmpriv.h"
|
#include "mmpriv.h"
|
||||||
#include "bseq.h"
|
#include "bseq.h"
|
||||||
#include "khash.h"
|
#include "khash.h"
|
||||||
|
#include <x86intrin.h>
|
||||||
|
|
||||||
|
#ifdef LISA_HASH
|
||||||
|
#include "lisa_hash.h"
|
||||||
|
extern lisa_hash<uint64_t, uint64_t> *lh;
|
||||||
|
#endif
|
||||||
|
|
||||||
|
#ifdef MANUAL_PROFILING
|
||||||
|
extern uint64_t num_reads, minimizer_hit_time, dp_chaining_time, alignment_time;
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
|
struct mm_tbuf_s {
|
||||||
|
void *km;
|
||||||
|
int rep_len, frag_gap;
|
||||||
|
};
|
||||||
|
|
||||||
mm_tbuf_t *mm_tbuf_init(void)
|
mm_tbuf_t *mm_tbuf_init(void)
|
||||||
{
|
{
|
||||||
@@ -75,7 +120,120 @@ static void collect_minimizers(void *km, const mm_mapopt_t *opt, const mm_idx_t
|
|||||||
#define heap_lt(a, b) ((a).x > (b).x)
|
#define heap_lt(a, b) ((a).x > (b).x)
|
||||||
KSORT_INIT(heap, mm128_t, heap_lt)
|
KSORT_INIT(heap, mm128_t, heap_lt)
|
||||||
|
|
||||||
static inline int skip_seed(int flag, uint64_t r, const mm_seed_t *q, const char *qname, int qlen, const mm_idx_t *mi, int *is_self)
|
typedef struct {
|
||||||
|
uint32_t n;
|
||||||
|
uint32_t q_pos, q_span;
|
||||||
|
uint32_t seg_id:31, is_tandem:1;
|
||||||
|
const uint64_t *cr;
|
||||||
|
} mm_match_t;
|
||||||
|
|
||||||
|
|
||||||
|
#ifdef LISA_HASH
|
||||||
|
static mm_match_t *collect_matches_lisa_hash(void *km, int *_n_m, int max_occ, const mm_idx_t *mi, const mm128_v *mv, int64_t *n_a, int *rep_len, int *n_mini_pos, uint64_t **mini_pos)
|
||||||
|
{
|
||||||
|
uint64_t** cr_batch = (uint64_t**) malloc((mv->n)*sizeof(uint64_t*));
|
||||||
|
int* t_batch = (int*)malloc((mv->n)*sizeof(int));
|
||||||
|
uint64_t* minimizers = (uint64_t*) malloc((mv->n)*sizeof(uint64_t));
|
||||||
|
int64_t* lisa_pos = (int64_t*) malloc((max(32, (int)mv->n))* sizeof(int64_t));
|
||||||
|
|
||||||
|
int rep_st = 0, rep_en = 0, n_m;
|
||||||
|
size_t i;
|
||||||
|
mm_match_t *m;
|
||||||
|
*n_mini_pos = 0;
|
||||||
|
*mini_pos = (uint64_t*)kmalloc(km, mv->n * sizeof(uint64_t));
|
||||||
|
m = (mm_match_t*)kmalloc(km, mv->n * sizeof(mm_match_t));
|
||||||
|
|
||||||
|
for (i = 0; i < mv->n; i++) {
|
||||||
|
mm128_t *p = &mv->a[i];
|
||||||
|
minimizers[i] = p->x>>8;
|
||||||
|
}
|
||||||
|
|
||||||
|
lh->mm_idx_get_batched(minimizers, mv->n, lisa_pos, cr_batch, t_batch);
|
||||||
|
for (i = 0, n_m = 0, *rep_len = 0, *n_a = 0; i < mv->n; ++i) {
|
||||||
|
const uint64_t *cr;
|
||||||
|
mm128_t *p = &mv->a[i];
|
||||||
|
uint32_t q_pos = (uint32_t)p->y, q_span = p->x & 0xff;
|
||||||
|
int t;
|
||||||
|
cr = cr_batch[i]; t = t_batch[i];
|
||||||
|
/*Correctness check for lisa_hash*/
|
||||||
|
#ifdef LISA_HASH_ASSERT
|
||||||
|
int t_minimap2_original;
|
||||||
|
const uint64_t *cr_minimap2_hash = mm_idx_get(mi, p->x>>8, &t);
|
||||||
|
cr_minimap2_hash = mm_idx_get(mi, p->x>>8, &t_minimap2_original);
|
||||||
|
assert(t == t_minimap2_original);
|
||||||
|
#endif
|
||||||
|
if (t >= max_occ) {
|
||||||
|
|
||||||
|
int en = (q_pos >> 1) + 1, st = en - q_span;
|
||||||
|
if (st > rep_en) {
|
||||||
|
*rep_len += rep_en - rep_st;
|
||||||
|
rep_st = st, rep_en = en;
|
||||||
|
} else rep_en = en;
|
||||||
|
} else {
|
||||||
|
#ifdef LISA_HASH_ASSERT
|
||||||
|
//Correctness assertion
|
||||||
|
for(int itr = 0; itr < t; itr++){
|
||||||
|
assert((cr[itr] == cr_minimap2_hash[itr]));
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
mm_match_t *q = &m[n_m++];
|
||||||
|
q->q_pos = q_pos, q->q_span = q_span, q->cr = cr, q->n = t, q->seg_id = p->y >> 32;
|
||||||
|
q->is_tandem = 0;
|
||||||
|
if (i > 0 && p->x>>8 == mv->a[i - 1].x>>8) q->is_tandem = 1;
|
||||||
|
if (i < mv->n - 1 && p->x>>8 == mv->a[i + 1].x>>8) q->is_tandem = 1;
|
||||||
|
*n_a += q->n;
|
||||||
|
(*mini_pos)[(*n_mini_pos)++] = (uint64_t)q_span<<32 | q_pos>>1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
free(cr_batch);
|
||||||
|
free(t_batch);
|
||||||
|
free(minimizers);
|
||||||
|
free(lisa_pos);
|
||||||
|
|
||||||
|
*rep_len += rep_en - rep_st;
|
||||||
|
*_n_m = n_m;
|
||||||
|
return m;
|
||||||
|
}
|
||||||
|
#endif
|
||||||
|
|
||||||
|
|
||||||
|
static mm_match_t *collect_matches(void *km, int *_n_m, int max_occ, const mm_idx_t *mi, const mm128_v *mv, int64_t *n_a, int *rep_len, int *n_mini_pos, uint64_t **mini_pos)
|
||||||
|
{
|
||||||
|
int rep_st = 0, rep_en = 0, n_m;
|
||||||
|
size_t i;
|
||||||
|
mm_match_t *m;
|
||||||
|
*n_mini_pos = 0;
|
||||||
|
*mini_pos = (uint64_t*)kmalloc(km, mv->n * sizeof(uint64_t));
|
||||||
|
m = (mm_match_t*)kmalloc(km, mv->n * sizeof(mm_match_t));
|
||||||
|
for (i = 0, n_m = 0, *rep_len = 0, *n_a = 0; i < mv->n; ++i) {
|
||||||
|
const uint64_t *cr;
|
||||||
|
mm128_t *p = &mv->a[i];
|
||||||
|
uint32_t q_pos = (uint32_t)p->y, q_span = p->x & 0xff;
|
||||||
|
int t;
|
||||||
|
cr = mm_idx_get(mi, p->x>>8, &t);
|
||||||
|
if (t >= max_occ) {
|
||||||
|
int en = (q_pos >> 1) + 1, st = en - q_span;
|
||||||
|
if (st > rep_en) {
|
||||||
|
*rep_len += rep_en - rep_st;
|
||||||
|
rep_st = st, rep_en = en;
|
||||||
|
} else rep_en = en;
|
||||||
|
} else {
|
||||||
|
mm_match_t *q = &m[n_m++];
|
||||||
|
q->q_pos = q_pos, q->q_span = q_span, q->cr = cr, q->n = t, q->seg_id = p->y >> 32;
|
||||||
|
q->is_tandem = 0;
|
||||||
|
if (i > 0 && p->x>>8 == mv->a[i - 1].x>>8) q->is_tandem = 1;
|
||||||
|
if (i < mv->n - 1 && p->x>>8 == mv->a[i + 1].x>>8) q->is_tandem = 1;
|
||||||
|
*n_a += q->n;
|
||||||
|
(*mini_pos)[(*n_mini_pos)++] = (uint64_t)q_span<<32 | q_pos>>1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
*rep_len += rep_en - rep_st;
|
||||||
|
*_n_m = n_m;
|
||||||
|
return m;
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline int skip_seed(int flag, uint64_t r, const mm_match_t *q, const char *qname, int qlen, const mm_idx_t *mi, int *is_self)
|
||||||
{
|
{
|
||||||
*is_self = 0;
|
*is_self = 0;
|
||||||
if (qname && (flag & (MM_F_NO_DIAG|MM_F_NO_DUAL))) {
|
if (qname && (flag & (MM_F_NO_DIAG|MM_F_NO_DUAL))) {
|
||||||
@@ -99,15 +257,58 @@ static inline int skip_seed(int flag, uint64_t r, const mm_seed_t *q, const char
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
static mm128_t *collect_seed_hits(void *km, const mm_mapopt_t *opt, int max_occ, const mm_idx_t *mi, const char *qname, const mm128_v *mv, int qlen, int64_t *n_a, int *rep_len,
|
||||||
|
int *n_mini_pos, uint64_t **mini_pos)
|
||||||
|
{
|
||||||
|
int i, n_m;
|
||||||
|
mm_match_t *m;
|
||||||
|
mm128_t *a;
|
||||||
|
#ifndef LISA_HASH
|
||||||
|
m = collect_matches(km, &n_m, max_occ, mi, mv, n_a, rep_len, n_mini_pos, mini_pos);
|
||||||
|
#else
|
||||||
|
m = collect_matches_lisa_hash(km, &n_m, max_occ, mi, mv, n_a, rep_len, n_mini_pos, mini_pos);
|
||||||
|
#endif
|
||||||
|
|
||||||
|
a = (mm128_t*)kmalloc(km, *n_a * sizeof(mm128_t));
|
||||||
|
for (i = 0, *n_a = 0; i < n_m; ++i) {
|
||||||
|
|
||||||
|
mm_match_t *q = &m[i];
|
||||||
|
const uint64_t *r = q->cr;
|
||||||
|
uint32_t k;
|
||||||
|
for (k = 0; k < q->n; ++k) {
|
||||||
|
uint64_t r_k = r[k];
|
||||||
|
int32_t is_self, rpos = (uint32_t)r_k >> 1;
|
||||||
|
mm128_t *p;
|
||||||
|
if (skip_seed(opt->flag, r_k, q, qname, qlen, mi, &is_self)) continue;
|
||||||
|
p = &a[(*n_a)++];
|
||||||
|
if ((r_k&1) == (q->q_pos&1)) { // forward strand
|
||||||
|
p->x = (r_k & 0xffffffff00000000ULL) | rpos;
|
||||||
|
p->y = (uint64_t)q->q_span << 32 | q->q_pos >> 1;
|
||||||
|
} else { // reverse strand
|
||||||
|
p->x = 1ULL<<63 | (r_k & 0xffffffff00000000ULL) | rpos;
|
||||||
|
p->y = (uint64_t)q->q_span << 32 | (qlen - ((q->q_pos>>1) + 1 - q->q_span) - 1);
|
||||||
|
}
|
||||||
|
p->y |= (uint64_t)q->seg_id << MM_SEED_SEG_SHIFT;
|
||||||
|
if (q->is_tandem) p->y |= MM_SEED_TANDEM;
|
||||||
|
if (is_self) p->y |= MM_SEED_SELF;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
kfree(km, m);
|
||||||
|
radix_sort_128x(a, a + (*n_a));
|
||||||
|
return a;
|
||||||
|
}
|
||||||
|
|
||||||
static mm128_t *collect_seed_hits_heap(void *km, const mm_mapopt_t *opt, int max_occ, const mm_idx_t *mi, const char *qname, const mm128_v *mv, int qlen, int64_t *n_a, int *rep_len,
|
static mm128_t *collect_seed_hits_heap(void *km, const mm_mapopt_t *opt, int max_occ, const mm_idx_t *mi, const char *qname, const mm128_v *mv, int qlen, int64_t *n_a, int *rep_len,
|
||||||
int *n_mini_pos, uint64_t **mini_pos)
|
int *n_mini_pos, uint64_t **mini_pos)
|
||||||
{
|
{
|
||||||
int i, n_m, heap_size = 0;
|
int i, n_m, heap_size = 0;
|
||||||
int64_t j, n_for = 0, n_rev = 0;
|
int64_t j, n_for = 0, n_rev = 0;
|
||||||
mm_seed_t *m;
|
mm_match_t *m;
|
||||||
mm128_t *a, *heap;
|
mm128_t *a, *heap;
|
||||||
|
|
||||||
m = mm_collect_matches(km, &n_m, qlen, max_occ, opt->max_max_occ, opt->occ_dist, mi, mv, n_a, rep_len, n_mini_pos, mini_pos);
|
m = collect_matches(km, &n_m, max_occ, mi, mv, n_a, rep_len, n_mini_pos, mini_pos);
|
||||||
|
|
||||||
heap = (mm128_t*)kmalloc(km, n_m * sizeof(mm128_t));
|
heap = (mm128_t*)kmalloc(km, n_m * sizeof(mm128_t));
|
||||||
a = (mm128_t*)kmalloc(km, *n_a * sizeof(mm128_t));
|
a = (mm128_t*)kmalloc(km, *n_a * sizeof(mm128_t));
|
||||||
@@ -121,7 +322,7 @@ static mm128_t *collect_seed_hits_heap(void *km, const mm_mapopt_t *opt, int max
|
|||||||
}
|
}
|
||||||
ks_heapmake_heap(heap_size, heap);
|
ks_heapmake_heap(heap_size, heap);
|
||||||
while (heap_size > 0) {
|
while (heap_size > 0) {
|
||||||
mm_seed_t *q = &m[heap->y>>32];
|
mm_match_t *q = &m[heap->y>>32];
|
||||||
mm128_t *p;
|
mm128_t *p;
|
||||||
uint64_t r = heap->x;
|
uint64_t r = heap->x;
|
||||||
int32_t is_self, rpos = (uint32_t)r >> 1;
|
int32_t is_self, rpos = (uint32_t)r >> 1;
|
||||||
@@ -165,50 +366,14 @@ static mm128_t *collect_seed_hits_heap(void *km, const mm_mapopt_t *opt, int max
|
|||||||
return a;
|
return a;
|
||||||
}
|
}
|
||||||
|
|
||||||
static mm128_t *collect_seed_hits(void *km, const mm_mapopt_t *opt, int max_occ, const mm_idx_t *mi, const char *qname, const mm128_v *mv, int qlen, int64_t *n_a, int *rep_len,
|
|
||||||
int *n_mini_pos, uint64_t **mini_pos)
|
|
||||||
{
|
|
||||||
int i, n_m;
|
|
||||||
mm_seed_t *m;
|
|
||||||
mm128_t *a;
|
|
||||||
m = mm_collect_matches(km, &n_m, qlen, max_occ, opt->max_max_occ, opt->occ_dist, mi, mv, n_a, rep_len, n_mini_pos, mini_pos);
|
|
||||||
a = (mm128_t*)kmalloc(km, *n_a * sizeof(mm128_t));
|
|
||||||
for (i = 0, *n_a = 0; i < n_m; ++i) {
|
|
||||||
mm_seed_t *q = &m[i];
|
|
||||||
const uint64_t *r = q->cr;
|
|
||||||
uint32_t k;
|
|
||||||
for (k = 0; k < q->n; ++k) {
|
|
||||||
int32_t is_self, rpos = (uint32_t)r[k] >> 1;
|
|
||||||
mm128_t *p;
|
|
||||||
if (skip_seed(opt->flag, r[k], q, qname, qlen, mi, &is_self)) continue;
|
|
||||||
p = &a[(*n_a)++];
|
|
||||||
if ((r[k]&1) == (q->q_pos&1)) { // forward strand
|
|
||||||
p->x = (r[k]&0xffffffff00000000ULL) | rpos;
|
|
||||||
p->y = (uint64_t)q->q_span << 32 | q->q_pos >> 1;
|
|
||||||
} else if (!(opt->flag & MM_F_QSTRAND)) { // reverse strand and not in the query-strand mode
|
|
||||||
p->x = 1ULL<<63 | (r[k]&0xffffffff00000000ULL) | rpos;
|
|
||||||
p->y = (uint64_t)q->q_span << 32 | (qlen - ((q->q_pos>>1) + 1 - q->q_span) - 1);
|
|
||||||
} else { // reverse strand; query-strand
|
|
||||||
int32_t len = mi->seq[r[k]>>32].len;
|
|
||||||
p->x = 1ULL<<63 | (r[k]&0xffffffff00000000ULL) | (len - (rpos + 1 - q->q_span) - 1); // coordinate only accurate for non-HPC seeds
|
|
||||||
p->y = (uint64_t)q->q_span << 32 | q->q_pos >> 1;
|
|
||||||
}
|
|
||||||
p->y |= (uint64_t)q->seg_id << MM_SEED_SEG_SHIFT;
|
|
||||||
if (q->is_tandem) p->y |= MM_SEED_TANDEM;
|
|
||||||
if (is_self) p->y |= MM_SEED_SELF;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
kfree(km, m);
|
|
||||||
radix_sort_128x(a, a + (*n_a));
|
|
||||||
return a;
|
|
||||||
}
|
|
||||||
|
|
||||||
static void chain_post(const mm_mapopt_t *opt, int max_chain_gap_ref, const mm_idx_t *mi, void *km, int qlen, int n_segs, const int *qlens, int *n_regs, mm_reg1_t *regs, mm128_t *a)
|
static void chain_post(const mm_mapopt_t *opt, int max_chain_gap_ref, const mm_idx_t *mi, void *km, int qlen, int n_segs, const int *qlens, int *n_regs, mm_reg1_t *regs, mm128_t *a)
|
||||||
{
|
{
|
||||||
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
||||||
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
||||||
if (n_segs <= 1) mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, 1, opt->max_gap * 0.8, n_regs, regs);
|
if (n_segs <= 1) mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, n_regs, regs);
|
||||||
else mm_select_sub_multi(km, opt->pri_ratio, 0.2f, 0.7f, max_chain_gap_ref, mi->k*2, opt->best_n, n_segs, qlens, n_regs, regs);
|
else mm_select_sub_multi(km, opt->pri_ratio, 0.2f, 0.7f, max_chain_gap_ref, mi->k*2, opt->best_n, n_segs, qlens, n_regs, regs);
|
||||||
|
if (!(opt->flag & (MM_F_SPLICE|MM_F_SR|MM_F_NO_LJOIN))) // long join not working well without primary chains
|
||||||
|
mm_join_long(km, opt, qlen, n_regs, regs, a);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -218,7 +383,7 @@ static mm_reg1_t *align_regs(const mm_mapopt_t *opt, const mm_idx_t *mi, void *k
|
|||||||
regs = mm_align_skeleton(km, opt, mi, qlen, seq, n_regs, regs, a); // this calls mm_filter_regs()
|
regs = mm_align_skeleton(km, opt, mi, qlen, seq, n_regs, regs, a); // this calls mm_filter_regs()
|
||||||
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
||||||
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
||||||
mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, 0, opt->max_gap * 0.8, n_regs, regs);
|
mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, n_regs, regs);
|
||||||
mm_set_sam_pri(*n_regs, regs);
|
mm_set_sam_pri(*n_regs, regs);
|
||||||
}
|
}
|
||||||
return regs;
|
return regs;
|
||||||
@@ -226,6 +391,11 @@ static mm_reg1_t *align_regs(const mm_mapopt_t *opt, const mm_idx_t *mi, void *k
|
|||||||
|
|
||||||
void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **seqs, int *n_regs, mm_reg1_t **regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **seqs, int *n_regs, mm_reg1_t **regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
||||||
{
|
{
|
||||||
|
|
||||||
|
#ifdef MANUAL_PROFILING
|
||||||
|
num_reads++;
|
||||||
|
#endif
|
||||||
|
|
||||||
int i, j, rep_len, qlen_sum, n_regs0, n_mini_pos;
|
int i, j, rep_len, qlen_sum, n_regs0, n_mini_pos;
|
||||||
int max_chain_gap_qry, max_chain_gap_ref, is_splice = !!(opt->flag & MM_F_SPLICE), is_sr = !!(opt->flag & MM_F_SR);
|
int max_chain_gap_qry, max_chain_gap_ref, is_splice = !!(opt->flag & MM_F_SPLICE), is_sr = !!(opt->flag & MM_F_SR);
|
||||||
uint32_t hash;
|
uint32_t hash;
|
||||||
@@ -235,7 +405,6 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
mm128_v mv = {0,0,0};
|
mm128_v mv = {0,0,0};
|
||||||
mm_reg1_t *regs0;
|
mm_reg1_t *regs0;
|
||||||
km_stat_t kmst;
|
km_stat_t kmst;
|
||||||
float chn_pen_gap, chn_pen_skip;
|
|
||||||
|
|
||||||
for (i = 0, qlen_sum = 0; i < n_segs; ++i)
|
for (i = 0, qlen_sum = 0; i < n_segs; ++i)
|
||||||
qlen_sum += qlens[i], n_regs[i] = 0, regs[i] = 0;
|
qlen_sum += qlens[i], n_regs[i] = 0, regs[i] = 0;
|
||||||
@@ -243,15 +412,24 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
if (qlen_sum == 0 || n_segs <= 0 || n_segs > MM_MAX_SEG) return;
|
if (qlen_sum == 0 || n_segs <= 0 || n_segs > MM_MAX_SEG) return;
|
||||||
if (opt->max_qlen > 0 && qlen_sum > opt->max_qlen) return;
|
if (opt->max_qlen > 0 && qlen_sum > opt->max_qlen) return;
|
||||||
|
|
||||||
hash = qname && !(opt->flag & MM_F_NO_HASH_NAME)? __ac_X31_hash_string(qname) : 0;
|
hash = qname? __ac_X31_hash_string(qname) : 0;
|
||||||
hash ^= __ac_Wang_hash(qlen_sum) + __ac_Wang_hash(opt->seed);
|
hash ^= __ac_Wang_hash(qlen_sum) + __ac_Wang_hash(opt->seed);
|
||||||
hash = __ac_Wang_hash(hash);
|
hash = __ac_Wang_hash(hash);
|
||||||
|
|
||||||
collect_minimizers(b->km, opt, mi, n_segs, qlens, seqs, &mv);
|
collect_minimizers(b->km, opt, mi, n_segs, qlens, seqs, &mv);
|
||||||
if (opt->q_occ_frac > 0.0f) mm_seed_mz_flt(b->km, &mv, opt->mid_occ, opt->q_occ_frac);
|
|
||||||
if (opt->flag & MM_F_HEAP_SORT) a = collect_seed_hits_heap(b->km, opt, opt->mid_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
if (opt->flag & MM_F_HEAP_SORT) a = collect_seed_hits_heap(b->km, opt, opt->mid_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||||
else a = collect_seed_hits(b->km, opt, opt->mid_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
else {
|
||||||
|
|
||||||
|
#ifdef MANUAL_PROFILING
|
||||||
|
uint64_t mm_hit_start = __rdtsc();
|
||||||
|
#endif
|
||||||
|
|
||||||
|
a = collect_seed_hits(b->km, opt, opt->mid_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||||
|
|
||||||
|
#ifdef MANUAL_PROFILING
|
||||||
|
minimizer_hit_time += (__rdtsc() - mm_hit_start);
|
||||||
|
#endif
|
||||||
|
}
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_SEED) {
|
if (mm_dbg_flag & MM_DBG_PRINT_SEED) {
|
||||||
fprintf(stderr, "RS\t%d\n", rep_len);
|
fprintf(stderr, "RS\t%d\n", rep_len);
|
||||||
for (i = 0; i < n_a; ++i)
|
for (i = 0; i < n_a; ++i)
|
||||||
@@ -269,28 +447,17 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
max_chain_gap_ref = opt->max_frag_len - qlen_sum;
|
max_chain_gap_ref = opt->max_frag_len - qlen_sum;
|
||||||
if (max_chain_gap_ref < opt->max_gap) max_chain_gap_ref = opt->max_gap;
|
if (max_chain_gap_ref < opt->max_gap) max_chain_gap_ref = opt->max_gap;
|
||||||
} else max_chain_gap_ref = opt->max_gap;
|
} else max_chain_gap_ref = opt->max_gap;
|
||||||
|
#ifdef MANUAL_PROFILING
|
||||||
|
uint64_t dp_start = __rdtsc();
|
||||||
|
#endif
|
||||||
|
a = mm_chain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score, opt->chain_gap_scale, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||||
|
|
||||||
chn_pen_gap = opt->chain_gap_scale * 0.01 * mi->k;
|
|
||||||
chn_pen_skip = opt->chain_skip_scale * 0.01 * mi->k;
|
|
||||||
if (opt->flag & MM_F_RMQ) {
|
|
||||||
a = mg_lchain_rmq(opt->max_gap, opt->rmq_inner_dist, opt->bw, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
|
||||||
chn_pen_gap, chn_pen_skip, n_a, a, &n_regs0, &u, b->km);
|
|
||||||
} else {
|
|
||||||
a = mg_lchain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score,
|
|
||||||
chn_pen_gap, chn_pen_skip, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (opt->bw_long > opt->bw && (opt->flag & (MM_F_SPLICE|MM_F_SR|MM_F_NO_LJOIN)) == 0 && n_segs == 1 && n_regs0 > 1) { // re-chain/long-join for long sequences
|
#ifdef MANUAL_PROFILING
|
||||||
int32_t st = (int32_t)a[0].y, en = (int32_t)a[(int32_t)u[0] - 1].y;
|
dp_chaining_time += (__rdtsc() - dp_start);
|
||||||
if (qlen_sum - (en - st) > opt->rmq_rescue_size || en - st > qlen_sum * opt->rmq_rescue_ratio) {
|
#endif
|
||||||
int32_t i;
|
|
||||||
for (i = 0, n_a = 0; i < n_regs0; ++i) n_a += (int32_t)u[i];
|
if (opt->max_occ > opt->mid_occ && rep_len > 0) {
|
||||||
kfree(b->km, u);
|
|
||||||
radix_sort_128x(a, a + n_a);
|
|
||||||
a = mg_lchain_rmq(opt->max_gap, opt->rmq_inner_dist, opt->bw_long, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
|
||||||
chn_pen_gap, chn_pen_skip, n_a, a, &n_regs0, &u, b->km);
|
|
||||||
}
|
|
||||||
} else if (opt->max_occ > opt->mid_occ && rep_len > 0 && !(opt->flag & MM_F_RMQ)) { // re-chain, mostly for short reads
|
|
||||||
int rechain = 0;
|
int rechain = 0;
|
||||||
if (n_regs0 > 0) { // test if the best chain has all the segments
|
if (n_regs0 > 0) { // test if the best chain has all the segments
|
||||||
int n_chained_segs = 1, max = 0, max_i = -1, max_off = -1, off = 0;
|
int n_chained_segs = 1, max = 0, max_i = -1, max_off = -1, off = 0;
|
||||||
@@ -310,34 +477,30 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
kfree(b->km, mini_pos);
|
kfree(b->km, mini_pos);
|
||||||
if (opt->flag & MM_F_HEAP_SORT) a = collect_seed_hits_heap(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
if (opt->flag & MM_F_HEAP_SORT) a = collect_seed_hits_heap(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||||
else a = collect_seed_hits(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
else a = collect_seed_hits(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||||
a = mg_lchain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score,
|
|
||||||
chn_pen_gap, chn_pen_skip, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
a = mm_chain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score, opt->chain_gap_scale, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
b->frag_gap = max_chain_gap_ref;
|
b->frag_gap = max_chain_gap_ref;
|
||||||
b->rep_len = rep_len;
|
b->rep_len = rep_len;
|
||||||
|
|
||||||
regs0 = mm_gen_regs(b->km, hash, qlen_sum, n_regs0, u, a, !!(opt->flag&MM_F_QSTRAND));
|
regs0 = mm_gen_regs(b->km, hash, qlen_sum, n_regs0, u, a);
|
||||||
if (mi->n_alt) {
|
if (mi->n_alt) {
|
||||||
mm_mark_alt(mi, n_regs0, regs0);
|
mm_mark_alt(mi, n_regs0, regs0);
|
||||||
mm_hit_sort(b->km, &n_regs0, regs0, opt->alt_drop); // this step can be merged into mm_gen_regs(); will do if this shows up in profile
|
mm_hit_sort(b->km, &n_regs0, regs0, opt->alt_drop); // this step can be merged into mm_gen_regs(); will do if this shows up in profile
|
||||||
}
|
}
|
||||||
|
|
||||||
if (mm_dbg_flag & (MM_DBG_PRINT_SEED|MM_DBG_PRINT_CHAIN))
|
if (mm_dbg_flag & MM_DBG_PRINT_SEED)
|
||||||
for (j = 0; j < n_regs0; ++j)
|
for (j = 0; j < n_regs0; ++j)
|
||||||
for (i = regs0[j].as; i < regs0[j].as + regs0[j].cnt; ++i)
|
for (i = regs0[j].as; i < regs0[j].as + regs0[j].cnt; ++i)
|
||||||
fprintf(stderr, "CN\t%d\t%s\t%d\t%c\t%d\t%d\t%d\n", j, mi->seq[a[i].x<<1>>33].name, (int32_t)a[i].x, "+-"[a[i].x>>63], (int32_t)a[i].y, (int32_t)(a[i].y>>32&0xff),
|
fprintf(stderr, "CN\t%d\t%s\t%d\t%c\t%d\t%d\t%d\n", j, mi->seq[a[i].x<<1>>33].name, (int32_t)a[i].x, "+-"[a[i].x>>63], (int32_t)a[i].y, (int32_t)(a[i].y>>32&0xff),
|
||||||
i == regs0[j].as? 0 : ((int32_t)a[i].y - (int32_t)a[i-1].y) - ((int32_t)a[i].x - (int32_t)a[i-1].x));
|
i == regs0[j].as? 0 : ((int32_t)a[i].y - (int32_t)a[i-1].y) - ((int32_t)a[i].x - (int32_t)a[i-1].x));
|
||||||
|
|
||||||
chain_post(opt, max_chain_gap_ref, mi, b->km, qlen_sum, n_segs, qlens, &n_regs0, regs0, a);
|
chain_post(opt, max_chain_gap_ref, mi, b->km, qlen_sum, n_segs, qlens, &n_regs0, regs0, a);
|
||||||
if (!is_sr && !(opt->flag&MM_F_QSTRAND)) {
|
if (!is_sr) mm_est_err(mi, qlen_sum, n_regs0, regs0, a, n_mini_pos, mini_pos);
|
||||||
mm_est_err(mi, qlen_sum, n_regs0, regs0, a, n_mini_pos, mini_pos);
|
|
||||||
n_regs0 = mm_filter_strand_retained(n_regs0, regs0);
|
|
||||||
}
|
|
||||||
|
|
||||||
if (n_segs == 1) { // uni-segment
|
if (n_segs == 1) { // uni-segment
|
||||||
regs0 = align_regs(opt, mi, b->km, qlens[0], seqs[0], &n_regs0, regs0, a);
|
regs0 = align_regs(opt, mi, b->km, qlens[0], seqs[0], &n_regs0, regs0, a);
|
||||||
regs0 = (mm_reg1_t*)realloc(regs0, sizeof(*regs0) * n_regs0);
|
|
||||||
mm_set_mapq(b->km, n_regs0, regs0, opt->min_chain_score, opt->a, rep_len, is_sr);
|
mm_set_mapq(b->km, n_regs0, regs0, opt->min_chain_score, opt->a, rep_len, is_sr);
|
||||||
n_regs[0] = n_regs0, regs[0] = regs0;
|
n_regs[0] = n_regs0, regs[0] = regs0;
|
||||||
} else { // multi-segment
|
} else { // multi-segment
|
||||||
@@ -364,9 +527,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||||
fprintf(stderr, "QM\t%s\t%d\tcap=%ld,nCore=%ld,largest=%ld\n", qname, qlen_sum, kmst.capacity, kmst.n_cores, kmst.largest);
|
fprintf(stderr, "QM\t%s\t%d\tcap=%ld,nCore=%ld,largest=%ld\n", qname, qlen_sum, kmst.capacity, kmst.n_cores, kmst.largest);
|
||||||
assert(kmst.n_blocks == kmst.n_cores); // otherwise, there is a memory leak
|
assert(kmst.n_blocks == kmst.n_cores); // otherwise, there is a memory leak
|
||||||
if (kmst.largest > 1U<<28 || (opt->cap_kalloc > 0 && kmst.capacity > opt->cap_kalloc)) {
|
if (kmst.largest > 1U<<28) {
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
|
||||||
fprintf(stderr, "[W::%s] reset thread-local memory after read %s\n", __func__, qname);
|
|
||||||
km_destroy(b->km);
|
km_destroy(b->km);
|
||||||
b->km = km_init();
|
b->km = km_init();
|
||||||
}
|
}
|
||||||
@@ -411,13 +572,11 @@ static void worker_for(void *_data, long i, int tid) // kt_for() callback
|
|||||||
step_t *s = (step_t*)_data;
|
step_t *s = (step_t*)_data;
|
||||||
int qlens[MM_MAX_SEG], j, off = s->seg_off[i], pe_ori = s->p->opt->pe_ori;
|
int qlens[MM_MAX_SEG], j, off = s->seg_off[i], pe_ori = s->p->opt->pe_ori;
|
||||||
const char *qseqs[MM_MAX_SEG];
|
const char *qseqs[MM_MAX_SEG];
|
||||||
double t = 0.0;
|
|
||||||
mm_tbuf_t *b = s->buf[tid];
|
mm_tbuf_t *b = s->buf[tid];
|
||||||
|
|
||||||
assert(s->n_seg[i] <= MM_MAX_SEG);
|
assert(s->n_seg[i] <= MM_MAX_SEG);
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME) {
|
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||||
fprintf(stderr, "QR\t%s\t%d\t%d\n", s->seq[off].name, tid, s->seq[off].l_seq);
|
fprintf(stderr, "QR\t%s\t%d\t%d\n", s->seq[off].name, tid, s->seq[off].l_seq);
|
||||||
t = realtime();
|
|
||||||
}
|
|
||||||
for (j = 0; j < s->n_seg[i]; ++j) {
|
for (j = 0; j < s->n_seg[i]; ++j) {
|
||||||
if (s->n_seg[i] == 2 && ((j == 0 && (pe_ori>>1&1)) || (j == 1 && (pe_ori&1))))
|
if (s->n_seg[i] == 2 && ((j == 0 && (pe_ori>>1&1)) || (j == 1 && (pe_ori&1))))
|
||||||
mm_revcomp_bseq(&s->seq[off + j]);
|
mm_revcomp_bseq(&s->seq[off + j]);
|
||||||
@@ -449,8 +608,6 @@ static void worker_for(void *_data, long i, int tid) // kt_for() callback
|
|||||||
r->rev = !r->rev;
|
r->rev = !r->rev;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
|
||||||
fprintf(stderr, "QT\t%s\t%d\t%.6f\n", s->seq[off].name, tid, realtime() - t);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
static void merge_hits(step_t *s)
|
static void merge_hits(step_t *s)
|
||||||
@@ -495,18 +652,10 @@ static void merge_hits(step_t *s)
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!(opt->flag&MM_F_SR) && s->seq[k].l_seq >= opt->rank_min_len)
|
|
||||||
mm_update_dp_max(s->seq[k].l_seq, s->n_reg[k], s->reg[k], opt->rank_frac, opt->a, opt->b);
|
|
||||||
for (j = 0; j < s->n_reg[k]; ++j) {
|
|
||||||
mm_reg1_t *r = &s->reg[k][j];
|
|
||||||
if (r->p) r->p->dp_max2 = 0; // reset ->dp_max2 as mm_set_parent() doesn't clear it; necessary with mm_update_dp_max()
|
|
||||||
r->subsc = 0; // this may not be necessary
|
|
||||||
r->n_sub = 0; // n_sub will be an underestimate as we don't see all the chains now, but it can't be accurate anyway
|
|
||||||
}
|
|
||||||
mm_hit_sort(km, &s->n_reg[k], s->reg[k], opt->alt_drop);
|
mm_hit_sort(km, &s->n_reg[k], s->reg[k], opt->alt_drop);
|
||||||
mm_set_parent(km, opt->mask_level, opt->mask_len, s->n_reg[k], s->reg[k], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
mm_set_parent(km, opt->mask_level, opt->mask_len, s->n_reg[k], s->reg[k], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
||||||
if (!(opt->flag & MM_F_ALL_CHAINS)) {
|
if (!(opt->flag & MM_F_ALL_CHAINS)) {
|
||||||
mm_select_sub(km, opt->pri_ratio, s->p->mi->k*2, opt->best_n, 0, opt->max_gap * 0.8, &s->n_reg[k], s->reg[k]);
|
mm_select_sub(km, opt->pri_ratio, s->p->mi->k*2, opt->best_n, &s->n_reg[k], s->reg[k]);
|
||||||
mm_set_sam_pri(s->n_reg[k], s->reg[k]);
|
mm_set_sam_pri(s->n_reg[k], s->reg[k]);
|
||||||
}
|
}
|
||||||
mm_set_mapq(km, s->n_reg[k], s->reg[k], opt->min_chain_score, opt->a, rep_len, !!(opt->flag & MM_F_SR));
|
mm_set_mapq(km, s->n_reg[k], s->reg[k], opt->min_chain_score, opt->a, rep_len, !!(opt->flag & MM_F_SR));
|
||||||
@@ -556,6 +705,7 @@ static void *worker_pipeline(void *shared, int step, void *in)
|
|||||||
else kt_for(p->n_threads, worker_for, in, ((step_t*)in)->n_frag);
|
else kt_for(p->n_threads, worker_for, in, ((step_t*)in)->n_frag);
|
||||||
return in;
|
return in;
|
||||||
} else if (step == 2) { // step 2: output
|
} else if (step == 2) { // step 2: output
|
||||||
|
|
||||||
void *km = 0;
|
void *km = 0;
|
||||||
step_t *s = (step_t*)in;
|
step_t *s = (step_t*)in;
|
||||||
const mm_idx_t *mi = p->mi;
|
const mm_idx_t *mi = p->mi;
|
||||||
@@ -564,6 +714,7 @@ static void *worker_pipeline(void *shared, int step, void *in)
|
|||||||
if ((p->opt->flag & MM_F_OUT_CS) && !(mm_dbg_flag & MM_DBG_NO_KALLOC)) km = km_init();
|
if ((p->opt->flag & MM_F_OUT_CS) && !(mm_dbg_flag & MM_DBG_NO_KALLOC)) km = km_init();
|
||||||
for (k = 0; k < s->n_frag; ++k) {
|
for (k = 0; k < s->n_frag; ++k) {
|
||||||
int seg_st = s->seg_off[k], seg_en = s->seg_off[k] + s->n_seg[k];
|
int seg_st = s->seg_off[k], seg_en = s->seg_off[k] + s->n_seg[k];
|
||||||
|
#ifndef DISABLE_OUTPUT
|
||||||
for (i = seg_st; i < seg_en; ++i) {
|
for (i = seg_st; i < seg_en; ++i) {
|
||||||
mm_bseq1_t *t = &s->seq[i];
|
mm_bseq1_t *t = &s->seq[i];
|
||||||
if (p->opt->split_prefix && p->n_parts == 0) { // then write to temporary files
|
if (p->opt->split_prefix && p->n_parts == 0) { // then write to temporary files
|
||||||
@@ -598,6 +749,7 @@ static void *worker_pipeline(void *shared, int step, void *in)
|
|||||||
mm_err_puts(p->str.s);
|
mm_err_puts(p->str.s);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
#endif
|
||||||
for (i = seg_st; i < seg_en; ++i) {
|
for (i = seg_st; i < seg_en; ++i) {
|
||||||
for (j = 0; j < s->n_reg[i]; ++j) free(s->reg[i][j].p);
|
for (j = 0; j < s->n_reg[i]; ++j) free(s->reg[i][j].p);
|
||||||
free(s->reg[i]);
|
free(s->reg[i]);
|
||||||
|
|||||||
@@ -1,3 +1,32 @@
|
|||||||
|
/* The MIT License
|
||||||
|
|
||||||
|
Copyright (c) 2018- Dana-Farber Cancer Institute
|
||||||
|
2017-2018 Broad Institute, Inc.
|
||||||
|
|
||||||
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
|
a copy of this software and associated documentation files (the
|
||||||
|
"Software"), to deal in the Software without restriction, including
|
||||||
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
|
the following conditions:
|
||||||
|
|
||||||
|
The above copyright notice and this permission notice shall be
|
||||||
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
SOFTWARE.
|
||||||
|
Modified Copyright (C) 2021 Intel Corporation
|
||||||
|
Contacts: Saurabh Kalikar <saurabh.kalikar@intel.com>;
|
||||||
|
Vasimuddin Md <vasimuddin.md@intel.com>; Sanchit Misra <sanchit.misra@intel.com>;
|
||||||
|
Chirag Jain <chirag@iisc.ac.in>; Heng Li <hli@jimmy.harvard.edu>
|
||||||
|
*/
|
||||||
#ifndef MINIMAP2_H
|
#ifndef MINIMAP2_H
|
||||||
#define MINIMAP2_H
|
#define MINIMAP2_H
|
||||||
|
|
||||||
@@ -5,45 +34,37 @@
|
|||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <sys/types.h>
|
#include <sys/types.h>
|
||||||
|
|
||||||
#define MM_VERSION "2.25-r1173"
|
#define MM_F_NO_DIAG 0x001 // no exact diagonal hit
|
||||||
|
#define MM_F_NO_DUAL 0x002 // skip pairs where query name is lexicographically larger than target name
|
||||||
#define MM_F_NO_DIAG (0x001LL) // no exact diagonal hit
|
#define MM_F_CIGAR 0x004
|
||||||
#define MM_F_NO_DUAL (0x002LL) // skip pairs where query name is lexicographically larger than target name
|
#define MM_F_OUT_SAM 0x008
|
||||||
#define MM_F_CIGAR (0x004LL)
|
#define MM_F_NO_QUAL 0x010
|
||||||
#define MM_F_OUT_SAM (0x008LL)
|
#define MM_F_OUT_CG 0x020
|
||||||
#define MM_F_NO_QUAL (0x010LL)
|
#define MM_F_OUT_CS 0x040
|
||||||
#define MM_F_OUT_CG (0x020LL)
|
#define MM_F_SPLICE 0x080 // splice mode
|
||||||
#define MM_F_OUT_CS (0x040LL)
|
#define MM_F_SPLICE_FOR 0x100 // match GT-AG
|
||||||
#define MM_F_SPLICE (0x080LL) // splice mode
|
#define MM_F_SPLICE_REV 0x200 // match CT-AC, the reverse complement of GT-AG
|
||||||
#define MM_F_SPLICE_FOR (0x100LL) // match GT-AG
|
#define MM_F_NO_LJOIN 0x400
|
||||||
#define MM_F_SPLICE_REV (0x200LL) // match CT-AC, the reverse complement of GT-AG
|
#define MM_F_OUT_CS_LONG 0x800
|
||||||
#define MM_F_NO_LJOIN (0x400LL)
|
#define MM_F_SR 0x1000
|
||||||
#define MM_F_OUT_CS_LONG (0x800LL)
|
#define MM_F_FRAG_MODE 0x2000
|
||||||
#define MM_F_SR (0x1000LL)
|
#define MM_F_NO_PRINT_2ND 0x4000
|
||||||
#define MM_F_FRAG_MODE (0x2000LL)
|
#define MM_F_2_IO_THREADS 0x8000
|
||||||
#define MM_F_NO_PRINT_2ND (0x4000LL)
|
#define MM_F_LONG_CIGAR 0x10000
|
||||||
#define MM_F_2_IO_THREADS (0x8000LL)
|
#define MM_F_INDEPEND_SEG 0x20000
|
||||||
#define MM_F_LONG_CIGAR (0x10000LL)
|
#define MM_F_SPLICE_FLANK 0x40000
|
||||||
#define MM_F_INDEPEND_SEG (0x20000LL)
|
#define MM_F_SOFTCLIP 0x80000
|
||||||
#define MM_F_SPLICE_FLANK (0x40000LL)
|
#define MM_F_FOR_ONLY 0x100000
|
||||||
#define MM_F_SOFTCLIP (0x80000LL)
|
#define MM_F_REV_ONLY 0x200000
|
||||||
#define MM_F_FOR_ONLY (0x100000LL)
|
#define MM_F_HEAP_SORT 0x400000
|
||||||
#define MM_F_REV_ONLY (0x200000LL)
|
#define MM_F_ALL_CHAINS 0x800000
|
||||||
#define MM_F_HEAP_SORT (0x400000LL)
|
#define MM_F_OUT_MD 0x1000000
|
||||||
#define MM_F_ALL_CHAINS (0x800000LL)
|
#define MM_F_COPY_COMMENT 0x2000000
|
||||||
#define MM_F_OUT_MD (0x1000000LL)
|
#define MM_F_EQX 0x4000000 // use =/X instead of M
|
||||||
#define MM_F_COPY_COMMENT (0x2000000LL)
|
#define MM_F_PAF_NO_HIT 0x8000000 // output unmapped reads to PAF
|
||||||
#define MM_F_EQX (0x4000000LL) // use =/X instead of M
|
#define MM_F_NO_END_FLT 0x10000000
|
||||||
#define MM_F_PAF_NO_HIT (0x8000000LL) // output unmapped reads to PAF
|
#define MM_F_HARD_MLEVEL 0x20000000
|
||||||
#define MM_F_NO_END_FLT (0x10000000LL)
|
#define MM_F_SAM_HIT_ONLY 0x40000000
|
||||||
#define MM_F_HARD_MLEVEL (0x20000000LL)
|
|
||||||
#define MM_F_SAM_HIT_ONLY (0x40000000LL)
|
|
||||||
#define MM_F_RMQ (0x80000000LL)
|
|
||||||
#define MM_F_QSTRAND (0x100000000LL)
|
|
||||||
#define MM_F_NO_INV (0x200000000LL)
|
|
||||||
#define MM_F_NO_HASH_NAME (0x400000000LL)
|
|
||||||
#define MM_F_SPLICE_OLD (0x800000000LL)
|
|
||||||
#define MM_F_SECONDARY_SEQ (0x1000000000LL) //output SEQ field for seqondary alignments using hard clipping
|
|
||||||
|
|
||||||
#define MM_I_HPC 0x1
|
#define MM_I_HPC 0x1
|
||||||
#define MM_I_NO_SEQ 0x2
|
#define MM_I_NO_SEQ 0x2
|
||||||
@@ -53,18 +74,6 @@
|
|||||||
|
|
||||||
#define MM_MAX_SEG 255
|
#define MM_MAX_SEG 255
|
||||||
|
|
||||||
#define MM_CIGAR_MATCH 0
|
|
||||||
#define MM_CIGAR_INS 1
|
|
||||||
#define MM_CIGAR_DEL 2
|
|
||||||
#define MM_CIGAR_N_SKIP 3
|
|
||||||
#define MM_CIGAR_SOFTCLIP 4
|
|
||||||
#define MM_CIGAR_HARDCLIP 5
|
|
||||||
#define MM_CIGAR_PADDING 6
|
|
||||||
#define MM_CIGAR_EQ_MATCH 7
|
|
||||||
#define MM_CIGAR_X_MISMATCH 8
|
|
||||||
|
|
||||||
#define MM_CIGAR_STR "MIDNSHP=XB"
|
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
extern "C" {
|
extern "C" {
|
||||||
#endif
|
#endif
|
||||||
@@ -113,7 +122,7 @@ typedef struct {
|
|||||||
int32_t mlen, blen; // seeded exact match length; seeded alignment block length
|
int32_t mlen, blen; // seeded exact match length; seeded alignment block length
|
||||||
int32_t n_sub; // number of suboptimal mappings
|
int32_t n_sub; // number of suboptimal mappings
|
||||||
int32_t score0; // initial chaining score (before chain merging/spliting)
|
int32_t score0; // initial chaining score (before chain merging/spliting)
|
||||||
uint32_t mapq:8, split:2, rev:1, inv:1, sam_pri:1, proper_frag:1, pe_thru:1, seg_split:1, seg_id:8, split_inv:1, is_alt:1, strand_retained:1, dummy:5;
|
uint32_t mapq:8, split:2, rev:1, inv:1, sam_pri:1, proper_frag:1, pe_thru:1, seg_split:1, seg_id:8, split_inv:1, is_alt:1, dummy:6;
|
||||||
uint32_t hash;
|
uint32_t hash;
|
||||||
float div;
|
float div;
|
||||||
mm_extra_t *p;
|
mm_extra_t *p;
|
||||||
@@ -133,23 +142,23 @@ typedef struct {
|
|||||||
|
|
||||||
int max_qlen; // max query length
|
int max_qlen; // max query length
|
||||||
|
|
||||||
int bw, bw_long; // bandwidth
|
int bw; // bandwidth
|
||||||
int max_gap, max_gap_ref; // break a chain if there are no minimizers in a max_gap window
|
int max_gap, max_gap_ref; // break a chain if there are no minimizers in a max_gap window
|
||||||
int max_frag_len;
|
int max_frag_len;
|
||||||
int max_chain_skip, max_chain_iter;
|
int max_chain_skip, max_chain_iter;
|
||||||
int min_cnt; // min number of minimizers on each chain
|
int min_cnt; // min number of minimizers on each chain
|
||||||
int min_chain_score; // min chaining score
|
int min_chain_score; // min chaining score
|
||||||
float chain_gap_scale;
|
float chain_gap_scale;
|
||||||
float chain_skip_scale;
|
|
||||||
int rmq_size_cap, rmq_inner_dist;
|
|
||||||
int rmq_rescue_size;
|
|
||||||
float rmq_rescue_ratio;
|
|
||||||
|
|
||||||
float mask_level;
|
float mask_level;
|
||||||
int mask_len;
|
int mask_len;
|
||||||
float pri_ratio;
|
float pri_ratio;
|
||||||
int best_n; // top best_n chains are subjected to DP alignment
|
int best_n; // top best_n chains are subjected to DP alignment
|
||||||
|
|
||||||
|
int max_join_long, max_join_short;
|
||||||
|
int min_join_flank_sc;
|
||||||
|
float min_join_flank_ratio;
|
||||||
|
|
||||||
float alt_drop;
|
float alt_drop;
|
||||||
|
|
||||||
int a, b, q, e, q2, e2; // matching score, mismatch, gap-open and gap-ext penalties
|
int a, b, q, e, q2, e2; // matching score, mismatch, gap-open and gap-ext penalties
|
||||||
@@ -163,21 +172,18 @@ typedef struct {
|
|||||||
int anchor_ext_len, anchor_ext_shift;
|
int anchor_ext_len, anchor_ext_shift;
|
||||||
float max_clip_ratio; // drop an alignment if BOTH ends are clipped above this ratio
|
float max_clip_ratio; // drop an alignment if BOTH ends are clipped above this ratio
|
||||||
|
|
||||||
int rank_min_len;
|
|
||||||
float rank_frac;
|
|
||||||
|
|
||||||
int pe_ori, pe_bonus;
|
int pe_ori, pe_bonus;
|
||||||
|
|
||||||
float mid_occ_frac; // only used by mm_mapopt_update(); see below
|
float mid_occ_frac; // only used by mm_mapopt_update(); see below
|
||||||
float q_occ_frac;
|
int32_t min_mid_occ;
|
||||||
int32_t min_mid_occ, max_mid_occ;
|
|
||||||
int32_t mid_occ; // ignore seeds with occurrences above this threshold
|
int32_t mid_occ; // ignore seeds with occurrences above this threshold
|
||||||
int32_t max_occ, max_max_occ, occ_dist;
|
int32_t max_occ;
|
||||||
int64_t mini_batch_size; // size of a batch of query bases to process in parallel
|
int64_t mini_batch_size; // size of a batch of query bases to process in parallel
|
||||||
int64_t max_sw_mat;
|
int64_t max_sw_mat;
|
||||||
int64_t cap_kalloc;
|
|
||||||
|
|
||||||
const char *split_prefix;
|
const char *split_prefix;
|
||||||
|
// Store minimizer hash to a file as key and list of values
|
||||||
|
int L_hash;
|
||||||
} mm_mapopt_t;
|
} mm_mapopt_t;
|
||||||
|
|
||||||
// index reader
|
// index reader
|
||||||
@@ -193,11 +199,6 @@ typedef struct {
|
|||||||
} mm_idx_reader_t;
|
} mm_idx_reader_t;
|
||||||
|
|
||||||
// memory buffer for thread-local storage during mapping
|
// memory buffer for thread-local storage during mapping
|
||||||
struct mm_tbuf_s {
|
|
||||||
void *km;
|
|
||||||
int rep_len, frag_gap;
|
|
||||||
};
|
|
||||||
|
|
||||||
typedef struct mm_tbuf_s mm_tbuf_t;
|
typedef struct mm_tbuf_s mm_tbuf_t;
|
||||||
|
|
||||||
// global variables
|
// global variables
|
||||||
@@ -297,6 +298,14 @@ mm_idx_t *mm_idx_load(FILE *fp);
|
|||||||
*/
|
*/
|
||||||
void mm_idx_dump(FILE *fp, const mm_idx_t *mi);
|
void mm_idx_dump(FILE *fp, const mm_idx_t *mi);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Store hash table from minimap2 index into a file
|
||||||
|
* @param f_name File name for output file
|
||||||
|
* @param mi minimap2 index
|
||||||
|
*/
|
||||||
|
void mm_idx_dump_hash(const char* f_name, const mm_idx_t *mi);
|
||||||
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Create an index from strings in memory
|
* Create an index from strings in memory
|
||||||
*
|
*
|
||||||
@@ -326,6 +335,21 @@ void mm_idx_stat(const mm_idx_t *idx);
|
|||||||
*/
|
*/
|
||||||
void mm_idx_destroy(mm_idx_t *mi);
|
void mm_idx_destroy(mm_idx_t *mi);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Destroy/deallocate an hash table index
|
||||||
|
*
|
||||||
|
* @param r minimap2 index
|
||||||
|
*/
|
||||||
|
void mm_idx_destroy_mm_hash(mm_idx_t *mi);
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Destroy/deallocate target sequences
|
||||||
|
*
|
||||||
|
* @param r minimap2 index
|
||||||
|
*/
|
||||||
|
void mm_idx_destroy_seq(mm_idx_t *mi);
|
||||||
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Initialize a thread-local buffer for mapping
|
* Initialize a thread-local buffer for mapping
|
||||||
*
|
*
|
||||||
|
|||||||
+56
-93
@@ -1,4 +1,4 @@
|
|||||||
.TH minimap2 1 "25 April 2023" "minimap2-2.25 (r1173)" "Bioinformatics tools"
|
.TH minimap2 1 "9 April 2021" "minimap2-2.18 (r1015)" "Bioinformatics tools"
|
||||||
.SH NAME
|
.SH NAME
|
||||||
.PP
|
.PP
|
||||||
minimap2 - mapping and alignment between collections of DNA sequences
|
minimap2 - mapping and alignment between collections of DNA sequences
|
||||||
@@ -77,21 +77,8 @@ SAM format.
|
|||||||
Minimizer k-mer length [15]
|
Minimizer k-mer length [15]
|
||||||
.TP
|
.TP
|
||||||
.BI -w \ INT
|
.BI -w \ INT
|
||||||
Minimizer window size [10]. A minimizer is the smallest k-mer
|
Minimizer window size [2/3 of k-mer length]. A minimizer is the smallest k-mer
|
||||||
in a window of w consecutive k-mers.
|
in a window of w consecutive k-mers.
|
||||||
.TP
|
|
||||||
.BI -j \ INT
|
|
||||||
Syncmer submer size [10]. Option
|
|
||||||
.B -j
|
|
||||||
and
|
|
||||||
.B -w
|
|
||||||
will override each: if
|
|
||||||
.B -w
|
|
||||||
is applied after
|
|
||||||
.BR -j ,
|
|
||||||
.B -j
|
|
||||||
will have no effect, and vice versa.
|
|
||||||
|
|
||||||
.TP
|
.TP
|
||||||
.B -H
|
.B -H
|
||||||
Use homopolymer-compressed (HPC) minimizers. An HPC sequence is constructed by
|
Use homopolymer-compressed (HPC) minimizers. An HPC sequence is constructed by
|
||||||
@@ -101,17 +88,16 @@ on the HPC sequence.
|
|||||||
.BI -I \ NUM
|
.BI -I \ NUM
|
||||||
Load at most
|
Load at most
|
||||||
.I NUM
|
.I NUM
|
||||||
target bases into RAM for indexing [8G]. If there are more than
|
target bases into RAM for indexing [4G]. If there are more than
|
||||||
.I NUM
|
.I NUM
|
||||||
bases in
|
bases in
|
||||||
.IR target.fa ,
|
.IR target.fa ,
|
||||||
minimap2 needs to read
|
minimap2 needs to read
|
||||||
.I query.fa
|
.I query.fa
|
||||||
multiple times to map it against each batch of target sequences. This would create a multi-part index.
|
multiple times to map it against each batch of target sequences.
|
||||||
.I NUM
|
.I NUM
|
||||||
may be ending with k/K/m/M/g/G. NB: mapping quality is incorrect given a
|
may be ending with k/K/m/M/g/G. NB: mapping quality is incorrect given a
|
||||||
multi-part index. See also option
|
multi-part index.
|
||||||
.BR --split-prefix .
|
|
||||||
.TP
|
.TP
|
||||||
.B --idx-no-seq
|
.B --idx-no-seq
|
||||||
Don't store target sequences in the index. It saves disk space and memory but
|
Don't store target sequences in the index. It saves disk space and memory but
|
||||||
@@ -159,38 +145,22 @@ or
|
|||||||
.B -xsr
|
.B -xsr
|
||||||
mode, which sets the threshold for a second round of seeding.
|
mode, which sets the threshold for a second round of seeding.
|
||||||
.TP
|
.TP
|
||||||
.BI -U \ INT1 [, INT2 ]
|
.BI --min-occ-floor \ INT
|
||||||
Lower and upper bounds of k-mer occurrences [10,1000000]. The final k-mer occurrence threshold is
|
Force minimap2 to always use k-mers occurring
|
||||||
.RI max{ INT1 ,\ min{ INT2 ,
|
|
||||||
.BR -f }}.
|
|
||||||
This option prevents excessively small or large
|
|
||||||
.B -f
|
|
||||||
estimated from the input reference. Available since r1034 and deprecating
|
|
||||||
.B --min-occ-floor
|
|
||||||
in earlier versions of minimap2.
|
|
||||||
.TP
|
|
||||||
.BI --q-occ-frac \ FLOAT
|
|
||||||
Discard a query minimizer if its occurrence is higher than
|
|
||||||
.I FLOAT
|
|
||||||
fraction of query minimizers and than the reference occurrence threshold
|
|
||||||
[0.01]. Set 0 to disable. Available since r1105.
|
|
||||||
.TP
|
|
||||||
.BI -e \ INT
|
|
||||||
Sample a high-frequency minimizer every
|
|
||||||
.I INT
|
.I INT
|
||||||
basepairs [500].
|
times or less [0]. In effect, the max occurrence threshold is set to
|
||||||
|
the
|
||||||
|
.RI max{ INT ,
|
||||||
|
.BR -f }.
|
||||||
.TP
|
.TP
|
||||||
.BI -g \ NUM
|
.BI -g \ INT
|
||||||
Stop chain enlongation if there are no minimizers within
|
Stop chain enlongation if there are no minimizers within
|
||||||
.IR NUM -bp
|
.IR INT -bp
|
||||||
[10k].
|
[10000].
|
||||||
.TP
|
.TP
|
||||||
.BI -r \ NUM1 [, NUM2 ]
|
.BI -r \ INT
|
||||||
Bandwidth for chaining and base alignment [500,20k].
|
Bandwidth used in chaining and DP-based alignment [500]. This option
|
||||||
.I NUM1
|
approximately controls the maximum gap size.
|
||||||
is used for initial chaining and alignment extension;
|
|
||||||
.I NUM2
|
|
||||||
for RMQ-based re-chaining and closing gaps in alignments.
|
|
||||||
.TP
|
.TP
|
||||||
.BI -n \ INT
|
.BI -n \ INT
|
||||||
Discard chains consisting of
|
Discard chains consisting of
|
||||||
@@ -264,10 +234,6 @@ Mark as secondary a chain that overlaps with a better chain by
|
|||||||
.I FLOAT
|
.I FLOAT
|
||||||
or more of the shorter chain [0.5]
|
or more of the shorter chain [0.5]
|
||||||
.TP
|
.TP
|
||||||
.BR --rmq = no | yes
|
|
||||||
Use the minigraph chaining algorithm [no]. The minigraph algorithm is better
|
|
||||||
for aligning contigs through long INDELs.
|
|
||||||
.TP
|
|
||||||
.B --hard-mask-level
|
.B --hard-mask-level
|
||||||
Honor option
|
Honor option
|
||||||
.B -M
|
.B -M
|
||||||
@@ -302,6 +268,10 @@ Disable the long gap patching heuristic. When this option is applied, the
|
|||||||
maximum alignment gap is mostly controlled by
|
maximum alignment gap is mostly controlled by
|
||||||
.BR -r .
|
.BR -r .
|
||||||
.TP
|
.TP
|
||||||
|
.BI --lj-min-ratio \ FLOAT
|
||||||
|
Fraction of query sequence length required to bridge a long gap [0.5]. A
|
||||||
|
smaller value helps to recover longer gaps, at the cost of more false gaps.
|
||||||
|
.TP
|
||||||
.B --splice
|
.B --splice
|
||||||
Enable the splice alignment mode.
|
Enable the splice alignment mode.
|
||||||
.TP
|
.TP
|
||||||
@@ -332,9 +302,6 @@ faster for short reads, but slower for long reads. [no]
|
|||||||
.B --no-pairing
|
.B --no-pairing
|
||||||
Treat two reads in a pair as independent reads. The mate related fields in SAM
|
Treat two reads in a pair as independent reads. The mate related fields in SAM
|
||||||
are still properly populated.
|
are still properly populated.
|
||||||
.TP
|
|
||||||
.B --no-hash-name
|
|
||||||
Produce the same alignment for identical sequences regardless of their sequence names.
|
|
||||||
.SS Alignment options
|
.SS Alignment options
|
||||||
.TP 10
|
.TP 10
|
||||||
.BI -A \ INT
|
.BI -A \ INT
|
||||||
@@ -445,11 +412,6 @@ alignment.
|
|||||||
.BI --cap-sw-mem \ NUM
|
.BI --cap-sw-mem \ NUM
|
||||||
Skip alignment if the DP matrix size is above
|
Skip alignment if the DP matrix size is above
|
||||||
.IR NUM .
|
.IR NUM .
|
||||||
Set 0 to disable [100m].
|
|
||||||
.TP
|
|
||||||
.BI --cap-kalloc \ NUM
|
|
||||||
Free thread-local kalloc memory reservoir if after the alignment the size of the reservoir above
|
|
||||||
.IR NUM .
|
|
||||||
Set 0 to disable [0].
|
Set 0 to disable [0].
|
||||||
.SS Input/output options
|
.SS Input/output options
|
||||||
.TP 10
|
.TP 10
|
||||||
@@ -561,47 +523,60 @@ Available
|
|||||||
.I STR
|
.I STR
|
||||||
are:
|
are:
|
||||||
.RS
|
.RS
|
||||||
.TP 10
|
.TP 8
|
||||||
.B map-ont
|
|
||||||
Align noisy long reads of ~10% error rate to a reference genome. This is the
|
|
||||||
default mode.
|
|
||||||
.TP
|
|
||||||
.B map-hifi
|
|
||||||
Align PacBio high-fidelity (HiFi) reads to a reference genome
|
|
||||||
.RB ( -k19
|
|
||||||
.B -w19 -U50,500 -g10k -A1 -B4 -O6,26 -E2,1
|
|
||||||
.BR -s200 ).
|
|
||||||
.TP
|
|
||||||
.B map-pb
|
.B map-pb
|
||||||
Align older PacBio continuous long (CLR) reads to a reference genome
|
PacBio/Oxford Nanopore read to reference mapping
|
||||||
.RB ( -Hk19 ).
|
.RB ( -Hk19 )
|
||||||
|
.TP
|
||||||
|
.B map-ont
|
||||||
|
Slightly more sensitive for Oxford Nanopore to reference mapping
|
||||||
|
.RB ( -k15 ).
|
||||||
|
For PacBio reads, HPC minimizers consistently leads to faster performance and
|
||||||
|
more sensitive results in comparison to normal minimizers. For Oxford Nanopore
|
||||||
|
data, normal minimizers are better, though not much. The effectiveness of HPC
|
||||||
|
is determined by the sequencing error mode.
|
||||||
.TP
|
.TP
|
||||||
.B asm5
|
.B asm5
|
||||||
Long assembly to reference mapping
|
Long assembly to reference mapping
|
||||||
.RB ( -k19
|
.RB ( -k19
|
||||||
.B -w19 -U50,500 --rmq -r1k,100k -g10k -A1 -B19 -O39,81 -E3,1 -s200 -z200
|
.B -w19 -A1 -B19 -O39,81 -E3,1 -s200 -z200 -N50
|
||||||
.BR -N50 ).
|
.BR --min-occ-floor=100 ).
|
||||||
Typically, the alignment will not extend to regions with 5% or higher sequence
|
Typically, the alignment will not extend to regions with 5% or higher sequence
|
||||||
divergence. Only use this preset if the average divergence is far below 5%.
|
divergence. Only use this preset if the average divergence is far below 5%.
|
||||||
.TP
|
.TP
|
||||||
.B asm10
|
.B asm10
|
||||||
Long assembly to reference mapping
|
Long assembly to reference mapping
|
||||||
.RB ( -k19
|
.RB ( -k19
|
||||||
.B -w19 -U50,500 --rmq -r1k,100k -g10k -A1 -B9 -O16,41 -E2,1 -s200 -z200
|
.B -w19 -A1 -B9 -O16,41 -E2,1 -s200 -z200 -N50
|
||||||
.BR -N50 ).
|
.BR --min-occ-floor=100 ).
|
||||||
Up to 10% sequence divergence.
|
Up to 10% sequence divergence.
|
||||||
.TP
|
.TP
|
||||||
.B asm20
|
.B asm20
|
||||||
Long assembly to reference mapping
|
Long assembly to reference mapping
|
||||||
.RB ( -k19
|
.RB ( -k19
|
||||||
.B -w10 -U50,500 --rmq -r1k,100k -g10k -A1 -B4 -O6,26 -E2,1 -s200 -z200
|
.B -w10 -A1 -B4 -O6,26 -E2,1 -s200 -z200 -N50
|
||||||
.BR -N50 ).
|
.BR --min-occ-floor=100 ).
|
||||||
Up to 20% sequence divergence.
|
Up to 20% sequence divergence.
|
||||||
.TP
|
.TP
|
||||||
|
.B ava-pb
|
||||||
|
PacBio all-vs-all overlap mapping
|
||||||
|
.RB ( -Hk19
|
||||||
|
.B -Xw5 -m100 -g10000 --max-chain-skip
|
||||||
|
.BR 25 ).
|
||||||
|
.TP
|
||||||
|
.B ava-ont
|
||||||
|
Oxford Nanopore all-vs-all overlap mapping
|
||||||
|
.RB ( -k15
|
||||||
|
.B -Xw5 -m100 -g10000 -r2000 --max-chain-skip
|
||||||
|
.BR 25 ).
|
||||||
|
Similarly, the major difference from
|
||||||
|
.B ava-pb
|
||||||
|
is that this preset is not using HPC minimizers.
|
||||||
|
.TP
|
||||||
.B splice
|
.B splice
|
||||||
Long-read spliced alignment
|
Long-read spliced alignment
|
||||||
.RB ( -k15
|
.RB ( -k15
|
||||||
.B -w5 --splice -g2k -G200k -A1 -B2 -O2,32 -E1,0 -C9 -z200 -ub --junc-bonus=9 --cap-sw-mem=0
|
.B -w5 --splice -g2000 -G200k -A1 -B2 -O2,32 -E1,0 -C9 -z200 -ub --junc-bonus=9
|
||||||
.BR --splice-flank=yes ).
|
.BR --splice-flank=yes ).
|
||||||
In the splice mode, 1) long deletions are taken as introns and represented as
|
In the splice mode, 1) long deletions are taken as introns and represented as
|
||||||
the
|
the
|
||||||
@@ -620,21 +595,9 @@ Long-read splice alignment for PacBio CCS reads
|
|||||||
.B sr
|
.B sr
|
||||||
Short single-end reads without splicing
|
Short single-end reads without splicing
|
||||||
.RB ( -k21
|
.RB ( -k21
|
||||||
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -b0 -r100 -p.5 -N20 -f1000,5000 -n2 -m25
|
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -r50 -p.5 -N20 -f1000,5000 -n2 -m20
|
||||||
.B -s40 -g100 -2K50m --heap-sort=yes
|
.B -s40 -g200 -2K50m --heap-sort=yes
|
||||||
.BR --secondary=no ).
|
.BR --secondary=no ).
|
||||||
.TP
|
|
||||||
.B ava-pb
|
|
||||||
PacBio CLR all-vs-all overlap mapping
|
|
||||||
.RB ( -Hk19
|
|
||||||
.B -Xw5 -e0
|
|
||||||
.BR -m100 ).
|
|
||||||
.TP
|
|
||||||
.B ava-ont
|
|
||||||
Oxford Nanopore all-vs-all overlap mapping
|
|
||||||
.RB ( -k15
|
|
||||||
.B -Xw5 -e0 -m100
|
|
||||||
.BR -r2k ).
|
|
||||||
.RE
|
.RE
|
||||||
.SS Miscellaneous options
|
.SS Miscellaneous options
|
||||||
.TP 10
|
.TP 10
|
||||||
|
|||||||
@@ -159,4 +159,3 @@ KRADIX_SORT_INIT(128x, mm128_t, sort_key_128x, 8)
|
|||||||
KRADIX_SORT_INIT(64, uint64_t, sort_key_64, 8)
|
KRADIX_SORT_INIT(64, uint64_t, sort_key_64, 8)
|
||||||
|
|
||||||
KSORT_INIT_GENERIC(uint32_t)
|
KSORT_INIT_GENERIC(uint32_t)
|
||||||
KSORT_INIT_GENERIC(uint64_t)
|
|
||||||
|
|||||||
-335
@@ -1,335 +0,0 @@
|
|||||||
#!/usr/bin/env k8
|
|
||||||
|
|
||||||
var getopt = function(args, ostr) {
|
|
||||||
var oli; // option letter list index
|
|
||||||
if (typeof(getopt.place) == 'undefined')
|
|
||||||
getopt.ind = 0, getopt.arg = null, getopt.place = -1;
|
|
||||||
if (getopt.place == -1) { // update scanning pointer
|
|
||||||
if (getopt.ind >= args.length || args[getopt.ind].charAt(getopt.place = 0) != '-') {
|
|
||||||
getopt.place = -1;
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
if (getopt.place + 1 < args[getopt.ind].length && args[getopt.ind].charAt(++getopt.place) == '-') { // found "--"
|
|
||||||
++getopt.ind;
|
|
||||||
getopt.place = -1;
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
var optopt = args[getopt.ind].charAt(getopt.place++); // character checked for validity
|
|
||||||
if (optopt == ':' || (oli = ostr.indexOf(optopt)) < 0) {
|
|
||||||
if (optopt == '-') return null; // if the user didn't specify '-' as an option, assume it means null.
|
|
||||||
if (getopt.place < 0) ++getopt.ind;
|
|
||||||
return '?';
|
|
||||||
}
|
|
||||||
if (oli+1 >= ostr.length || ostr.charAt(++oli) != ':') { // don't need argument
|
|
||||||
getopt.arg = null;
|
|
||||||
if (getopt.place < 0 || getopt.place >= args[getopt.ind].length) ++getopt.ind, getopt.place = -1;
|
|
||||||
} else { // need an argument
|
|
||||||
if (getopt.place >= 0 && getopt.place < args[getopt.ind].length)
|
|
||||||
getopt.arg = args[getopt.ind].substr(getopt.place);
|
|
||||||
else if (args.length <= ++getopt.ind) { // no arg
|
|
||||||
getopt.place = -1;
|
|
||||||
if (ostr.length > 0 && ostr.charAt(0) == ':') return ':';
|
|
||||||
return '?';
|
|
||||||
} else getopt.arg = args[getopt.ind]; // white space
|
|
||||||
getopt.place = -1;
|
|
||||||
++getopt.ind;
|
|
||||||
}
|
|
||||||
return optopt;
|
|
||||||
}
|
|
||||||
|
|
||||||
function read_fastx(file, buf)
|
|
||||||
{
|
|
||||||
if (file.readline(buf) < 0) return null;
|
|
||||||
var m, line = buf.toString();
|
|
||||||
if ((m = /^([>@])(\S+)/.exec(line)) == null)
|
|
||||||
throw Error("wrong fastx format");
|
|
||||||
var is_fq = (m[1] == '@');
|
|
||||||
var name = m[2];
|
|
||||||
if (file.readline(buf) < 0)
|
|
||||||
throw Error("missing sequence line");
|
|
||||||
var seq = buf.toString();
|
|
||||||
if (is_fq) { // skip quality
|
|
||||||
file.readline(buf);
|
|
||||||
file.readline(buf);
|
|
||||||
}
|
|
||||||
return [name, seq];
|
|
||||||
}
|
|
||||||
|
|
||||||
function filter_paf(a, opt)
|
|
||||||
{
|
|
||||||
if (a.length == 0) return;
|
|
||||||
var k = 0;
|
|
||||||
for (var i = 0; i < a.length; ++i) {
|
|
||||||
var ai = a[i];
|
|
||||||
if (ai[10] < opt.min_blen) continue;
|
|
||||||
if (ai[9] < ai[10] * opt.min_iden) continue;
|
|
||||||
var clip = [0, 0];
|
|
||||||
if (ai[4] == '+') {
|
|
||||||
clip[0] = ai[2] < ai[7]? ai[2] : ai[7];
|
|
||||||
clip[1] = ai[1] - ai[3] < ai[6] - ai[8]? ai[1] - ai[3] : ai[6] - ai[8];
|
|
||||||
} else {
|
|
||||||
clip[0] = ai[2] < ai[6] - ai[8]? ai[2] : ai[6] - ai[8];
|
|
||||||
clip[1] = ai[1] - ai[3] < ai[7]? ai[1] - ai[3] : ai[7];
|
|
||||||
}
|
|
||||||
if (clip[0] > opt.max_clip_len || clip[1] > opt.max_clip_len) continue;
|
|
||||||
a[k++] = ai;
|
|
||||||
}
|
|
||||||
a.length = k;
|
|
||||||
}
|
|
||||||
|
|
||||||
function parse_events(t, ev, id, buf)
|
|
||||||
{
|
|
||||||
var re = /(:(\d+))|(([\+\-\*])([a-z]+))/g;
|
|
||||||
var m, cs = null;
|
|
||||||
for (var j = 12; j < t.length; ++j) {
|
|
||||||
if ((m = /^cs:Z:(\S+)/.exec(t[j])) != null) {
|
|
||||||
cs = m[1].toLowerCase();
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (cs == null) {
|
|
||||||
warn("Warning: no cs tag for read '" + t[0] + "'");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
var st = t[2], en = t[3];
|
|
||||||
var x = st;
|
|
||||||
while ((m = re.exec(cs)) != null) {
|
|
||||||
var l;
|
|
||||||
if (m[2] != null) { // an identitcal match ":\d+"
|
|
||||||
l = parseInt(m[2]);
|
|
||||||
// [start, end, type, index, changed_base]
|
|
||||||
ev.push([x, x + l, 0, id]);
|
|
||||||
} else {
|
|
||||||
if (m[4] == '*') {
|
|
||||||
l = 1;
|
|
||||||
ev.push([x, x + 1, 1, id, m[5][0]]);
|
|
||||||
} else if (m[4] == '+') {
|
|
||||||
l = m[5].length;
|
|
||||||
ev.push([x, x + l, 2, id]);
|
|
||||||
} else if (m[4] == '-') {
|
|
||||||
l = 0;
|
|
||||||
ev.push([x, x, -1, id, m[5]]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
x += l;
|
|
||||||
}
|
|
||||||
if (x != en)
|
|
||||||
throw Error("inconsistent cs for read '" + t[0] + "'");
|
|
||||||
}
|
|
||||||
|
|
||||||
function find_het_sub(ev, a, opt)
|
|
||||||
{
|
|
||||||
var n = a.length, last0_i = -1, h = [], d = [];
|
|
||||||
for (var i = 0; i < n; ++i) h[i] = [], d[i] = [];
|
|
||||||
for (var i = 0; i < ev.length; ++i) {
|
|
||||||
if (ev[i][2] == 0) {
|
|
||||||
if (last0_i < 0 || ev[i][0] != ev[last0_i][0]) last0_i = i;
|
|
||||||
else if (ev[i][1] > ev[last0_i][1])
|
|
||||||
last0_i = i;
|
|
||||||
} else if (ev[i][2] == 1 && last0_i >= 0 && ev[i][0] < ev[last0_i][1]) {
|
|
||||||
if (ev[last0_i][1] - ev[last0_i][0] >= opt.min_mlen) {
|
|
||||||
if (opt.dbg_ev) print("EV", ev[last0_i].join("\t"), "|", ev[i].join("\t"));
|
|
||||||
var e0 = ev[last0_i], hl = h[e0[3]];
|
|
||||||
if (hl.length == 0 || hl[hl.length-1][0] != e0[0])
|
|
||||||
hl.push([e0[0], e0[1]]);
|
|
||||||
d[ev[i][3]].push([ev[i][0], e0[1] - e0[0]]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
var b = [];
|
|
||||||
for (var i = 0; i < n; ++i) {
|
|
||||||
var sh = 0, dh = 0;
|
|
||||||
for (var j = 0; j < h[i].length; ++j)
|
|
||||||
sh += h[i][j][1] - h[i][j][0];
|
|
||||||
for (var j = 0; j < d[i].length; ++j)
|
|
||||||
dh += d[i][j][1];
|
|
||||||
// [start, end, index, #consistent, lenConsistent, #conflictive, lenConflictive, identity, mlen]
|
|
||||||
b[i] = [a[i][2], a[i][3], i, h[i].length, sh, d[i].length, dh, a[i][9] / a[i][10], a[i][9]];
|
|
||||||
}
|
|
||||||
return b;
|
|
||||||
}
|
|
||||||
|
|
||||||
function flt_utg_for_ec(b, opt)
|
|
||||||
{
|
|
||||||
var k = 0;
|
|
||||||
for (var i = 0; i < b.length; ++i) {
|
|
||||||
var bi = b[i];
|
|
||||||
if (bi[4] == 0 && bi[6] == 0) b[k++] = bi; // entirely ambiguous
|
|
||||||
else if (bi[6] < (bi[4] + bi[6]) * opt.max_ratio0) b[k++] = bi;
|
|
||||||
}
|
|
||||||
b.length = k;
|
|
||||||
if (b.length == 0) return;
|
|
||||||
// find the longest contiguous segment
|
|
||||||
b.sort(function(x,y) { return x[0]-y[0] });
|
|
||||||
var st = b[0][0], en = b[0][1], max_st = 0, max_en = 0, max_max_en = en;
|
|
||||||
for (var i = 1; i < b.length; ++i) {
|
|
||||||
if (b[i][0] > en) {
|
|
||||||
if (en - st > max_en - max_st)
|
|
||||||
max_st = st, max_en = en;
|
|
||||||
st = b[i][0], en = b[i][1];
|
|
||||||
} else {
|
|
||||||
en = en > b[i][1]? en : b[i][1];
|
|
||||||
}
|
|
||||||
max_max_en = max_max_en > b[i][1]? max_max_en : b[i][1];
|
|
||||||
}
|
|
||||||
if (en - st > max_en - max_st)
|
|
||||||
max_st = st, max_en = en;
|
|
||||||
if (max_max_en != en || st != b[0][0]) {
|
|
||||||
var k = 0;
|
|
||||||
for (var i = 0; i < b.length; ++i)
|
|
||||||
if (b[i][0] < max_en && b[i][1] > max_st)
|
|
||||||
b[k++] = b[i];
|
|
||||||
b.length = k;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function flt_utg_for_bin(b, opt) // filter out alignments clearly on the wrong phase
|
|
||||||
{
|
|
||||||
var k = 0;
|
|
||||||
for (var i = 0; i < b.length; ++i) {
|
|
||||||
var bi = b[i];
|
|
||||||
if (bi[4] + bi[6] == 0 || bi[4] >= (bi[4] + bi[6]) * opt.max_ratio0) b[k++] = bi;
|
|
||||||
}
|
|
||||||
b.length = k;
|
|
||||||
}
|
|
||||||
|
|
||||||
function ec_core(b, n_a, ev, buf, ecb) // error correction
|
|
||||||
{
|
|
||||||
var intv = [];
|
|
||||||
for (var i = 0; i < n_a; ++i)
|
|
||||||
intv[i] = null;
|
|
||||||
intv[b[0][2]] = [b[0][0], b[0][1]];
|
|
||||||
var en = b[0][1];
|
|
||||||
for (var i = 1; i < b.length; ++i) {
|
|
||||||
if (b[i][1] <= en) continue;
|
|
||||||
intv[b[i][2]] = [en, b[i][1]];
|
|
||||||
en = b[i][1];
|
|
||||||
}
|
|
||||||
var k = 0;
|
|
||||||
ecb.capacity = buf.capacity;
|
|
||||||
ecb.length = 0;
|
|
||||||
for (var i = 0; i < ev.length; ++i) {
|
|
||||||
var e = ev[i], I = intv[e[3]];
|
|
||||||
if (I == null) continue;
|
|
||||||
if (e[0] >= I[0] && e[0] < I[1]) { // this is to reduce duplicated events around junctions
|
|
||||||
//print("X", e.join("\t"));
|
|
||||||
if (e[2] == 0) {
|
|
||||||
ecb.length += e[1] - e[0];
|
|
||||||
for (var j = e[0]; j < e[1]; ++j)
|
|
||||||
ecb[k++] = buf[j];
|
|
||||||
} else if (e[2] == 1) {
|
|
||||||
++ecb.length;
|
|
||||||
ecb[k++] = e[4].charCodeAt(0);
|
|
||||||
} else if (e[2] < 0) {
|
|
||||||
ecb.length += e[4].length;
|
|
||||||
for (var j = 0; j < e[4].length; ++j)
|
|
||||||
ecb[k++] = e[4].charCodeAt(j);
|
|
||||||
} // else, skip e[2] == 2
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (ecb.length != k) throw Error("BUG!");
|
|
||||||
}
|
|
||||||
|
|
||||||
function process_paf(a, opt, fp_seq, buf, ecb)
|
|
||||||
{
|
|
||||||
if (a.length == 0) return;
|
|
||||||
var len = a[0][1], name = a[0][0], seq = null;
|
|
||||||
if (len < opt.min_rlen) return;
|
|
||||||
if (fp_seq) {
|
|
||||||
var ret;
|
|
||||||
while ((ret = read_fastx(fp_seq, buf)) != null)
|
|
||||||
if (ret[0] == a[0][0])
|
|
||||||
break;
|
|
||||||
if (ret == null)
|
|
||||||
throw Error("failed to find sequence for read '" + a[0][0] + "'");
|
|
||||||
name = ret[0], seq = ret[1];
|
|
||||||
if (seq.length != len)
|
|
||||||
throw Error("inconsistent length for read '" + name + "'");
|
|
||||||
}
|
|
||||||
filter_paf(a, opt);
|
|
||||||
if (a.length == 0) return;
|
|
||||||
var ev = [];
|
|
||||||
for (var i = 0; i < a.length; ++i)
|
|
||||||
parse_events(a[i], ev, i, buf);
|
|
||||||
ev.sort(function(x,y) { return x[0]!=y[0]? x[0]-y[0] : x[2]-y[2] });
|
|
||||||
if (seq == null) print("SQ", name, a[0][1], a.length);
|
|
||||||
var b = find_het_sub(ev, a, opt);
|
|
||||||
if (opt.ec) flt_utg_for_ec(b, opt);
|
|
||||||
else flt_utg_for_bin(b, opt);
|
|
||||||
if (seq == null) {
|
|
||||||
for (var i = 0; i < b.length; ++i) {
|
|
||||||
var m, ai = a[b[i][2]], score = 0;
|
|
||||||
for (var j = 10; j < ai.length; ++j)
|
|
||||||
if ((m = /^AS:i:(\d+)/.exec(ai[j])) != null)
|
|
||||||
score = m[1];
|
|
||||||
print("TS", b[i][2], b[i][0], b[i][1], ai.slice(5, 9).join("\t"), b[i].slice(3, 7).join("\t"), score);
|
|
||||||
}
|
|
||||||
print("//");
|
|
||||||
} else { // error correction
|
|
||||||
if (b.length == 0) return;
|
|
||||||
buf.set(seq, 0);
|
|
||||||
ec_core(b, a.length, ev, buf, ecb);
|
|
||||||
print(">" + name);
|
|
||||||
print(ecb);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function main(args)
|
|
||||||
{
|
|
||||||
var c, opt = { min_rlen:5000, min_blen:5000, min_iden:0.8, min_mlen:5, max_clip_len:500, max_ratio0:0.25, dbg_ev:false };
|
|
||||||
while ((c = getopt(args, "l:b:d:m:c:r:E")) != null) {
|
|
||||||
if (c == 'l') opt.min_rlen = parseInt(getopt.arg);
|
|
||||||
else if (c == 'b') opt.min_blen = parseInt(getopt.arg);
|
|
||||||
else if (c == 'd') opt.min_iden = parseFloat(getopt.arg);
|
|
||||||
else if (c == 'm') opt.min_slen = parseInt(getopt.arg);
|
|
||||||
else if (c == 'c') opt.max_clip_len = parseInt(getopt.arg);
|
|
||||||
else if (c == 'r') opt.max_ratio0 = parseFloat(getopt.arg);
|
|
||||||
else if (c == 'E') opt.dbg_ev = true;
|
|
||||||
}
|
|
||||||
if (args.length - getopt.ind < 1) {
|
|
||||||
print("Usage: mmphase.js [options] <map-with-cs.paf> [reads.fa]");
|
|
||||||
print("Options:");
|
|
||||||
print(" -l INT min read length [" + opt.min_rlen + "]");
|
|
||||||
print(" -b INT min alignment length [" + opt.min_blen + "]");
|
|
||||||
print(" -d FLOAT min identity [" + opt.min_iden + "]");
|
|
||||||
print(" -s INT min match length [" + opt.min_mlen + "]");
|
|
||||||
print(" -c INT max clip length [" + opt.max_clip_len + "]");
|
|
||||||
print(" -r FLOAT initial ratio for haplotype filtering [" + opt.max_ratio0 + "]");
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
opt.ec = args.length - getopt.ind < 2? false : true;
|
|
||||||
if (!opt.ec) {
|
|
||||||
print("CC");
|
|
||||||
print("CC", "SQ qName qLen nHits");
|
|
||||||
print("CC", "TS index qStart qEnd tName tLen tStart tEnd nConsistent lCons nConflictive lConf score");
|
|
||||||
print("CC");
|
|
||||||
}
|
|
||||||
|
|
||||||
var buf = new Bytes(), ecb = new Bytes();
|
|
||||||
var fp_paf = new File(args[getopt.ind]);
|
|
||||||
var fp_seq = args.length - getopt.ind >= 2? new File(args[getopt.ind+1]) : null;
|
|
||||||
var a = [];
|
|
||||||
while (fp_paf.readline(buf) >= 0) {
|
|
||||||
var t = buf.toString().split("\t");
|
|
||||||
if (a.length > 0 && a[0][0] != t[0]) {
|
|
||||||
process_paf(a, opt, fp_seq, buf, ecb);
|
|
||||||
a.length = 0;
|
|
||||||
}
|
|
||||||
for (var i = 1; i <= 3; ++i) t[i] = parseInt(t[i]);
|
|
||||||
if (t[1] < opt.min_rlen) continue;
|
|
||||||
for (var i = 6; i <= 10; ++i) t[i] = parseInt(t[i]);
|
|
||||||
if (t[10] < opt.min_blen) continue;
|
|
||||||
a.push(t);
|
|
||||||
}
|
|
||||||
if (a.length >= 0)
|
|
||||||
process_paf(a, opt, fp_seq, buf, ecb);
|
|
||||||
if (fp_seq) fp_seq.close();
|
|
||||||
fp_paf.close();
|
|
||||||
ecb.destroy();
|
|
||||||
buf.destroy();
|
|
||||||
}
|
|
||||||
|
|
||||||
var ret = main(arguments)
|
|
||||||
exit(ret)
|
|
||||||
+61
-839
File diff suppressed because it is too large
Load Diff
@@ -13,7 +13,6 @@
|
|||||||
#define MM_DBG_PRINT_QNAME 0x2
|
#define MM_DBG_PRINT_QNAME 0x2
|
||||||
#define MM_DBG_PRINT_SEED 0x4
|
#define MM_DBG_PRINT_SEED 0x4
|
||||||
#define MM_DBG_PRINT_ALN_SEQ 0x8
|
#define MM_DBG_PRINT_ALN_SEQ 0x8
|
||||||
#define MM_DBG_PRINT_CHAIN 0x10
|
|
||||||
|
|
||||||
#define MM_SEED_LONG_JOIN (1ULL<<40)
|
#define MM_SEED_LONG_JOIN (1ULL<<40)
|
||||||
#define MM_SEED_IGNORE (1ULL<<41)
|
#define MM_SEED_IGNORE (1ULL<<41)
|
||||||
@@ -37,14 +36,6 @@
|
|||||||
extern "C" {
|
extern "C" {
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
typedef struct {
|
|
||||||
uint32_t n;
|
|
||||||
uint32_t q_pos;
|
|
||||||
uint32_t q_span:31, flt:1;
|
|
||||||
uint32_t seg_id:31, is_tandem:1;
|
|
||||||
const uint64_t *cr;
|
|
||||||
} mm_seed_t;
|
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int n_u, n_a;
|
int n_u, n_a;
|
||||||
uint64_t *u;
|
uint64_t *u;
|
||||||
@@ -61,44 +52,32 @@ uint32_t ks_ksmall_uint32_t(size_t n, uint32_t arr[], size_t kk);
|
|||||||
|
|
||||||
void mm_sketch(void *km, const char *str, int len, int w, int k, uint32_t rid, int is_hpc, mm128_v *p);
|
void mm_sketch(void *km, const char *str, int len, int w, int k, uint32_t rid, int is_hpc, mm128_v *p);
|
||||||
|
|
||||||
mm_seed_t *mm_collect_matches(void *km, int *_n_m, int qlen, int max_occ, int max_max_occ, int dist, const mm_idx_t *mi, const mm128_v *mv, int64_t *n_a, int *rep_len, int *n_mini_pos, uint64_t **mini_pos);
|
|
||||||
void mm_seed_mz_flt(void *km, mm128_v *mv, int32_t q_occ_max, float q_occ_frac);
|
|
||||||
|
|
||||||
double mm_event_identity(const mm_reg1_t *r);
|
|
||||||
int mm_write_sam_hdr(const mm_idx_t *mi, const char *rg, const char *ver, int argc, char *argv[]);
|
int mm_write_sam_hdr(const mm_idx_t *mi, const char *rg, const char *ver, int argc, char *argv[]);
|
||||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag);
|
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int opt_flag);
|
||||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len);
|
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int opt_flag, int rep_len);
|
||||||
void mm_write_sam(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int n_regs, const mm_reg1_t *regs);
|
void mm_write_sam(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int n_regs, const mm_reg1_t *regs);
|
||||||
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regs, const mm_reg1_t *const* regs, void *km, int64_t opt_flag);
|
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regs, const mm_reg1_t *const* regs, void *km, int opt_flag);
|
||||||
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int64_t opt_flag, int rep_len);
|
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int opt_flag, int rep_len);
|
||||||
|
|
||||||
void mm_idxopt_init(mm_idxopt_t *opt);
|
void mm_idxopt_init(mm_idxopt_t *opt);
|
||||||
const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n);
|
const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n);
|
||||||
int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f);
|
int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f);
|
||||||
int mm_idx_getseq2(const mm_idx_t *mi, int is_rev, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq);
|
mm128_t *mm_chain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float gap_scale, int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
||||||
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a);
|
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a);
|
||||||
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a, int is_qstrand);
|
|
||||||
|
|
||||||
mm128_t *mm_chain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float gap_scale,
|
|
||||||
int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
|
||||||
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
|
||||||
int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
|
||||||
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
|
||||||
int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
|
||||||
|
|
||||||
|
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a);
|
||||||
void mm_mark_alt(const mm_idx_t *mi, int n, mm_reg1_t *r);
|
void mm_mark_alt(const mm_idx_t *mi, int n, mm_reg1_t *r);
|
||||||
void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a, int is_qstrand);
|
void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a);
|
||||||
void mm_sync_regs(void *km, int n_regs, mm_reg1_t *regs);
|
void mm_sync_regs(void *km, int n_regs, mm_reg1_t *regs);
|
||||||
int mm_squeeze_a(void *km, int n_regs, mm_reg1_t *regs, mm128_t *a);
|
int mm_squeeze_a(void *km, int n_regs, mm_reg1_t *regs, mm128_t *a);
|
||||||
int mm_set_sam_pri(int n, mm_reg1_t *r);
|
int mm_set_sam_pri(int n, mm_reg1_t *r);
|
||||||
void mm_set_parent(void *km, float mask_level, int mask_len, int n, mm_reg1_t *r, int sub_diff, int hard_mask_level, float alt_diff_frac);
|
void mm_set_parent(void *km, float mask_level, int mask_len, int n, mm_reg1_t *r, int sub_diff, int hard_mask_level, float alt_diff_frac);
|
||||||
void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int check_strand, int min_strand_sc, int *n_, mm_reg1_t *r);
|
void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int *n_, mm_reg1_t *r);
|
||||||
void mm_select_sub_multi(void *km, float pri_ratio, float pri1, float pri2, int max_gap_ref, int min_diff, int best_n, int n_segs, const int *qlens, int *n_, mm_reg1_t *r);
|
void mm_select_sub_multi(void *km, float pri_ratio, float pri1, float pri2, int max_gap_ref, int min_diff, int best_n, int n_segs, const int *qlens, int *n_, mm_reg1_t *r);
|
||||||
int mm_filter_strand_retained(int n_regs, mm_reg1_t *r);
|
|
||||||
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs);
|
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs);
|
||||||
|
void mm_join_long(void *km, const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs, mm128_t *a);
|
||||||
void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r, float alt_diff_frac);
|
void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r, float alt_diff_frac);
|
||||||
void mm_set_mapq(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr);
|
void mm_set_mapq(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr);
|
||||||
void mm_update_dp_max(int qlen, int n_regs, mm_reg1_t *regs, float frac, int a, int b);
|
|
||||||
|
|
||||||
void mm_est_err(const mm_idx_t *mi, int qlen, int n_regs, mm_reg1_t *regs, const mm128_t *a, int32_t n, const uint64_t *mini_pos);
|
void mm_est_err(const mm_idx_t *mi, int qlen, int n_regs, mm_reg1_t *regs, const mm128_t *a, int32_t n, const uint64_t *mini_pos);
|
||||||
|
|
||||||
@@ -115,16 +94,6 @@ void mm_err_puts(const char *str);
|
|||||||
void mm_err_fwrite(const void *p, size_t size, size_t nitems, FILE *fp);
|
void mm_err_fwrite(const void *p, size_t size, size_t nitems, FILE *fp);
|
||||||
void mm_err_fread(void *p, size_t size, size_t nitems, FILE *fp);
|
void mm_err_fread(void *p, size_t size, size_t nitems, FILE *fp);
|
||||||
|
|
||||||
static inline float mg_log2(float x) // NB: this doesn't work when x<2
|
|
||||||
{
|
|
||||||
union { float f; uint32_t i; } z = { x };
|
|
||||||
float log_2 = ((z.i >> 23) & 255) - 128;
|
|
||||||
z.i &= ~(255 << 23);
|
|
||||||
z.i += 127 << 23;
|
|
||||||
log_2 += (-0.34484843f * z.f + 2.02466578f) * z.f - 0.67487759f;
|
|
||||||
return log_2;
|
|
||||||
}
|
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ void mm_idxopt_init(mm_idxopt_t *opt)
|
|||||||
opt->k = 15, opt->w = 10, opt->flag = 0;
|
opt->k = 15, opt->w = 10, opt->flag = 0;
|
||||||
opt->bucket_bits = 14;
|
opt->bucket_bits = 14;
|
||||||
opt->mini_batch_size = 50000000;
|
opt->mini_batch_size = 50000000;
|
||||||
opt->batch_size = 8000000000ULL;
|
opt->batch_size = 4000000000ULL;
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_mapopt_init(mm_mapopt_t *opt)
|
void mm_mapopt_init(mm_mapopt_t *opt)
|
||||||
@@ -16,32 +16,27 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
|||||||
memset(opt, 0, sizeof(mm_mapopt_t));
|
memset(opt, 0, sizeof(mm_mapopt_t));
|
||||||
opt->seed = 11;
|
opt->seed = 11;
|
||||||
opt->mid_occ_frac = 2e-4f;
|
opt->mid_occ_frac = 2e-4f;
|
||||||
opt->min_mid_occ = 10;
|
|
||||||
opt->max_mid_occ = 1000000;
|
|
||||||
opt->sdust_thres = 0; // no SDUST masking
|
opt->sdust_thres = 0; // no SDUST masking
|
||||||
opt->q_occ_frac = 0.01f;
|
|
||||||
|
|
||||||
opt->min_cnt = 3;
|
opt->min_cnt = 3;
|
||||||
opt->min_chain_score = 40;
|
opt->min_chain_score = 40;
|
||||||
opt->bw = 500, opt->bw_long = 20000;
|
opt->bw = 500;
|
||||||
opt->max_gap = 5000;
|
opt->max_gap = 5000;
|
||||||
opt->max_gap_ref = -1;
|
opt->max_gap_ref = -1;
|
||||||
opt->max_chain_skip = 25;
|
opt->max_chain_skip = 25;
|
||||||
opt->max_chain_iter = 5000;
|
opt->max_chain_iter = 5000;
|
||||||
opt->rmq_inner_dist = 1000;
|
opt->chain_gap_scale = 1.0f;
|
||||||
opt->rmq_size_cap = 100000;
|
|
||||||
opt->rmq_rescue_size = 1000;
|
|
||||||
opt->rmq_rescue_ratio = 0.1f;
|
|
||||||
opt->chain_gap_scale = 0.8f;
|
|
||||||
opt->chain_skip_scale = 0.0f;
|
|
||||||
opt->max_max_occ = 4095;
|
|
||||||
opt->occ_dist = 500;
|
|
||||||
|
|
||||||
opt->mask_level = 0.5f;
|
opt->mask_level = 0.5f;
|
||||||
opt->mask_len = INT_MAX;
|
opt->mask_len = INT_MAX;
|
||||||
opt->pri_ratio = 0.8f;
|
opt->pri_ratio = 0.8f;
|
||||||
opt->best_n = 5;
|
opt->best_n = 5;
|
||||||
|
|
||||||
|
opt->max_join_long = 20000;
|
||||||
|
opt->max_join_short = 2000;
|
||||||
|
opt->min_join_flank_sc = 1000;
|
||||||
|
opt->min_join_flank_ratio = 0.5f;
|
||||||
|
|
||||||
opt->alt_drop = 0.15f;
|
opt->alt_drop = 0.15f;
|
||||||
|
|
||||||
opt->a = 2, opt->b = 4, opt->q = 4, opt->e = 2, opt->q2 = 24, opt->e2 = 1;
|
opt->a = 2, opt->b = 4, opt->q = 4, opt->e = 2, opt->q2 = 24, opt->e2 = 1;
|
||||||
@@ -53,11 +48,6 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
|||||||
opt->anchor_ext_len = 20, opt->anchor_ext_shift = 6;
|
opt->anchor_ext_len = 20, opt->anchor_ext_shift = 6;
|
||||||
opt->max_clip_ratio = 1.0f;
|
opt->max_clip_ratio = 1.0f;
|
||||||
opt->mini_batch_size = 500000000;
|
opt->mini_batch_size = 500000000;
|
||||||
opt->max_sw_mat = 100000000;
|
|
||||||
opt->cap_kalloc = 1000000000;
|
|
||||||
|
|
||||||
opt->rank_min_len = 500;
|
|
||||||
opt->rank_frac = 0.9f;
|
|
||||||
|
|
||||||
opt->pe_ori = 0; // FF
|
opt->pe_ori = 0; // FF
|
||||||
opt->pe_bonus = 33;
|
opt->pe_bonus = 33;
|
||||||
@@ -67,14 +57,10 @@ void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
|||||||
{
|
{
|
||||||
if ((opt->flag & MM_F_SPLICE_FOR) || (opt->flag & MM_F_SPLICE_REV))
|
if ((opt->flag & MM_F_SPLICE_FOR) || (opt->flag & MM_F_SPLICE_REV))
|
||||||
opt->flag |= MM_F_SPLICE;
|
opt->flag |= MM_F_SPLICE;
|
||||||
if (opt->mid_occ <= 0) {
|
if (opt->mid_occ <= 0)
|
||||||
opt->mid_occ = mm_idx_cal_max_occ(mi, opt->mid_occ_frac);
|
opt->mid_occ = mm_idx_cal_max_occ(mi, opt->mid_occ_frac);
|
||||||
if (opt->mid_occ < opt->min_mid_occ)
|
if (opt->mid_occ < opt->min_mid_occ)
|
||||||
opt->mid_occ = opt->min_mid_occ;
|
opt->mid_occ = opt->min_mid_occ;
|
||||||
if (opt->max_mid_occ > opt->min_mid_occ && opt->mid_occ > opt->max_mid_occ)
|
|
||||||
opt->mid_occ = opt->max_mid_occ;
|
|
||||||
}
|
|
||||||
if (opt->bw_long < opt->bw) opt->bw_long = opt->bw;
|
|
||||||
if (mm_verbose >= 3)
|
if (mm_verbose >= 3)
|
||||||
fprintf(stderr, "[M::%s::%.3f*%.2f] mid_occ = %d\n", __func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), opt->mid_occ);
|
fprintf(stderr, "[M::%s::%.3f*%.2f] mid_occ = %d\n", __func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), opt->mid_occ);
|
||||||
}
|
}
|
||||||
@@ -82,7 +68,7 @@ void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
|||||||
void mm_mapopt_max_intron_len(mm_mapopt_t *opt, int max_intron_len)
|
void mm_mapopt_max_intron_len(mm_mapopt_t *opt, int max_intron_len)
|
||||||
{
|
{
|
||||||
if ((opt->flag & MM_F_SPLICE) && max_intron_len > 0)
|
if ((opt->flag & MM_F_SPLICE) && max_intron_len > 0)
|
||||||
opt->max_gap_ref = opt->bw = opt->bw_long = max_intron_len;
|
opt->max_gap_ref = opt->bw = max_intron_len;
|
||||||
}
|
}
|
||||||
|
|
||||||
int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||||
@@ -90,44 +76,37 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
|||||||
if (preset == 0) {
|
if (preset == 0) {
|
||||||
mm_idxopt_init(io);
|
mm_idxopt_init(io);
|
||||||
mm_mapopt_init(mo);
|
mm_mapopt_init(mo);
|
||||||
} else if (strcmp(preset, "map-ont") == 0) { // this is the same as the default
|
|
||||||
} else if (strcmp(preset, "ava-ont") == 0) {
|
} else if (strcmp(preset, "ava-ont") == 0) {
|
||||||
io->flag = 0, io->k = 15, io->w = 5;
|
io->flag = 0, io->k = 15, io->w = 5;
|
||||||
mo->flag |= MM_F_ALL_CHAINS | MM_F_NO_DIAG | MM_F_NO_DUAL | MM_F_NO_LJOIN;
|
mo->flag |= MM_F_ALL_CHAINS | MM_F_NO_DIAG | MM_F_NO_DUAL | MM_F_NO_LJOIN;
|
||||||
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_chain_skip = 25;
|
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_gap = 10000, mo->max_chain_skip = 25;
|
||||||
mo->bw = mo->bw_long = 2000;
|
mo->bw = 2000;
|
||||||
mo->occ_dist = 0;
|
|
||||||
} else if (strcmp(preset, "map10k") == 0 || strcmp(preset, "map-pb") == 0) {
|
|
||||||
io->flag |= MM_I_HPC, io->k = 19;
|
|
||||||
} else if (strcmp(preset, "ava-pb") == 0) {
|
} else if (strcmp(preset, "ava-pb") == 0) {
|
||||||
io->flag |= MM_I_HPC, io->k = 19, io->w = 5;
|
io->flag |= MM_I_HPC, io->k = 19, io->w = 5;
|
||||||
mo->flag |= MM_F_ALL_CHAINS | MM_F_NO_DIAG | MM_F_NO_DUAL | MM_F_NO_LJOIN;
|
mo->flag |= MM_F_ALL_CHAINS | MM_F_NO_DIAG | MM_F_NO_DUAL | MM_F_NO_LJOIN;
|
||||||
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_chain_skip = 25;
|
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_gap = 10000, mo->max_chain_skip = 25;
|
||||||
mo->bw_long = mo->bw;
|
} else if (strcmp(preset, "map10k") == 0 || strcmp(preset, "map-pb") == 0) {
|
||||||
mo->occ_dist = 0;
|
io->flag |= MM_I_HPC, io->k = 19;
|
||||||
} else if (strcmp(preset, "map-hifi") == 0 || strcmp(preset, "map-ccs") == 0) {
|
} else if (strcmp(preset, "map-ont") == 0) {
|
||||||
|
io->flag = 0, io->k = 15;
|
||||||
|
} else if (strcmp(preset, "asm5") == 0) {
|
||||||
io->flag = 0, io->k = 19, io->w = 19;
|
io->flag = 0, io->k = 19, io->w = 19;
|
||||||
mo->max_gap = 10000;
|
mo->a = 1, mo->b = 19, mo->q = 39, mo->q2 = 81, mo->e = 3, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
||||||
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1;
|
mo->min_mid_occ = 100;
|
||||||
mo->occ_dist = 500;
|
|
||||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
|
||||||
mo->min_dp_max = 200;
|
mo->min_dp_max = 200;
|
||||||
} else if (strncmp(preset, "asm", 3) == 0) {
|
mo->best_n = 50;
|
||||||
|
} else if (strcmp(preset, "asm10") == 0) {
|
||||||
io->flag = 0, io->k = 19, io->w = 19;
|
io->flag = 0, io->k = 19, io->w = 19;
|
||||||
mo->bw = 1000, mo->bw_long = 100000;
|
mo->a = 1, mo->b = 9, mo->q = 16, mo->q2 = 41, mo->e = 2, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
||||||
mo->max_gap = 10000;
|
mo->min_mid_occ = 100;
|
||||||
mo->flag |= MM_F_RMQ;
|
mo->min_dp_max = 200;
|
||||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
mo->best_n = 50;
|
||||||
|
} else if (strcmp(preset, "asm20") == 0) {
|
||||||
|
io->flag = 0, io->k = 19, io->w = 10;
|
||||||
|
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
||||||
|
mo->min_mid_occ = 100;
|
||||||
mo->min_dp_max = 200;
|
mo->min_dp_max = 200;
|
||||||
mo->best_n = 50;
|
mo->best_n = 50;
|
||||||
if (strcmp(preset, "asm5") == 0) {
|
|
||||||
mo->a = 1, mo->b = 19, mo->q = 39, mo->q2 = 81, mo->e = 3, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
|
||||||
} else if (strcmp(preset, "asm10") == 0) {
|
|
||||||
mo->a = 1, mo->b = 9, mo->q = 16, mo->q2 = 41, mo->e = 2, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
|
||||||
} else if (strcmp(preset, "asm20") == 0) {
|
|
||||||
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
|
||||||
io->w = 10;
|
|
||||||
} else return -1;
|
|
||||||
} else if (strcmp(preset, "short") == 0 || strcmp(preset, "sr") == 0) {
|
} else if (strcmp(preset, "short") == 0 || strcmp(preset, "sr") == 0) {
|
||||||
io->flag = 0, io->k = 21, io->w = 11;
|
io->flag = 0, io->k = 21, io->w = 11;
|
||||||
mo->flag |= MM_F_SR | MM_F_FRAG_MODE | MM_F_NO_PRINT_2ND | MM_F_2_IO_THREADS | MM_F_HEAP_SORT;
|
mo->flag |= MM_F_SR | MM_F_FRAG_MODE | MM_F_NO_PRINT_2ND | MM_F_2_IO_THREADS | MM_F_HEAP_SORT;
|
||||||
@@ -137,7 +116,7 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
|||||||
mo->end_bonus = 10;
|
mo->end_bonus = 10;
|
||||||
mo->max_frag_len = 800;
|
mo->max_frag_len = 800;
|
||||||
mo->max_gap = 100;
|
mo->max_gap = 100;
|
||||||
mo->bw = mo->bw_long = 100;
|
mo->bw = 100;
|
||||||
mo->pri_ratio = 0.5f;
|
mo->pri_ratio = 0.5f;
|
||||||
mo->min_cnt = 2;
|
mo->min_cnt = 2;
|
||||||
mo->min_chain_score = 25;
|
mo->min_chain_score = 25;
|
||||||
@@ -149,8 +128,7 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
|||||||
} else if (strncmp(preset, "splice", 6) == 0 || strcmp(preset, "cdna") == 0) {
|
} else if (strncmp(preset, "splice", 6) == 0 || strcmp(preset, "cdna") == 0) {
|
||||||
io->flag = 0, io->k = 15, io->w = 5;
|
io->flag = 0, io->k = 15, io->w = 5;
|
||||||
mo->flag |= MM_F_SPLICE | MM_F_SPLICE_FOR | MM_F_SPLICE_REV | MM_F_SPLICE_FLANK;
|
mo->flag |= MM_F_SPLICE | MM_F_SPLICE_FOR | MM_F_SPLICE_REV | MM_F_SPLICE_FLANK;
|
||||||
mo->max_sw_mat = 0;
|
mo->max_gap = 2000, mo->max_gap_ref = mo->bw = 200000;
|
||||||
mo->max_gap = 2000, mo->max_gap_ref = mo->bw = mo->bw_long = 200000;
|
|
||||||
mo->a = 1, mo->b = 2, mo->q = 2, mo->e = 1, mo->q2 = 32, mo->e2 = 0;
|
mo->a = 1, mo->b = 2, mo->q = 2, mo->e = 1, mo->q2 = 32, mo->e2 = 0;
|
||||||
mo->noncan = 9;
|
mo->noncan = 9;
|
||||||
mo->junc_bonus = 9;
|
mo->junc_bonus = 9;
|
||||||
@@ -163,16 +141,6 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
|||||||
|
|
||||||
int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
||||||
{
|
{
|
||||||
if (mo->bw > mo->bw_long) {
|
|
||||||
if (mm_verbose >= 1)
|
|
||||||
fprintf(stderr, "[ERROR]\033[1;31m with '-rNUM1,NUM2', NUM1 (%d) can't be larger than NUM2 (%d)\033[0m\n", mo->bw, mo->bw_long);
|
|
||||||
return -8;
|
|
||||||
}
|
|
||||||
if ((mo->flag & MM_F_RMQ) && (mo->flag & (MM_F_SR|MM_F_SPLICE))) {
|
|
||||||
if (mm_verbose >= 1)
|
|
||||||
fprintf(stderr, "[ERROR]\033[1;31m --rmq doesn't work with --sr or --splice\033[0m\n");
|
|
||||||
return -7;
|
|
||||||
}
|
|
||||||
if (mo->split_prefix && (mo->flag & (MM_F_OUT_CS|MM_F_OUT_MD))) {
|
if (mo->split_prefix && (mo->flag & (MM_F_OUT_CS|MM_F_OUT_MD))) {
|
||||||
if (mm_verbose >= 1)
|
if (mm_verbose >= 1)
|
||||||
fprintf(stderr, "[ERROR]\033[1;31m --cs or --MD doesn't work with --split-prefix\033[0m\n");
|
fprintf(stderr, "[ERROR]\033[1;31m --cs or --MD doesn't work with --split-prefix\033[0m\n");
|
||||||
@@ -225,10 +193,5 @@ int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
|||||||
fprintf(stderr, "[ERROR]\033[1;31m -X/-P and --secondary=no can't be applied at the same time\033[0m\n");
|
fprintf(stderr, "[ERROR]\033[1;31m -X/-P and --secondary=no can't be applied at the same time\033[0m\n");
|
||||||
return -5;
|
return -5;
|
||||||
}
|
}
|
||||||
if ((mo->flag & MM_F_QSTRAND) && ((mo->flag & (MM_F_OUT_SAM|MM_F_SPLICE|MM_F_FRAG_MODE)) || (io->flag & MM_I_HPC))) {
|
|
||||||
if (mm_verbose >= 1)
|
|
||||||
fprintf(stderr, "[ERROR]\033[1;31m --qstrand doesn't work with -a, -H, --frag or --splice\033[0m\n");
|
|
||||||
return -5;
|
|
||||||
}
|
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,2 +0,0 @@
|
|||||||
[build-system]
|
|
||||||
requires = ["setuptools", "wheel", "Cython"]
|
|
||||||
+1
-1
@@ -144,7 +144,7 @@ properties:
|
|||||||
* **mlen**: length of the matching bases in the alignment, excluding ambiguous
|
* **mlen**: length of the matching bases in the alignment, excluding ambiguous
|
||||||
base matches.
|
base matches.
|
||||||
|
|
||||||
* **NM**: number of mismatches, gaps and ambiguous positions in the alignment
|
* **NM**: number of mismatches, gaps and ambiguous poistions in the alignment
|
||||||
|
|
||||||
* **trans_strand**: transcript strand. +1 if on the forward strand; -1 if on the
|
* **trans_strand**: transcript strand. +1 if on the forward strand; -1 if on the
|
||||||
reverse strand; 0 if unknown
|
reverse strand; 0 if unknown
|
||||||
|
|||||||
+4
-18
@@ -13,28 +13,22 @@ cdef extern from "minimap.h":
|
|||||||
int64_t flag
|
int64_t flag
|
||||||
int seed
|
int seed
|
||||||
int sdust_thres
|
int sdust_thres
|
||||||
|
|
||||||
int max_qlen
|
int max_qlen
|
||||||
|
int bw
|
||||||
int bw, bw_long
|
|
||||||
int max_gap, max_gap_ref
|
int max_gap, max_gap_ref
|
||||||
int max_frag_len
|
int max_frag_len
|
||||||
int max_chain_skip, max_chain_iter
|
int max_chain_skip, max_chain_iter
|
||||||
int min_cnt
|
int min_cnt
|
||||||
int min_chain_score
|
int min_chain_score
|
||||||
float chain_gap_scale
|
float chain_gap_scale
|
||||||
float chain_skip_scale
|
|
||||||
int rmq_size_cap, rmq_inner_dist
|
|
||||||
int rmq_rescue_size
|
|
||||||
float rmq_rescue_ratio
|
|
||||||
|
|
||||||
float mask_level
|
float mask_level
|
||||||
int mask_len
|
int mask_len
|
||||||
float pri_ratio
|
float pri_ratio
|
||||||
int best_n
|
int best_n
|
||||||
|
int max_join_long, max_join_short
|
||||||
|
int min_join_flank_sc
|
||||||
|
float min_join_flank_ratio
|
||||||
float alt_drop
|
float alt_drop
|
||||||
|
|
||||||
int a, b, q, e, q2, e2
|
int a, b, q, e, q2, e2
|
||||||
int sc_ambi
|
int sc_ambi
|
||||||
int noncan
|
int noncan
|
||||||
@@ -45,21 +39,13 @@ cdef extern from "minimap.h":
|
|||||||
int min_ksw_len
|
int min_ksw_len
|
||||||
int anchor_ext_len, anchor_ext_shift
|
int anchor_ext_len, anchor_ext_shift
|
||||||
float max_clip_ratio
|
float max_clip_ratio
|
||||||
|
|
||||||
int rank_min_len
|
|
||||||
float rank_frac
|
|
||||||
|
|
||||||
int pe_ori, pe_bonus
|
int pe_ori, pe_bonus
|
||||||
|
|
||||||
float mid_occ_frac
|
float mid_occ_frac
|
||||||
float q_occ_frac
|
|
||||||
int32_t min_mid_occ
|
int32_t min_mid_occ
|
||||||
int32_t mid_occ
|
int32_t mid_occ
|
||||||
int32_t max_occ
|
int32_t max_occ
|
||||||
int64_t mini_batch_size
|
int64_t mini_batch_size
|
||||||
int64_t max_sw_mat
|
int64_t max_sw_mat
|
||||||
int64_t cap_kalloc
|
|
||||||
|
|
||||||
const char *split_prefix
|
const char *split_prefix
|
||||||
|
|
||||||
int mm_set_opt(char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
int mm_set_opt(char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||||
|
|||||||
+2
-4
@@ -3,7 +3,7 @@ from libc.stdlib cimport free
|
|||||||
cimport cmappy
|
cimport cmappy
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
__version__ = '2.25'
|
__version__ = '2.18'
|
||||||
|
|
||||||
cmappy.mm_reset_timer()
|
cmappy.mm_reset_timer()
|
||||||
|
|
||||||
@@ -82,7 +82,7 @@ cdef class Alignment:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def cigar_str(self):
|
def cigar_str(self):
|
||||||
return "".join(map(lambda x: str(x[0]) + 'MIDNSHP=XB'[x[1]], self._cigar))
|
return "".join(map(lambda x: str(x[0]) + 'MIDNSH'[x[1]], self._cigar))
|
||||||
|
|
||||||
def __str__(self):
|
def __str__(self):
|
||||||
if self._strand > 0: strand = '+'
|
if self._strand > 0: strand = '+'
|
||||||
@@ -172,7 +172,6 @@ cdef class Aligner:
|
|||||||
cdef cmappy.mm_mapopt_t map_opt
|
cdef cmappy.mm_mapopt_t map_opt
|
||||||
|
|
||||||
if self._idx == NULL: return
|
if self._idx == NULL: return
|
||||||
if ((self.map_opt.flag & 4) and (self._idx.flag & 2)): return
|
|
||||||
map_opt = self.map_opt
|
map_opt = self.map_opt
|
||||||
if max_frag_len is not None: map_opt.max_frag_len = max_frag_len
|
if max_frag_len is not None: map_opt.max_frag_len = max_frag_len
|
||||||
if extra_flags is not None: map_opt.flag |= extra_flags
|
if extra_flags is not None: map_opt.flag |= extra_flags
|
||||||
@@ -218,7 +217,6 @@ cdef class Aligner:
|
|||||||
cdef int l
|
cdef int l
|
||||||
cdef char *s
|
cdef char *s
|
||||||
if self._idx == NULL: return
|
if self._idx == NULL: return
|
||||||
if ((self.map_opt.flag & 4) and (self._idx.flag & 2)): return
|
|
||||||
s = cmappy.mappy_fetch_seq(self._idx, name.encode(), start, end, &l)
|
s = cmappy.mappy_fetch_seq(self._idx, name.encode(), start, end, &l)
|
||||||
if l == 0: return None
|
if l == 0: return None
|
||||||
r = s[:l] if isinstance(s, str) else s[:l].decode()
|
r = s[:l] if isinstance(s, str) else s[:l].decode()
|
||||||
|
|||||||
@@ -1,131 +0,0 @@
|
|||||||
#include "mmpriv.h"
|
|
||||||
#include "kalloc.h"
|
|
||||||
#include "ksort.h"
|
|
||||||
|
|
||||||
void mm_seed_mz_flt(void *km, mm128_v *mv, int32_t q_occ_max, float q_occ_frac)
|
|
||||||
{
|
|
||||||
mm128_t *a;
|
|
||||||
size_t i, j, st;
|
|
||||||
if (mv->n <= q_occ_max || q_occ_frac <= 0.0f || q_occ_max <= 0) return;
|
|
||||||
a = Kmalloc(km, mm128_t, mv->n);
|
|
||||||
for (i = 0; i < mv->n; ++i)
|
|
||||||
a[i].x = mv->a[i].x, a[i].y = i;
|
|
||||||
radix_sort_128x(a, a + mv->n);
|
|
||||||
for (st = 0, i = 1; i <= mv->n; ++i) {
|
|
||||||
if (i == mv->n || a[i].x != a[st].x) {
|
|
||||||
int32_t cnt = i - st;
|
|
||||||
if (cnt > q_occ_max && cnt > mv->n * q_occ_frac)
|
|
||||||
for (j = st; j < i; ++j)
|
|
||||||
mv->a[a[j].y].x = 0;
|
|
||||||
st = i;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
kfree(km, a);
|
|
||||||
for (i = j = 0; i < mv->n; ++i)
|
|
||||||
if (mv->a[i].x != 0)
|
|
||||||
mv->a[j++] = mv->a[i];
|
|
||||||
mv->n = j;
|
|
||||||
}
|
|
||||||
|
|
||||||
mm_seed_t *mm_seed_collect_all(void *km, const mm_idx_t *mi, const mm128_v *mv, int32_t *n_m_)
|
|
||||||
{
|
|
||||||
mm_seed_t *m;
|
|
||||||
size_t i;
|
|
||||||
int32_t k;
|
|
||||||
m = (mm_seed_t*)kmalloc(km, mv->n * sizeof(mm_seed_t));
|
|
||||||
for (i = k = 0; i < mv->n; ++i) {
|
|
||||||
const uint64_t *cr;
|
|
||||||
mm_seed_t *q;
|
|
||||||
mm128_t *p = &mv->a[i];
|
|
||||||
uint32_t q_pos = (uint32_t)p->y, q_span = p->x & 0xff;
|
|
||||||
int t;
|
|
||||||
cr = mm_idx_get(mi, p->x>>8, &t);
|
|
||||||
if (t == 0) continue;
|
|
||||||
q = &m[k++];
|
|
||||||
q->q_pos = q_pos, q->q_span = q_span, q->cr = cr, q->n = t, q->seg_id = p->y >> 32;
|
|
||||||
q->is_tandem = q->flt = 0;
|
|
||||||
if (i > 0 && p->x>>8 == mv->a[i - 1].x>>8) q->is_tandem = 1;
|
|
||||||
if (i < mv->n - 1 && p->x>>8 == mv->a[i + 1].x>>8) q->is_tandem = 1;
|
|
||||||
}
|
|
||||||
*n_m_ = k;
|
|
||||||
return m;
|
|
||||||
}
|
|
||||||
|
|
||||||
#define MAX_MAX_HIGH_OCC 128
|
|
||||||
|
|
||||||
void mm_seed_select(int32_t n, mm_seed_t *a, int len, int max_occ, int max_max_occ, int dist)
|
|
||||||
{ // for high-occ minimizers, choose up to max_high_occ in each high-occ streak
|
|
||||||
extern void ks_heapdown_uint64_t(size_t i, size_t n, uint64_t*);
|
|
||||||
extern void ks_heapmake_uint64_t(size_t n, uint64_t*);
|
|
||||||
int32_t i, last0, m;
|
|
||||||
uint64_t b[MAX_MAX_HIGH_OCC]; // this is to avoid a heap allocation
|
|
||||||
|
|
||||||
if (n == 0 || n == 1) return;
|
|
||||||
for (i = m = 0; i < n; ++i)
|
|
||||||
if (a[i].n > max_occ) ++m;
|
|
||||||
if (m == 0) return; // no high-frequency k-mers; do nothing
|
|
||||||
for (i = 0, last0 = -1; i <= n; ++i) {
|
|
||||||
if (i == n || a[i].n <= max_occ) {
|
|
||||||
if (i - last0 > 1) {
|
|
||||||
int32_t ps = last0 < 0? 0 : (uint32_t)a[last0].q_pos>>1;
|
|
||||||
int32_t pe = i == n? len : (uint32_t)a[i].q_pos>>1;
|
|
||||||
int32_t j, k, st = last0 + 1, en = i;
|
|
||||||
int32_t max_high_occ = (int32_t)((double)(pe - ps) / dist + .499);
|
|
||||||
if (max_high_occ > 0) {
|
|
||||||
if (max_high_occ > MAX_MAX_HIGH_OCC)
|
|
||||||
max_high_occ = MAX_MAX_HIGH_OCC;
|
|
||||||
for (j = st, k = 0; j < en && k < max_high_occ; ++j, ++k)
|
|
||||||
b[k] = (uint64_t)a[j].n<<32 | j;
|
|
||||||
ks_heapmake_uint64_t(k, b); // initialize the binomial heap
|
|
||||||
for (; j < en; ++j) { // if there are more, choose top max_high_occ
|
|
||||||
if (a[j].n < (int32_t)(b[0]>>32)) { // then update the heap
|
|
||||||
b[0] = (uint64_t)a[j].n<<32 | j;
|
|
||||||
ks_heapdown_uint64_t(0, k, b);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (j = 0; j < k; ++j) a[(uint32_t)b[j]].flt = 1;
|
|
||||||
}
|
|
||||||
for (j = st; j < en; ++j) a[j].flt ^= 1;
|
|
||||||
for (j = st; j < en; ++j)
|
|
||||||
if (a[j].n > max_max_occ)
|
|
||||||
a[j].flt = 1;
|
|
||||||
}
|
|
||||||
last0 = i;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
mm_seed_t *mm_collect_matches(void *km, int *_n_m, int qlen, int max_occ, int max_max_occ, int dist, const mm_idx_t *mi, const mm128_v *mv, int64_t *n_a, int *rep_len, int *n_mini_pos, uint64_t **mini_pos)
|
|
||||||
{
|
|
||||||
int rep_st = 0, rep_en = 0, n_m, n_m0;
|
|
||||||
size_t i;
|
|
||||||
mm_seed_t *m;
|
|
||||||
*n_mini_pos = 0;
|
|
||||||
*mini_pos = (uint64_t*)kmalloc(km, mv->n * sizeof(uint64_t));
|
|
||||||
m = mm_seed_collect_all(km, mi, mv, &n_m0);
|
|
||||||
if (dist > 0 && max_max_occ > max_occ) {
|
|
||||||
mm_seed_select(n_m0, m, qlen, max_occ, max_max_occ, dist);
|
|
||||||
} else {
|
|
||||||
for (i = 0; i < n_m0; ++i)
|
|
||||||
if (m[i].n > max_occ)
|
|
||||||
m[i].flt = 1;
|
|
||||||
}
|
|
||||||
for (i = 0, n_m = 0, *rep_len = 0, *n_a = 0; i < n_m0; ++i) {
|
|
||||||
mm_seed_t *q = &m[i];
|
|
||||||
//fprintf(stderr, "X\t%d\t%d\t%d\n", q->q_pos>>1, q->n, q->flt);
|
|
||||||
if (q->flt) {
|
|
||||||
int en = (q->q_pos >> 1) + 1, st = en - q->q_span;
|
|
||||||
if (st > rep_en) {
|
|
||||||
*rep_len += rep_en - rep_st;
|
|
||||||
rep_st = st, rep_en = en;
|
|
||||||
} else rep_en = en;
|
|
||||||
} else {
|
|
||||||
*n_a += q->n;
|
|
||||||
(*mini_pos)[(*n_mini_pos)++] = (uint64_t)q->q_span<<32 | q->q_pos>>1;
|
|
||||||
m[n_m++] = *q;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
*rep_len += rep_en - rep_st;
|
|
||||||
*_n_m = n_m;
|
|
||||||
return m;
|
|
||||||
}
|
|
||||||
@@ -1,40 +1,29 @@
|
|||||||
try:
|
try:
|
||||||
from setuptools import setup, Extension
|
from setuptools import setup, Extension
|
||||||
from setuptools.command.build_ext import build_ext
|
|
||||||
except ImportError:
|
except ImportError:
|
||||||
from distutils.core import setup
|
from distutils.core import setup
|
||||||
from distutils.extension import Extension
|
from distutils.extension import Extension
|
||||||
from distutils.command.build_ext import build_ext
|
|
||||||
|
|
||||||
import sys, platform, subprocess
|
import sys, platform
|
||||||
|
|
||||||
|
sys.path.append('python')
|
||||||
|
|
||||||
|
extra_compile_args = ['-DHAVE_KALLOC']
|
||||||
|
include_dirs = ["."]
|
||||||
|
|
||||||
|
if platform.machine() in ["aarch64", "arm64"]:
|
||||||
|
include_dirs.append("sse2neon/")
|
||||||
|
extra_compile_args.extend(['-ftree-vectorize', '-DKSW_SSE2_ONLY', '-D__SSE2__'])
|
||||||
|
else:
|
||||||
|
extra_compile_args.append('-msse4.1') # WARNING: ancient x86_64 CPUs don't have SSE4
|
||||||
|
|
||||||
def readme():
|
def readme():
|
||||||
with open('python/README.rst') as f:
|
with open('python/README.rst') as f:
|
||||||
return f.read()
|
return f.read()
|
||||||
|
|
||||||
|
|
||||||
class LibMM2Build(build_ext):
|
|
||||||
# Uses Makefile to build library, avoids duplicating logic
|
|
||||||
# determining which objects to compile but does require
|
|
||||||
# end users to have Make (since precompiled wheels are not
|
|
||||||
# distributed on PyPI).
|
|
||||||
def run(self):
|
|
||||||
def compile_libminimap2(*args, **kwargs):
|
|
||||||
cmd = ['make', 'libminimap2.a'] + list(args)
|
|
||||||
subprocess.check_call(cmd)
|
|
||||||
options = []
|
|
||||||
if platform.machine() in ["aarch64", "arm64"]:
|
|
||||||
options = ["arm_neon=1", "aarch64=1"]
|
|
||||||
self.execute(
|
|
||||||
compile_libminimap2, options,
|
|
||||||
'Compiling libminimap2 using Makefile')
|
|
||||||
build_ext.run(self)
|
|
||||||
|
|
||||||
|
|
||||||
setup(
|
setup(
|
||||||
name = 'mappy',
|
name = 'mappy',
|
||||||
version = '2.25',
|
version = '2.18',
|
||||||
url = 'https://github.com/lh3/minimap2',
|
url = 'https://github.com/lh3/minimap2',
|
||||||
description = 'Minimap2 python binding',
|
description = 'Minimap2 python binding',
|
||||||
long_description = readme(),
|
long_description = readme(),
|
||||||
@@ -43,15 +32,16 @@ setup(
|
|||||||
license = 'MIT',
|
license = 'MIT',
|
||||||
keywords = 'sequence-alignment',
|
keywords = 'sequence-alignment',
|
||||||
scripts = ['python/minimap2.py'],
|
scripts = ['python/minimap2.py'],
|
||||||
cmdclass = {'build_ext': LibMM2Build},
|
ext_modules = [Extension('mappy',
|
||||||
ext_modules = [
|
sources = ['python/mappy.pyx', 'align.c', 'bseq.c', 'chain.c', 'format.c', 'hit.c', 'index.c', 'pe.c', 'options.c',
|
||||||
Extension(
|
'ksw2_extd2_sse.c', 'ksw2_exts2_sse.c', 'ksw2_extz2_sse.c', 'ksw2_ll_sse.c',
|
||||||
'mappy',
|
'kalloc.c', 'kthread.c', 'map.c', 'misc.c', 'sdust.c', 'sketch.c', 'esterr.c', 'splitidx.c'],
|
||||||
sources = ['python/mappy.pyx'],
|
depends = ['minimap.h', 'bseq.h', 'kalloc.h', 'kdq.h', 'khash.h', 'kseq.h', 'ksort.h',
|
||||||
depends = ['python/cmappy.h', 'python/cmappy.pxd'],
|
'ksw2.h', 'kthread.h', 'kvec.h', 'mmpriv.h', 'sdust.h',
|
||||||
include_dirs = ['.'],
|
'python/cmappy.h', 'python/cmappy.pxd'],
|
||||||
extra_objects = ['libminimap2.a'],
|
extra_compile_args = extra_compile_args,
|
||||||
libraries = ['z', 'm', 'pthread'])],
|
include_dirs = include_dirs,
|
||||||
|
libraries = ['z', 'm', 'pthread'])],
|
||||||
classifiers = [
|
classifiers = [
|
||||||
'Development Status :: 5 - Production/Stable',
|
'Development Status :: 5 - Production/Stable',
|
||||||
'License :: OSI Approved :: MIT License',
|
'License :: OSI Approved :: MIT License',
|
||||||
|
|||||||
@@ -338,123 +338,3 @@
|
|||||||
Title = {Introducing difference recurrence relations for faster semi-global alignment of long sequences},
|
Title = {Introducing difference recurrence relations for faster semi-global alignment of long sequences},
|
||||||
Volume = {19},
|
Volume = {19},
|
||||||
Year = {2018}}
|
Year = {2018}}
|
||||||
|
|
||||||
@article{Li:2018ab,
|
|
||||||
Author = {Li, Heng},
|
|
||||||
Journal = {Bioinformatics},
|
|
||||||
Pages = {3094-3100},
|
|
||||||
Title = {Minimap2: pairwise alignment for nucleotide sequences},
|
|
||||||
Volume = {34},
|
|
||||||
Year = {2018}}
|
|
||||||
|
|
||||||
@article{Jain:2020aa,
|
|
||||||
Author = {Jain, Chirag and others},
|
|
||||||
Journal = {Bioinformatics},
|
|
||||||
Pages = {i111-i118},
|
|
||||||
Title = {Weighted minimizer sampling improves long read mapping},
|
|
||||||
Volume = {36},
|
|
||||||
Year = {2020}}
|
|
||||||
|
|
||||||
@article{Miga:2020aa,
|
|
||||||
Author = {Miga, Karen H and others},
|
|
||||||
Journal = {Nature},
|
|
||||||
Pages = {79-84},
|
|
||||||
Title = {Telomere-to-telomere assembly of a complete human {X} chromosome},
|
|
||||||
Volume = {585},
|
|
||||||
Year = {2020}}
|
|
||||||
|
|
||||||
@article {Jain2020.11.01.363887,
|
|
||||||
author = {Jain, Chirag and others},
|
|
||||||
title = {A long read mapping method for highly repetitive reference sequences},
|
|
||||||
elocation-id = {2020.11.01.363887},
|
|
||||||
year = {2020},
|
|
||||||
doi = {10.1101/2020.11.01.363887},
|
|
||||||
publisher = {Cold Spring Harbor Laboratory},
|
|
||||||
URL = {https://www.biorxiv.org/content/early/2020/11/02/2020.11.01.363887},
|
|
||||||
eprint = {https://www.biorxiv.org/content/early/2020/11/02/2020.11.01.363887.full.pdf},
|
|
||||||
journal = {bioRxiv}
|
|
||||||
}
|
|
||||||
|
|
||||||
@article{Li:2020aa,
|
|
||||||
Author = {Li, Heng and others},
|
|
||||||
Journal = {Genome Biol},
|
|
||||||
Pages = {265},
|
|
||||||
Title = {The design and construction of reference pangenome graphs with minigraph},
|
|
||||||
Volume = {21},
|
|
||||||
Year = {2020}}
|
|
||||||
|
|
||||||
@article{Ren:2021aa,
|
|
||||||
Author = {Ren, Jingwen and Chaisson, Mark J P},
|
|
||||||
Journal = {PLoS Comput Biol},
|
|
||||||
Pages = {e1009078},
|
|
||||||
Title = {lra: A long read aligner for sequences and contigs},
|
|
||||||
Volume = {17},
|
|
||||||
Year = {2021}}
|
|
||||||
|
|
||||||
@inproceedings{DBLP:conf/wabi/AbouelhodaO03,
|
|
||||||
Author = {Mohamed Ibrahim Abouelhoda and Enno Ohlebusch},
|
|
||||||
Booktitle = {Algorithms in Bioinformatics, Third International Workshop, {WABI} 2003, Budapest, Hungary, September 15-20, 2003, Proceedings},
|
|
||||||
Crossref = {DBLP:conf/wabi/2003},
|
|
||||||
Pages = {1--16},
|
|
||||||
Title = {A Local Chaining Algorithm and Its Applications in Comparative Genomics},
|
|
||||||
Year = {2003}}
|
|
||||||
|
|
||||||
@article{Ono:2021aa,
|
|
||||||
Author = {Ono, Yukiteru and others},
|
|
||||||
Journal = {Bioinformatics},
|
|
||||||
Pages = {589-595},
|
|
||||||
Title = {{PBSIM2}: a simulator for long-read sequencers with a novel generative model of quality scores},
|
|
||||||
Volume = {37},
|
|
||||||
Year = {2021}}
|
|
||||||
|
|
||||||
@article{Sedlazeck:2018ab,
|
|
||||||
Author = {Sedlazeck, Fritz J and others},
|
|
||||||
Journal = {Nat Methods},
|
|
||||||
Pages = {461-468},
|
|
||||||
Title = {Accurate detection of complex structural variations using single-molecule sequencing},
|
|
||||||
Volume = {15},
|
|
||||||
Year = {2018}}
|
|
||||||
|
|
||||||
@article{Jeffares:2017aa,
|
|
||||||
Author = {Jeffares, Daniel C and others},
|
|
||||||
Journal = {Nat Commun},
|
|
||||||
Pages = {14061},
|
|
||||||
Title = {Transient structural variations have strong effects on quantitative traits and reproductive isolation in fission yeast},
|
|
||||||
Volume = {8},
|
|
||||||
Year = {2017}}
|
|
||||||
|
|
||||||
@article{Zook:2020aa,
|
|
||||||
Author = {Zook, Justin M and others},
|
|
||||||
Journal = {Nat Biotechnol},
|
|
||||||
Pages = {1347-1355},
|
|
||||||
Title = {A robust benchmark for detection of germline large deletions and insertions},
|
|
||||||
Volume = {38},
|
|
||||||
Year = {2020}}
|
|
||||||
|
|
||||||
@article{Harpak:2017aa,
|
|
||||||
Author = {Harpak, Arbel and others},
|
|
||||||
Journal = {Proc Natl Acad Sci U S A},
|
|
||||||
Pages = {12779-12784},
|
|
||||||
Title = {Frequent nonallelic gene conversion on the human lineage and its effect on the divergence of gene duplicates},
|
|
||||||
Volume = {114},
|
|
||||||
Year = {2017}}
|
|
||||||
|
|
||||||
@article{Li:2018aa,
|
|
||||||
Author = {Li, Heng and others},
|
|
||||||
Journal = {Nat Methods},
|
|
||||||
Month = {Aug},
|
|
||||||
Number = {8},
|
|
||||||
Pages = {595-597},
|
|
||||||
Title = {A synthetic-diploid benchmark for accurate variant-calling evaluation},
|
|
||||||
Volume = {15},
|
|
||||||
Year = {2018}}
|
|
||||||
|
|
||||||
@article{Gu:1995wt,
|
|
||||||
author = {Gu, X and Li, W H},
|
|
||||||
journal = {J Mol Evol},
|
|
||||||
month = {Apr},
|
|
||||||
number = {4},
|
|
||||||
pages = {464-73},
|
|
||||||
title = {The size distribution of insertions and deletions in human and rodent pseudogenes suggests the logarithmic gap penalty for sequence alignment},
|
|
||||||
volume = {40},
|
|
||||||
year = {1995}}
|
|
||||||
|
|||||||
@@ -1,240 +0,0 @@
|
|||||||
\documentclass{bioinfo}
|
|
||||||
\copyrightyear{2021}
|
|
||||||
\pubyear{2021}
|
|
||||||
|
|
||||||
\usepackage{graphicx}
|
|
||||||
\usepackage{hyperref}
|
|
||||||
\usepackage{url}
|
|
||||||
\usepackage{amsmath}
|
|
||||||
\usepackage[ruled,vlined]{algorithm2e}
|
|
||||||
\newcommand\mycommfont[1]{\footnotesize\rmfamily{\it #1}}
|
|
||||||
\SetCommentSty{mycommfont}
|
|
||||||
\SetKwComment{Comment}{$\triangleright$\ }{}
|
|
||||||
|
|
||||||
\usepackage{natbib}
|
|
||||||
\bibliographystyle{apalike}
|
|
||||||
|
|
||||||
\DeclareMathOperator*{\argmax}{argmax}
|
|
||||||
|
|
||||||
\begin{document}
|
|
||||||
\firstpage{1}
|
|
||||||
|
|
||||||
\title[Improvements to minimap2]{New strategies to improve minimap2 alignment accuracy}
|
|
||||||
\author[Li]{Heng Li$^{1,2}$}
|
|
||||||
\address{$^1$Dana-Farber Cancer Institute, 450 Brookline Ave, Boston, MA 02215, USA,
|
|
||||||
$^2$Harvard Medical School, 10 Shattuck St, Boston, MA 02215, USA}
|
|
||||||
|
|
||||||
\maketitle
|
|
||||||
|
|
||||||
\begin{abstract}
|
|
||||||
|
|
||||||
\section{Summary:} We present several recent improvements to minimap2, a
|
|
||||||
versatile pairwise aligner for nucleotide sequences. Now minimap2 v2.22 can
|
|
||||||
more accurately map long reads to highly repetitive regions and align through
|
|
||||||
insertions or deletions up to 100kb by default, addressing major weakness in
|
|
||||||
minimap2 v2.18 or earlier.
|
|
||||||
|
|
||||||
\section{Availability and implementation:}
|
|
||||||
\href{https://github.com/lh3/minimap2}{https://github.com/lh3/minimap2}
|
|
||||||
|
|
||||||
\section{Contact:} hli@ds.dfci.harvard.edu
|
|
||||||
\end{abstract}
|
|
||||||
|
|
||||||
\section{Introduction}
|
|
||||||
Minimap2~\citep{Li:2018ab} is widely used for maping long sequence
|
|
||||||
reads and assembly contigs. \citet{Jain:2020aa} found minimap2 v2.18 or earlier occasionally
|
|
||||||
misaligned reads from highly repetitive regions as minimap2 ignored seeds of
|
|
||||||
high occurrence. They also noticed minimap2 may misplace reads with structural
|
|
||||||
variations (SVs) in such regions~\citep{Jain2020.11.01.363887}. These
|
|
||||||
misalignments have become a pressing issue in the advent of
|
|
||||||
temolere-to-telomore human assembly~\citep{Miga:2020aa}. Meanwhile, old minimap2
|
|
||||||
was unable to efficiently align long insertions/deletions (INDELs) and often
|
|
||||||
breaks an alignment around variable-number tandem repeats (VNTRs). This has
|
|
||||||
inspired new chaining algorithms~\citep{Li:2020aa,Ren:2021aa} which are not
|
|
||||||
integrated into minimap2. Here we will describe recent efforts implemented
|
|
||||||
in v2.19 through v2.22 to improve mapping results.
|
|
||||||
|
|
||||||
\begin{methods}
|
|
||||||
\section{Methods}
|
|
||||||
|
|
||||||
\subsection{Rescuing high-occurrence $k$-mers}\label{sec:high-occ}
|
|
||||||
Minimap2 keeps all $k$-mer minimizers~\citep{Roberts:2004fv} during indexing. Its original
|
|
||||||
implementation only selected low-occurrence minimizers during mapping. The
|
|
||||||
cutoff is a few hundred for mapping long reads against a human genome. If a
|
|
||||||
read habors only a few or even no low-occurrence minimizers, it will fail
|
|
||||||
chaining due to insufficient anchors.
|
|
||||||
|
|
||||||
To resolve this issue, we implemented a new heuristic to add additional
|
|
||||||
minimizers. Suppose we are looking at two adjacent low-occurence $k$-mers
|
|
||||||
located at position $x_1$ and $x_2$, respectively. If $|x_1-x_2|\ge L$,
|
|
||||||
minimap2 v2.22 additionally selects $\lfloor|x_1-x_2|/L\rfloor$ minimizers
|
|
||||||
of the lowest occurrence among minimizers between $x_1$ and $x_2$. Here
|
|
||||||
parameter $L$ controls the frequency of sampling. It defaults to 500.
|
|
||||||
This strategy adds necessary anchors at the cost of increasing total alignment
|
|
||||||
time by a few percent on real data.
|
|
||||||
|
|
||||||
\subsection{Aligning through longer INDELs}
|
|
||||||
The original minimap2 may fail to align long INDELs due to its chaining
|
|
||||||
heuristics. Briefly, minimap2 applies dynamic programming (DP) to chain
|
|
||||||
minimizer anchors. This is a quadratic algorithm, slow for chaining
|
|
||||||
contigs. For acceptable performance, the original minimap2 uses a 500bp band by
|
|
||||||
default, which means a gap longer than 500bp will stop chaining.
|
|
||||||
To align through longer gaps, older minimap2 implemented a long-join heurstic as follows.
|
|
||||||
If there is an INDEL longer than 500bp and the two chains around the INDEL
|
|
||||||
have no overlaps on either the query or the reference sequence, minimap2 may
|
|
||||||
join the two short chains later.
|
|
||||||
This heuristic may fail around VNTRs because short chains
|
|
||||||
often have overlaps in VNTRs. More subtly, minimap2 may escape the inner DP
|
|
||||||
loop early, again for performance, if the chaining result is not improved for
|
|
||||||
50 iterations. When there is a copy number change in a long segmental
|
|
||||||
duplication, the early escape may break around the event even if users
|
|
||||||
specify a large band.
|
|
||||||
|
|
||||||
In minigraph~\citep{Li:2020aa}, we developed a new chaining algorithm that
|
|
||||||
finds up to 1kb INDELs with DP-based chaining and goes through longer INDELs with a
|
|
||||||
subquadratic algorithm~\citep{DBLP:conf/wabi/AbouelhodaO03}. We ported the same
|
|
||||||
algorithm to minimap2 for contig mapping. For long-read mapping, the minigraph
|
|
||||||
algorithm is slower. Minimap2 v2.22 still uses the DP-based algorithm to
|
|
||||||
find short chains and then invokes the minigraph algorithm to rechain anchors in
|
|
||||||
these short chains. The rechaining step achieves the same goal as long-join
|
|
||||||
but is more reliable because it can resolve overlaps between short chains. The old
|
|
||||||
long-join heuristic has since been removed.
|
|
||||||
|
|
||||||
\subsection{Properly mapping long reads with SVs}
|
|
||||||
The original minimap2 ranks an alignment by its Smith-Waterman score and
|
|
||||||
outputs the best scoring alignment. However, when there are SVs on the read,
|
|
||||||
the best scoring alignment is sometimes not the correct alignment.
|
|
||||||
\citet{Jain2020.11.01.363887} resolved this dilemma by altering the mapping
|
|
||||||
algorithm.
|
|
||||||
|
|
||||||
In our view, this problem is rooted in inapropriate scoring: affine-gap penalty
|
|
||||||
over-penalizes a long INDEL that was often evolutionarily created in one event.
|
|
||||||
We should not penalize a SV by a function linear in the SV length. Minimap2 v2.22 instead rescores
|
|
||||||
an alignment with the following scoring function. Suppose an alignment consists
|
|
||||||
of $M$ matching bases, $N$ substitutions and $G$ gap opens, we empirically
|
|
||||||
score the alignment with
|
|
||||||
$$
|
|
||||||
S=M-\frac{N+G}{2d}-\sum_{i=1}^G\log_2(1+g_i)
|
|
||||||
$$
|
|
||||||
where $g_i\ge1$ is the length of the $i$-th gap and
|
|
||||||
$$
|
|
||||||
d=\max\left\{\frac{N+G}{M+N+G},0.02\right\}
|
|
||||||
$$
|
|
||||||
It approximates per-base sequence divergence except with the smallest value set
|
|
||||||
to 2\%. As an analogy to affine-gap scoring, the matching score in our scheme
|
|
||||||
is 1, the mismatch and gap open penalties are both $1/2d$ and the gap extension
|
|
||||||
penalty is a logarithm function of the gap length~\citep{Gu:1995wt}. Our scoring gives a long SV
|
|
||||||
a much milder penalty. In terms of time complexity, scoring an alignment is
|
|
||||||
linear in the length of the alignment. The time spent on rescoring is negligible in
|
|
||||||
practice.
|
|
||||||
|
|
||||||
%If we assume sequences evolve under a duplication-mutation model, we may have a
|
|
||||||
%better way to choose the best alignment. If a long read can be mapped to $n$
|
|
||||||
%loci, we can take the read as the template and build a
|
|
||||||
%pseudo-multi-sequence-alignment (pMSA) of $n+1$ sequences. In this pMSA, we say
|
|
||||||
%a site on the read is informative if the $n$ reference subsequences differ at
|
|
||||||
%the position.
|
|
||||||
|
|
||||||
\end{methods}
|
|
||||||
|
|
||||||
\section{Results}
|
|
||||||
|
|
||||||
\begin{table}
|
|
||||||
\processtable{Evaluation of minimap2 v2.22}
|
|
||||||
{\footnotesize\label{tab:1}\begin{tabular}{p{4.2cm}rrrr}
|
|
||||||
\toprule
|
|
||||||
$[$Benchmark$]$ Metric & v2.22 & v2.18 & Winno & lra \\
|
|
||||||
\midrule
|
|
||||||
$[$sim-map$]$ \% mapped reads at Q10 & 97.9 & 97.6 & {\bf 99.0}& 97.3 \\
|
|
||||||
$[$sim-map$]$ err. rate at Q10 (phredQ) & {\bf 52} & {\bf 52} & 38 & 24 \\
|
|
||||||
$[$winno-cmp$]$ rate of diff. (phredQ) & {\bf 41} & 37 & truth & 18 \\
|
|
||||||
$[$winno-cmp$]$ CPU time (hour) & {\bf 5.0} & 5.3 & 71.8 & 13.1 \\
|
|
||||||
$[$winno-cmp$]$ peak RAM (Gb) & 17.1 & 14.4 & {\bf 9.6} & 12.4 \\
|
|
||||||
$[$sim-sv$]$ \% false negative rate & {\bf 0.5} & 2.0 & {\bf 0.5} & 1.4 \\
|
|
||||||
$[$sim-sv$]$ \% false discovery rate & {\bf 0.0} & 0.1 & {\bf 0.0} & 0.1 \\
|
|
||||||
$[$real-sv-1k$]$ \% false negative rate & {\bf 7.3} & 20.0 & 13.0 & N/A \\
|
|
||||||
$[$real-sv-1k$]$ \% false discovery rate & 2.7 & {\bf 2.4} & 2.7 & N/A \\
|
|
||||||
\botrule
|
|
||||||
\end{tabular}}
|
|
||||||
{In $[$sim-map$]$, 152,713 reads were simulated from the CHM13 telomere-to-telomere assembly v1.1
|
|
||||||
(AC: GCA\_009914755.3) with pbsim2~\citep{Ono:2021aa}: ``pbsim2 -{}-hmm\_model R94.model -{}-length-min
|
|
||||||
5000 -{}-length-mean 20000 -{}-accuracy-mean 0.95''. Alignments of mapping quality
|
|
||||||
10 or higher were evaluated by ``paftools.js mapeval''. The mapping error rate
|
|
||||||
is measured in the phred scale: if the error rate is $e$, $-10\log_{10}e$ is
|
|
||||||
reported in the table. In $[$winno-cmp$]$, 1.39 million CHM13 HiFi reads from
|
|
||||||
SRR11292121 were mapped against the same CHM13 assembly. 99.3\% of them were mapped by Winnowmap2
|
|
||||||
at mapping quality 10 or higher and were taken as ground truth to evaluate
|
|
||||||
minimap2 and lra with ``paftools.js pafcmp''. $[$sim-sv$]$ simulated 1,000
|
|
||||||
50bp to 1000bp INDELs from chr8 in CHM13 using SURVIVOR~\citep{Jeffares:2017aa} and simulated Nanopore
|
|
||||||
reads at 30-fold coverage with the same pbsim2 command line. SVs were called with
|
|
||||||
``sniffles -q 10''~\citep{Sedlazeck:2018ab} and compared to the simulated truth with ``SURVIVOR eval
|
|
||||||
call.vcf truth.bed 50''. In $[$real-sv-1k$]$, small and long variants were
|
|
||||||
called by dipcall-0.3~\citep{Li:2018aa} for HG002 assemblies (AC: GCA\_018852605.1 and
|
|
||||||
GCA\_018852615.1) and compared to the GIAB truth~\citep{Zook:2020aa} using ``truvari -r 2000 -s
|
|
||||||
1000 -S 400 -{}-multimatch -{}-passonly'' which sets the minimum INDEL size to 1kb in evaluation. }
|
|
||||||
\end{table}
|
|
||||||
|
|
||||||
We evaluated minimap2 v2.22 along with v2.18, Winnowmap2 v2.03 and lra v1.3.2
|
|
||||||
(Table~\ref{tab:1}), using the default setting of each mapper according to the input data types.
|
|
||||||
Both versions of minimap2 achieved high mapping accuracy on
|
|
||||||
simulated Nanopore reads (sim-map). Winnowmap2 aligned more reads at mapping
|
|
||||||
quality 10 or higher (mapQ10). However, it may occasionally assign a high mapping
|
|
||||||
quality to a read with multiple identical best alignments. This reduced its
|
|
||||||
mapping accuracy.
|
|
||||||
|
|
||||||
In lack of groud truth for real data, we took Winnowmap2 mapping as ground
|
|
||||||
truth to evaluate other mappers (winno-cmp in Table~\ref{tab:1}). Out of 1,378,092 reads with mapQ10
|
|
||||||
alignments by Winnowmap2, minimap2 v2.22 could map all of them. 118 reads, less
|
|
||||||
than 0.01\% of all reads, were mapped differently by v2.22. 51 of them have
|
|
||||||
multiple identical best alignments. We believe these are more likely to be
|
|
||||||
Winnowmap2 errors. Most of the remaining 67 (=118-51) reads have multiple
|
|
||||||
highly similar but not identical alignments.
|
|
||||||
Minimap2 v2.18 is less consistent with 275 differences including 30 unmapped
|
|
||||||
reads mappable by both Winnowmap2 and v2.22.
|
|
||||||
|
|
||||||
For the minimizer rescuing parameter $L$ in Section~\ref{sec:high-occ},
|
|
||||||
we set its default to 500 such that v2.22 has comparable performance to v2.18 given simulated PacBio and Nanopore human reads.
|
|
||||||
To see the effect of this parameter on real data, we tried several different $L$ values.
|
|
||||||
v2.22 gave 99 mapping differences at $L=200$,
|
|
||||||
118 at $L=500$ (default), 167 at $L=750$ and 224 differences at $L=1000$ in comparison to Winnowmap2.
|
|
||||||
$L=200$ is 28\% slower than the default while $L=1000$ is 9\% faster.
|
|
||||||
Changing the default minimizer window size (option ``-w'')
|
|
||||||
and the initial minimizer occurrence cutoff (option ``-f'')
|
|
||||||
also affects performance and accuracy to a similar magnitude.
|
|
||||||
|
|
||||||
The two benchmarks above only evaluate read mappings when there are no variations between the reads and the reference.
|
|
||||||
To measure the mapping accuracy in the presence of SVs (sim-sv), we reproduced
|
|
||||||
the results by~\citep{Jain2020.11.01.363887}. Minimap2 v2.22 is as good as
|
|
||||||
Winnowmap2 now. Note that we were setting the Sniffles mapping quality
|
|
||||||
threshold to 10 in consistent with the benchmarks above. If we used the
|
|
||||||
default threshold 20, v2.22 would miss additional five SVs (accounting for
|
|
||||||
0.5\% of simulated SVs). For four out of these five missing SVs, minimap2 v2.22
|
|
||||||
mapped more variant reads than Winnowmap2. Sniffles did not call these SVs
|
|
||||||
because minimap2 tended to give them conservative mapping quality. It is worth
|
|
||||||
noting that the simulation here only considers a simple scenario in evolution.
|
|
||||||
Non-allelic gene conversions, which happen often in segmental
|
|
||||||
duplications~\citep{Harpak:2017aa}, would obscure the optimal mapping
|
|
||||||
strategies. How much such simple SV simulation informs real-world SV calling
|
|
||||||
remains a question.
|
|
||||||
|
|
||||||
To see if minimap2 v2.22 could improve long INDEL alignment, we ran dipcall on
|
|
||||||
contig-to-reference alignments and focused on INDELs longer than 1kb
|
|
||||||
(real-sv-1k). v2.22 is more sensitive at comparable specificity, confirming its
|
|
||||||
advantage in more contiguous alignment. We could not get dipcall to work well with lra,
|
|
||||||
so did not report the numbers.
|
|
||||||
|
|
||||||
Minimap2 spends most computing time on base alignment. As recent improvements
|
|
||||||
in v2.22 incur little additional computing and do not change the base alignment
|
|
||||||
algorithm, the new version has similar performance to older versions. It is
|
|
||||||
consistently faster than Winnowmap2 by several times. Sometimes simple
|
|
||||||
heuristics can be as effective as more sophisticated yet slower solutions.
|
|
||||||
|
|
||||||
\section*{Acknowledgements}
|
|
||||||
We thank Arang Rhie and Chirag Jain for providing motivating examples for which
|
|
||||||
older minimap2 underperforms.
|
|
||||||
|
|
||||||
\paragraph{Funding\textcolon} This work is funded by NHGRI grant R01HG010040.
|
|
||||||
|
|
||||||
\bibliography{minimap2}
|
|
||||||
|
|
||||||
\end{document}
|
|
||||||
Reference in New Issue
Block a user