mirror of
https://github.com/lh3/minimap2.git
synced 2026-09-24 12:28:12 +08:00
Compare commits
149
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e88e6ea5d5 | ||
|
|
66c90fdb83 | ||
|
|
3b2eca139a | ||
|
|
23d4edfa31 | ||
|
|
249c180b29 | ||
|
|
798ea0a4a3 | ||
|
|
bedd87f61f | ||
|
|
03540c47b3 | ||
|
|
3b1deac0a5 | ||
|
|
de90f2e655 | ||
|
|
ba186a4c78 | ||
|
|
ba2f19ba37 | ||
|
|
c7cdb758db | ||
|
|
7358a1ead1 | ||
|
|
32f552957e | ||
|
|
a05edfa5ec | ||
|
|
8e81145817 | ||
|
|
e37f5ffe39 | ||
|
|
8a1d52bcbe | ||
|
|
f7271a7c24 | ||
|
|
70393eb46e | ||
|
|
9d049f0562 | ||
|
|
5180b70ff3 | ||
|
|
2392e54fe2 | ||
|
|
629c11728e | ||
|
|
59488f0271 | ||
|
|
7e33fde82b | ||
|
|
c4fe52fb07 | ||
|
|
ead1cfbaca | ||
|
|
83a535f148 | ||
|
|
f3af29a8aa | ||
|
|
cf7eaef367 | ||
|
|
2411887d8e | ||
|
|
1a8373bb84 | ||
|
|
161ae7ff73 | ||
|
|
8a6edab847 | ||
|
|
15118dd521 | ||
|
|
2546999639 | ||
|
|
52fafe0fed | ||
|
|
5f449c5cae | ||
|
|
b046052d82 | ||
|
|
5cc3d2239f | ||
|
|
581f2d7123 | ||
|
|
52dbd439bc | ||
|
|
260a68d232 | ||
|
|
177eef259d | ||
|
|
459ce04c84 | ||
|
|
e6cce019e4 | ||
|
|
7025b0b941 | ||
|
|
fe6a0bb337 | ||
|
|
3f7147864b | ||
|
|
c83589b9ea | ||
|
|
ce7a59f412 | ||
|
|
15471bd629 | ||
|
|
ca19463268 | ||
|
|
4f8d1bc360 | ||
|
|
1776c0c645 | ||
|
|
9febf532c1 | ||
|
|
cec23131e4 | ||
|
|
ef09ccf104 | ||
|
|
e74dfd1aa9 | ||
|
|
f31705bb4a | ||
|
|
41d7ccb191 | ||
|
|
34a41197d7 | ||
|
|
9626b3e716 | ||
|
|
379728726a | ||
|
|
4f91558160 | ||
|
|
ec3bc6efd7 | ||
|
|
8ec8866100 | ||
|
|
9e7247cff9 | ||
|
|
cd66777bfb | ||
|
|
5d7d25e92d | ||
|
|
2a3793bbd2 | ||
|
|
f97008a10e | ||
|
|
10502e2a78 | ||
|
|
4422c0c6f9 | ||
|
|
42a11e1d58 | ||
|
|
76df351fa8 | ||
|
|
d065d3bead | ||
|
|
ac146fe7bc | ||
|
|
6c96078ed0 | ||
|
|
bbb4f97e52 | ||
|
|
b7f4d8a0f4 | ||
|
|
e81927e7a1 | ||
|
|
f7dc5799c5 | ||
|
|
817cb81cb0 | ||
|
|
0f5608c4a4 | ||
|
|
e8823a3709 | ||
|
|
7edeec67b0 | ||
|
|
feb92d32ea | ||
|
|
cdbd96be0c | ||
|
|
ba52c79024 | ||
|
|
cd9ccfa069 | ||
|
|
86b716448c | ||
|
|
9ab95be1bb | ||
|
|
b51e859945 | ||
|
|
a9037dc16c | ||
|
|
9729fa99ad | ||
|
|
b6ff332de1 | ||
|
|
77abafaaf3 | ||
|
|
507d39af15 | ||
|
|
827ca4b461 | ||
|
|
d3dde2fdd4 | ||
|
|
7db2e8d21a | ||
|
|
0b41dd26a2 | ||
|
|
2b47846cd6 | ||
|
|
67dd906a80 | ||
|
|
1b0bb7b0ba | ||
|
|
1c4b7e8a48 | ||
|
|
ecbc399fa2 | ||
|
|
4dfd495cc2 | ||
|
|
194b457e79 | ||
|
|
75c8933511 | ||
|
|
1025993469 | ||
|
|
a3253d1a6b | ||
|
|
2da649d1d7 | ||
|
|
f995f55610 | ||
|
|
c9874e2dc5 | ||
|
|
28a37a017a | ||
|
|
ccb0f7b05d | ||
|
|
66db9da7d8 | ||
|
|
cd2b19035b | ||
|
|
9c0e2c67f8 | ||
|
|
da7109fd29 | ||
|
|
2b3403f094 | ||
|
|
3e16e4e39d | ||
|
|
f47e8a525e | ||
|
|
c172df7d2d | ||
|
|
9e6fdd376b | ||
|
|
29f67a1666 | ||
|
|
adde608a42 | ||
|
|
f10dff78dc | ||
|
|
d97bba9f27 | ||
|
|
50775362bb | ||
|
|
0a5e386359 | ||
|
|
cb56fb762a | ||
|
|
e2451e497a | ||
|
|
d2de282d21 | ||
|
|
48cb80ea94 | ||
|
|
6a4b9f9082 | ||
|
|
a7a01fe5bd | ||
|
|
9dceae59a0 | ||
|
|
20a3987082 | ||
|
|
eb3ed6993d | ||
|
|
7996f04008 | ||
|
|
d2e14705e7 | ||
|
|
24f50f38e8 | ||
|
|
04e015d803 | ||
|
|
040f74102c |
@@ -0,0 +1,21 @@
|
||||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
matrix:
|
||||
compiler: [gcc, clang]
|
||||
|
||||
steps:
|
||||
- name: Checkout minimap2
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: Compile with ${{ matrix.compiler }}
|
||||
run: make CC=${{ matrix.compiler }}
|
||||
@@ -0,0 +1,6 @@
|
||||
[submodule "lib/simde"]
|
||||
path = lib/simde
|
||||
url = https://github.com/nemequ/simde.git
|
||||
[submodule "ext/TAL"]
|
||||
path = ext/TAL
|
||||
url = https://github.com/IntelLabs/Trans-Omics-Acceleration-Library.git
|
||||
+5
-1
@@ -6,6 +6,10 @@ matrix:
|
||||
- language: c
|
||||
compiler: clang
|
||||
script: make
|
||||
- arch: arm64
|
||||
language: c
|
||||
compiler: gcc
|
||||
script: make arm_neon=1 aarch64=1
|
||||
- language: python
|
||||
python: "2.7"
|
||||
before_install: pip install cython
|
||||
@@ -15,6 +19,6 @@ matrix:
|
||||
before_install: pip install cython
|
||||
script: python setup.py build_ext
|
||||
- language: python
|
||||
python: "3.6"
|
||||
python: "3.9"
|
||||
before_install: pip install cython
|
||||
script: python setup.py build_ext
|
||||
|
||||
@@ -4,7 +4,6 @@ include ksw2_dispatch.c
|
||||
include main.c
|
||||
include README.md
|
||||
include sse2neon/emmintrin.h
|
||||
include python/mappy.c
|
||||
include python/cmappy.h
|
||||
include python/cmappy.pxd
|
||||
include python/mappy.pyx
|
||||
|
||||
@@ -1,14 +1,58 @@
|
||||
CFLAGS= -g -Wall -O2 -Wc++-compat #-Wextra
|
||||
CPPFLAGS= -DHAVE_KALLOC
|
||||
INCLUDES=
|
||||
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o chain.o align.o hit.o map.o format.o pe.o esterr.o splitidx.o ksw2_ll_sse.o
|
||||
CPPFLAGS= -DHAVE_KALLOC #-march=native #-DALIGN_AVX -DPARALLEL_CHAINING #-DMANUAL_PROFILING
|
||||
COMP_FLAG = -march=native
|
||||
|
||||
ifeq ($(avx2_compile), 1)
|
||||
COMP_FLAG = -mavx2
|
||||
endif
|
||||
|
||||
#CPPFLAGS= -DHAVE_KALLOC -mavx2 -DALIGN_AVX -DAPPLY_AVX2 -DPARALLEL_CHAINING #-DLISA_HASH -DUINT64 -DVECTORIZE #-DMANUAL_PROFILING
|
||||
#CPPFLAGS= -DHAVE_KALLOC -mavx2 -DPARALLEL_CHAINING #-DMANUAL_PROFILING
|
||||
|
||||
OPT_FLAGS= -DPARALLEL_CHAINING -DALIGN_AVX -DAPPLY_AVX2
|
||||
OPT_FLAGS+=$(COMP_FLAG)
|
||||
ifeq ($(lhash_index), 1)
|
||||
CPPFLAGS+= -DLISA_INDEX
|
||||
endif
|
||||
ifeq ($(lhash), 1)
|
||||
OPT_FLAGS+= -DLISA_HASH -DUINT64 -DVECTORIZE
|
||||
endif
|
||||
ifeq ($(manual_profile), 1)
|
||||
CPPFLAGS+= -DMANUAL_PROFILING
|
||||
endif
|
||||
|
||||
#ifeq ($(use_avx2), 1)
|
||||
# OPT_FLAGS+= -DAPPLY_AVX2
|
||||
#endif
|
||||
|
||||
ifeq ($(disable_output), 1)
|
||||
CPPFLAGS+= -DDISABLE_OUTPUT
|
||||
endif
|
||||
|
||||
ifeq ($(no_opt),)
|
||||
CPPFLAGS+= $(OPT_FLAGS)
|
||||
endif
|
||||
|
||||
|
||||
|
||||
#INCLUDES=
|
||||
#INCLUDES= -I./ext/TAL_offline/src/LISA-hash #-I./ext/TAL/src/dynamic-programming
|
||||
INCLUDES= -I./ext/TAL/src/LISA-hash -I./ext/TAL/src/dynamic-programming
|
||||
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o \
|
||||
lchain.o align.o hit.o seed.o map.o format.o pe.o esterr.o splitidx.o \
|
||||
ksw2_ll_sse.o
|
||||
PROG= minimap2
|
||||
PROG_EXTRA= sdust minimap2-lite
|
||||
LIBS= -lm -lz -lpthread
|
||||
|
||||
CC=$(CXX)
|
||||
ifeq ($(CC), g++)
|
||||
CC=g++ -std=c++11
|
||||
endif
|
||||
|
||||
ifeq ($(arm_neon),) # if arm_neon is not defined
|
||||
ifeq ($(sse2only),) # if sse2only is not defined
|
||||
OBJS+=ksw2_extz2_sse41.o ksw2_extd2_sse41.o ksw2_exts2_sse41.o ksw2_extz2_sse2.o ksw2_extd2_sse2.o ksw2_exts2_sse2.o ksw2_dispatch.o
|
||||
OBJS+=ksw2_extz2_sse41.o ksw2_extd2_sse41.o ksw2_exts2_sse41.o ksw2_extz2_sse2.o ksw2_extd2_sse2.o ksw2_exts2_sse2.o ksw2_dispatch.o ksw2_extd2_avx.o
|
||||
else # if sse2only is defined
|
||||
OBJS+=ksw2_extz2_sse.o ksw2_extd2_sse.o ksw2_exts2_sse.o
|
||||
endif
|
||||
@@ -54,6 +98,17 @@ libminimap2.a:$(OBJS)
|
||||
sdust:sdust.c kalloc.o kalloc.h kdq.h kvec.h kseq.h ketopt.h sdust.h
|
||||
$(CC) -D_SDUST_MAIN $(CFLAGS) $< kalloc.o -o $@ -lz
|
||||
|
||||
multi:
|
||||
$(MAKE) clean
|
||||
$(MAKE)
|
||||
mv minimap2 mm2-fast
|
||||
$(MAKE) clean
|
||||
$(MAKE) lhash=1
|
||||
mv minimap2 mm2-fast-lhash
|
||||
$(MAKE) clean
|
||||
$(MAKE) no_opt=1
|
||||
mv minimap2 mm2-fast-no-opt
|
||||
|
||||
# SSE-specific targets on x86/x86_64
|
||||
|
||||
ifeq ($(arm_neon),) # if arm_neon is defined, compile this target with the default setting (i.e. no -msse2)
|
||||
@@ -103,26 +158,28 @@ depend:
|
||||
|
||||
# DO NOT DELETE
|
||||
|
||||
align.o: minimap.h mmpriv.h bseq.h ksw2.h kalloc.h
|
||||
align.o: minimap.h mmpriv.h bseq.h kseq.h ksw2.h kalloc.h
|
||||
bseq.o: bseq.h kvec.h kalloc.h kseq.h
|
||||
chain.o: minimap.h mmpriv.h bseq.h kalloc.h
|
||||
esterr.o: mmpriv.h minimap.h bseq.h
|
||||
esterr.o: mmpriv.h minimap.h bseq.h kseq.h
|
||||
example.o: minimap.h kseq.h
|
||||
format.o: kalloc.h mmpriv.h minimap.h bseq.h
|
||||
hit.o: mmpriv.h minimap.h bseq.h kalloc.h khash.h
|
||||
index.o: kthread.h bseq.h minimap.h mmpriv.h kvec.h kalloc.h khash.h
|
||||
format.o: kalloc.h mmpriv.h minimap.h bseq.h kseq.h
|
||||
hit.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h khash.h
|
||||
index.o: kthread.h bseq.h minimap.h mmpriv.h kseq.h kvec.h kalloc.h khash.h
|
||||
index.o: ksort.h
|
||||
kalloc.o: kalloc.h
|
||||
ksw2_extd2_sse.o: ksw2.h kalloc.h
|
||||
ksw2_exts2_sse.o: ksw2.h kalloc.h
|
||||
ksw2_extz2_sse.o: ksw2.h kalloc.h
|
||||
ksw2_ll_sse.o: ksw2.h kalloc.h
|
||||
kthread.o: kthread.h
|
||||
main.o: bseq.h minimap.h mmpriv.h ketopt.h
|
||||
map.o: kthread.h kvec.h kalloc.h sdust.h mmpriv.h minimap.h bseq.h khash.h
|
||||
map.o: ksort.h
|
||||
misc.o: mmpriv.h minimap.h bseq.h ksort.h
|
||||
options.o: mmpriv.h minimap.h bseq.h
|
||||
pe.o: mmpriv.h minimap.h bseq.h kvec.h kalloc.h ksort.h
|
||||
sdust.o: kalloc.h kdq.h kvec.h ketopt.h sdust.h
|
||||
sketch.o: kvec.h kalloc.h mmpriv.h minimap.h bseq.h
|
||||
splitidx.o: mmpriv.h minimap.h bseq.h
|
||||
lchain.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h krmq.h
|
||||
main.o: bseq.h minimap.h mmpriv.h kseq.h ketopt.h
|
||||
map.o: kthread.h kvec.h kalloc.h sdust.h mmpriv.h minimap.h bseq.h kseq.h
|
||||
map.o: khash.h ksort.h
|
||||
misc.o: mmpriv.h minimap.h bseq.h kseq.h ksort.h
|
||||
options.o: mmpriv.h minimap.h bseq.h kseq.h
|
||||
pe.o: mmpriv.h minimap.h bseq.h kseq.h kvec.h kalloc.h ksort.h
|
||||
sdust.o: kalloc.h kdq.h kvec.h sdust.h
|
||||
seed.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h ksort.h
|
||||
sketch.o: kvec.h kalloc.h mmpriv.h minimap.h bseq.h kseq.h
|
||||
splitidx.o: mmpriv.h minimap.h bseq.h kseq.h
|
||||
|
||||
@@ -0,0 +1,97 @@
|
||||
CFLAGS= -g -Wall -O2 -Wc++-compat #-Wextra
|
||||
CPPFLAGS= -DHAVE_KALLOC -DUSE_SIMDE -DSIMDE_ENABLE_NATIVE_ALIASES
|
||||
INCLUDES= -Ilib/simde
|
||||
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o lchain.o align.o hit.o map.o format.o pe.o seed.o esterr.o splitidx.o \
|
||||
ksw2_extz2_simde.o ksw2_extd2_simde.o ksw2_exts2_simde.o ksw2_ll_simde.o
|
||||
PROG= minimap2
|
||||
PROG_EXTRA= sdust minimap2-lite
|
||||
LIBS= -lm -lz -lpthread
|
||||
|
||||
|
||||
ifneq ($(arm_neon),) # if arm_neon is defined
|
||||
ifeq ($(aarch64),) #if aarch64 is not defined
|
||||
CFLAGS+=-D_FILE_OFFSET_BITS=64 -mfpu=neon -fsigned-char
|
||||
else #if aarch64 is defined
|
||||
CFLAGS+=-D_FILE_OFFSET_BITS=64 -fsigned-char
|
||||
endif
|
||||
endif
|
||||
|
||||
ifneq ($(asan),)
|
||||
CFLAGS+=-fsanitize=address
|
||||
LIBS+=-fsanitize=address
|
||||
endif
|
||||
|
||||
ifneq ($(tsan),)
|
||||
CFLAGS+=-fsanitize=thread
|
||||
LIBS+=-fsanitize=thread
|
||||
endif
|
||||
|
||||
.PHONY:all extra clean depend
|
||||
.SUFFIXES:.c .o
|
||||
|
||||
.c.o:
|
||||
$(CC) -c $(CFLAGS) $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||
|
||||
all:$(PROG)
|
||||
|
||||
extra:all $(PROG_EXTRA)
|
||||
|
||||
minimap2:main.o libminimap2.a
|
||||
$(CC) $(CFLAGS) main.o -o $@ -L. -lminimap2 $(LIBS)
|
||||
|
||||
minimap2-lite:example.o libminimap2.a
|
||||
$(CC) $(CFLAGS) $< -o $@ -L. -lminimap2 $(LIBS)
|
||||
|
||||
libminimap2.a:$(OBJS)
|
||||
$(AR) -csru $@ $(OBJS)
|
||||
|
||||
sdust:sdust.c kalloc.o kalloc.h kdq.h kvec.h kseq.h ketopt.h sdust.h
|
||||
$(CC) -D_SDUST_MAIN $(CFLAGS) $< kalloc.o -o $@ -lz
|
||||
|
||||
ksw2_ll_simde.o:ksw2_ll_sse.c ksw2.h kalloc.h
|
||||
$(CC) -c $(CFLAGS) -msse2 $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||
|
||||
ksw2_extz2_simde.o:ksw2_extz2_sse.c ksw2.h kalloc.h
|
||||
$(CC) -c $(CFLAGS) -msse4.1 $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||
|
||||
ksw2_extd2_simde.o:ksw2_extd2_sse.c ksw2.h kalloc.h
|
||||
$(CC) -c $(CFLAGS) -msse4.1 $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||
|
||||
ksw2_exts2_simde.o:ksw2_exts2_sse.c ksw2.h kalloc.h
|
||||
$(CC) -c $(CFLAGS) -msse4.1 $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||
|
||||
# other non-file targets
|
||||
|
||||
clean:
|
||||
rm -fr gmon.out *.o a.out $(PROG) $(PROG_EXTRA) *~ *.a *.dSYM build dist mappy*.so mappy.c python/mappy.c mappy.egg*
|
||||
|
||||
depend:
|
||||
(LC_ALL=C; export LC_ALL; makedepend -Y -- $(CFLAGS) $(CPPFLAGS) -- *.c)
|
||||
|
||||
# DO NOT DELETE
|
||||
|
||||
align.o: minimap.h mmpriv.h bseq.h kseq.h ksw2.h kalloc.h
|
||||
bseq.o: bseq.h kvec.h kalloc.h kseq.h
|
||||
chain.o: minimap.h mmpriv.h bseq.h kseq.h kalloc.h
|
||||
esterr.o: mmpriv.h minimap.h bseq.h kseq.h
|
||||
example.o: minimap.h kseq.h
|
||||
format.o: kalloc.h mmpriv.h minimap.h bseq.h kseq.h
|
||||
hit.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h khash.h
|
||||
index.o: kthread.h bseq.h minimap.h mmpriv.h kseq.h kvec.h kalloc.h khash.h
|
||||
index.o: ksort.h
|
||||
kalloc.o: kalloc.h
|
||||
ksw2_extd2_sse.o: ksw2.h kalloc.h
|
||||
ksw2_exts2_sse.o: ksw2.h kalloc.h
|
||||
ksw2_extz2_sse.o: ksw2.h kalloc.h
|
||||
ksw2_ll_sse.o: ksw2.h kalloc.h
|
||||
kthread.o: kthread.h
|
||||
main.o: bseq.h minimap.h mmpriv.h kseq.h ketopt.h
|
||||
map.o: kthread.h kvec.h kalloc.h sdust.h mmpriv.h minimap.h bseq.h kseq.h
|
||||
map.o: khash.h ksort.h
|
||||
misc.o: mmpriv.h minimap.h bseq.h kseq.h ksort.h
|
||||
options.o: mmpriv.h minimap.h bseq.h kseq.h
|
||||
pe.o: mmpriv.h minimap.h bseq.h kseq.h kvec.h kalloc.h ksort.h
|
||||
sdust.o: kalloc.h kdq.h kvec.h sdust.h
|
||||
self-chain.o: minimap.h kseq.h
|
||||
sketch.o: kvec.h kalloc.h mmpriv.h minimap.h bseq.h kseq.h
|
||||
splitidx.o: mmpriv.h minimap.h bseq.h kseq.h
|
||||
@@ -1,3 +1,176 @@
|
||||
Release 2.22-r1101 (7 August 2021)
|
||||
----------------------------------
|
||||
|
||||
When choosing the best alignment, this release uses logarithm gap penalty and
|
||||
query-specific mismatch penalty. It improves the sensitivity to long INDELs in
|
||||
repetitive regions.
|
||||
|
||||
Other notable changes:
|
||||
|
||||
* Bugfix: fixed an indirect memory leak that may waste a large amount of
|
||||
memory given highly repetitive reference such as a 16S RNA database (#749).
|
||||
All versions of minimap2 have this issue.
|
||||
|
||||
* New feature: added --cap-kalloc to reduce the peak memory. This option is
|
||||
not enabled by default but may become the default in future releases.
|
||||
|
||||
Known issue:
|
||||
|
||||
* Minimap2 may take a long time to map a read (#771). So far it is not clear
|
||||
if this happens to v2.18 and earlier versions.
|
||||
|
||||
(2.22: 7 August 2021, r1101)
|
||||
|
||||
|
||||
|
||||
Release 2.21-r1071 (6 July 2021)
|
||||
--------------------------------
|
||||
|
||||
This release fixed a regression in short-read mapping introduced in v2.19
|
||||
(#776). It also fixed invalid comparisons of uninitialized variables, though
|
||||
these are harmless (#752). Long-read alignment should be identical to v2.20.
|
||||
|
||||
(2.21: 6 July 2021, r1071)
|
||||
|
||||
|
||||
|
||||
Release 2.20-r1061 (27 May 2021)
|
||||
--------------------------------
|
||||
|
||||
This release fixed a bug in the Python module and improves the command-line
|
||||
compatibiliity with v2.18. In v2.19, if `-r` is specified with an `asm*` preset,
|
||||
users would get alignments more fragmented than v2.18. This could be an issue
|
||||
for existing pipelines specifying `-r`. This release resolves this issue.
|
||||
|
||||
(2.20: 27 May 2021, r1061)
|
||||
|
||||
|
||||
|
||||
Release 2.19-r1057 (26 May 2021)
|
||||
--------------------------------
|
||||
|
||||
This release includes a few important improvements backported from unimap:
|
||||
|
||||
* Improvement: more contiguous alignment through long INDELs. This is enabled
|
||||
by the minigraph chaining algorithm. All `asm*` presets now use the new
|
||||
algorithm. They can find INDELs up to 100kb and may be faster for
|
||||
chromosome-long contigs. The default mode and `map*` presets use this
|
||||
algorithm to replace the long-join heuristic.
|
||||
|
||||
* Improvement: better alignment in highly repetitive regions by rescuing
|
||||
high-occurrence seeds. If the distance between two adjacent seeds is too
|
||||
large, attempt to choose a fraction of high-occurrence seeds in-between.
|
||||
Minimap2 now produces fewer clippings and alignment break points in long
|
||||
satellite regions.
|
||||
|
||||
* Improvement: allow to specify an interval of k-mer occurrences with `-U`.
|
||||
For repeat-rich genomes, the automatic k-mer occurrence threshold determined
|
||||
by `-f` may be too large and makes alignment impractically slow. The new
|
||||
option protects against such cases. Enabled for `asm*` and `map-hifi`.
|
||||
|
||||
* New feature: added the `map-hifi` preset for maping PacBio High-Fidelity
|
||||
(HiFi) reads.
|
||||
|
||||
* Change to the default: apply `--cap-sw-mem=100m` for genomic alignment.
|
||||
|
||||
* Bugfix: minimap2 could not generate an index file with `-xsr` (#734).
|
||||
|
||||
This release represents the most signficant algorithmic change since v2.1 in
|
||||
2017. With features backported from unimap, minimap2 now has similar power to
|
||||
unimap for contig alignment. Unimap will remain an experimental project and is
|
||||
no longer recommended over minimap2. Sorry for reverting the recommendation in
|
||||
short time.
|
||||
|
||||
(2.19: 26 May 2021, r1057)
|
||||
|
||||
|
||||
|
||||
Release 2.18-r1015 (9 April 2021)
|
||||
---------------------------------
|
||||
|
||||
This release fixes multiple rare bugs in minimap2 and adds additional
|
||||
functionality to paftools.js.
|
||||
|
||||
Changes to minimap2:
|
||||
|
||||
* Bugfix: a rare segfault caused by an off-by-one error (#489)
|
||||
|
||||
* Bugfix: minimap2 segfaulted due to an uninitilized variable (#622 and #625).
|
||||
|
||||
* Bugfix: minimap2 parsed spaces as field separators in BED (#721). This led
|
||||
to issues when the BED name column contains spaces.
|
||||
|
||||
* Bugfix: minimap2 `--split-prefix` did not work with long reference names
|
||||
(#394).
|
||||
|
||||
* Bugfix: option `--junc-bonus` didn't work (#513)
|
||||
|
||||
* Bugfix: minimap2 didn't return 1 on I/O errors (#532)
|
||||
|
||||
* Bugfix: the `de:f` tag (sequence divergence) could be negative if there were
|
||||
ambiguous bases
|
||||
|
||||
* Bugfix: fixed two undefined behaviors caused by calling memcpy() on
|
||||
zero-length blocks (#443)
|
||||
|
||||
* Bugfix: there were duplicated SAM @SQ lines if option `--split-prefix` is in
|
||||
use (#400 and #527)
|
||||
|
||||
* Bugfix: option -K had to be smaller than 2 billion (#491). This was caused
|
||||
by a 32-bit integer overflow.
|
||||
|
||||
* Improvement: optionally compile against SIMDe (#597). Minimap2 should work
|
||||
with IBM POWER CPUs, though this has not been tested. To compile with SIMDe,
|
||||
please use `make -f Makefile.simde`.
|
||||
|
||||
* Improvement: more informative error message for I/O errors (#454) and for
|
||||
FASTQ parsing errors (#510)
|
||||
|
||||
* Improvement: abort given malformatted RG line (#541)
|
||||
|
||||
* Improvement: better formula to estimate the `dv:f` tag (approximate sequence
|
||||
divergence). See DOI:10.1101/2021.01.15.426881.
|
||||
|
||||
* New feature: added the `--mask-len` option to fine control the removal of
|
||||
redundant hits (#659). The default behavior is unchanged.
|
||||
|
||||
Changes to mappy:
|
||||
|
||||
* Bugfix: mappy caused segmentation fault if the reference index is not
|
||||
present (#413).
|
||||
|
||||
* Bugfix: fixed a memory leak via 238b6bb3
|
||||
|
||||
* Change: always require Cython to compile the mappy module (#723). Older
|
||||
mappy packages at PyPI bundled the C source code generated by Cython such
|
||||
that end users did not need to install Cython to compile mappy. However, as
|
||||
Python 3.9 is breaking backward compatibility, older mappy does not work
|
||||
with Python 3.9 anymore. We have to add this Cython dependency as a
|
||||
workaround.
|
||||
|
||||
Changes to paftools.js:
|
||||
|
||||
* Bugfix: the "part10-" line from asmgene was wrong (#581)
|
||||
|
||||
* Improvement: compatibility with GTF files from GenBank (#422)
|
||||
|
||||
* New feature: asmgene also checks missing multi-copy genes
|
||||
|
||||
* New feature: added the misjoin command to evaluate large-scale misjoins and
|
||||
megabase-long inversions.
|
||||
|
||||
Although given the many bug fixes and minor improvements, the core algorithm
|
||||
stays the same. This version of minimap2 produces nearly identical alignments
|
||||
to v2.17 except very rare corner cases.
|
||||
|
||||
Now unimap is recommended over minimap2 for aligning long contigs against a
|
||||
reference genome. It often takes less wall-clock time and is much more
|
||||
sensitive to long insertions and deletions.
|
||||
|
||||
(2.18: 9 April 2021, r1015)
|
||||
|
||||
|
||||
|
||||
Release 2.17-r941 (4 May 2019)
|
||||
------------------------------
|
||||
|
||||
|
||||
@@ -1,7 +1,80 @@
|
||||
## mm2-fast
|
||||
### Introduction
|
||||
mm2-fast is an accelerated implementation of minimap2 on modern CPUs. mm2-fast accelerates all the three major modules of minimap2: (a) seeding, (b) chaining, and (c) pairwise alignment, achieving up to 1.8x speedup using AVX512 over minimap2.
|
||||
mm2-fast is a drop-in replacement of minimap2, providing the same functionality with the exact same output.
|
||||
In the current version, all the modules are optimized using **AVX-512** and **AVX2** vectorization. Detailed benchmark results are available in our [preprint](https://doi.org/10.1101/2021.07.21.453294).
|
||||
|
||||
### System requirement
|
||||
Operating System: Linux
|
||||
mm2-fast was tested using g++ (GCC) 9.2.0 and icpc version 19.1.3.304
|
||||
Architecture: x86\_64 CPUs with [AVX512, AVX2](https://en.wikipedia.org/wiki/Advanced_Vector_Extensions)
|
||||
Memory requirement: ~30GB for human genome
|
||||
|
||||
### Installation
|
||||
Clone the *fast-contrib-v2.22* branch from minimap2 github page. The source code can be compiled by using *make* command. It only takes a few seconds.
|
||||
```
|
||||
git clone --recursive https://github.com/lh3/minimap2.git -b fast-contrib-v2.22 mm2-fast
|
||||
cd mm2-fast
|
||||
make
|
||||
```
|
||||
|
||||
### Usage
|
||||
The usage of mm2-fast is same as minimap2. Here is an example of mapping ONT reads with test data.
|
||||
```sh
|
||||
./minimap2 -ax map-ont test/MT-human.fa test/MT-orang.fa > mm2-fast_output
|
||||
```
|
||||
|
||||
### Accuracy evaluation
|
||||
As mm2-fast is an accelerated version of minimap2-v2.22, the output of mm2-fast can be verified against minimap2-v2.22. Note that the optimized chaining in mm2-fast is strictly required to be run with a chaining parameter *max-chain-skip=infinity*. Note that having parameter *max-chain-skip=infinity* leads to higher chaining precision. Therefore, for correctness verification, minimap2 should run with a larger value of *max-chain-skip* parameter. Follow the below steps to verify the accuracy of mm2-fast.
|
||||
```sh
|
||||
git clone --recursive https://github.com/lh3/minimap2.git -b fast-contrib-v2.22 mm2-fast
|
||||
cd mm2-fast && make
|
||||
./minimap2 -ax map-ont test/MT-human.fa test/MT-orang.fa --max-chain-skip=1000000 > mm2-fast_output
|
||||
```
|
||||
```sh
|
||||
git clone https://github.com/lh3/minimap2.git -b v2.22
|
||||
cd minimap2 && make
|
||||
./minimap2 -ax map-ont test/MT-human.fa test/MT-orang.fa --max-chain-skip=1000000 > minimap2_output
|
||||
```
|
||||
The output generated by minimap2 and mm2-fast should match.
|
||||
```sh
|
||||
diff minimap2_output mm2-fast_output > diff_result
|
||||
```
|
||||
The file ```diff_result``` should be empty, meaning a difference of 0 lines.
|
||||
|
||||
### Advanced options
|
||||
The default compilation using make applies two optimizations: vectorized chaining and sequence alignment. The learned-indexes based seeding is disabled by default as it requires availability of [Rust](https://en.wikipedia.org/wiki/Rust_(programming_language)). This is because the learned hash-table uses an external training library that runs on Rust. Rust is trivial to install, see https://rustup.rs/ and add its path to .bashrc file. Rust installation only takes a few seconds. Following are the steps to enable learned hash table optimization in mm2-fast:
|
||||
```sh
|
||||
# Start by building learned hash table index for optimized seeding module
|
||||
./build_rmi.sh test/MT-human.fa map-ont ##Takes two arguments: 1. path-to-reference-seq-file 2. preset.
|
||||
##For human genome, this step should take around 2-3 minutes to finish.
|
||||
|
||||
# Next, compile and run the mapping phase
|
||||
make clean && make lhash=1
|
||||
./minimap2 -ax map-ont test/MT-human.fa test/MT-orang.fa > mm2-fast-lhash_output
|
||||
```
|
||||
To compile mm2-fast with all optimizations turned off and switch back to default minimap2, use the following command during compilation. This could be useful for debugging.
|
||||
```sh
|
||||
make clean && make no_opt=1
|
||||
```
|
||||
|
||||
### Performance
|
||||
We have observed up to 1.8x speedup across datasets (please refer to the paper for more details). For example, for the randomly sampled 100K reads from ["HG002\_GM24385\_1\_2\_3\_Guppy\_3.6.0\_prom.fastq.gz"](https://precision.fda.gov/challenges/10/view), minimap2 takes 92 seconds, while mm2-fast takes 54 seconds to map against the human genome on a 28 cores Intel® Xeon® Platinum 8280 CPUs. Our sampled datasets with 100K reads are available [here](https://drive.google.com/drive/folders/1131j7ejHdT7QZnjxLcTLi5qqwYcfFbuv).
|
||||
|
||||
### Future Plans
|
||||
|
||||
|
||||
### Citations
|
||||
["Accelerating long-read analysis on modern CPUs"](https://doi.org/10.1101/2021.07.21.453294); Saurabh Kalikar, Chirag Jain, Vasimuddin Md, Sanchit Misra; BioRxiv 2021
|
||||
|
||||
---
|
||||
The original README content of minimap2 follows.
|
||||
|
||||
|
||||
[](https://github.com/lh3/minimap2/releases)
|
||||
[](https://anaconda.org/bioconda/minimap2)
|
||||
[](https://pypi.python.org/pypi/mappy)
|
||||
[](https://travis-ci.org/lh3/minimap2)
|
||||
[](https://github.com/lh3/minimap2/actions)
|
||||
## <a name="started"></a>Getting Started
|
||||
```sh
|
||||
git clone https://github.com/lh3/minimap2
|
||||
@@ -12,19 +85,22 @@ cd minimap2 && make
|
||||
./minimap2 -x map-ont -d MT-human-ont.mmi test/MT-human.fa
|
||||
./minimap2 -a MT-human-ont.mmi test/MT-orang.fa > test.sam
|
||||
# use presets (no test data)
|
||||
./minimap2 -ax map-pb ref.fa pacbio.fq.gz > aln.sam # PacBio genomic reads
|
||||
./minimap2 -ax map-pb ref.fa pacbio.fq.gz > aln.sam # PacBio CLR genomic reads
|
||||
./minimap2 -ax map-ont ref.fa ont.fq.gz > aln.sam # Oxford Nanopore genomic reads
|
||||
./minimap2 -ax asm20 ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio CCS genomic reads
|
||||
./minimap2 -ax map-hifi ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.19 or later)
|
||||
./minimap2 -ax asm20 ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.18 or earlier)
|
||||
./minimap2 -ax sr ref.fa read1.fa read2.fa > aln.sam # short genomic paired-end reads
|
||||
./minimap2 -ax splice ref.fa rna-reads.fa > aln.sam # spliced long reads (strand unknown)
|
||||
./minimap2 -ax splice -uf -k14 ref.fa reads.fa > aln.sam # noisy Nanopore Direct RNA-seq
|
||||
./minimap2 -ax splice:hq -uf ref.fa query.fa > aln.sam # Final PacBio Iso-seq or traditional cDNA
|
||||
./minimap2 -ax splice --junc-bed anno.bed12 ref.fa query.fa > aln.sam # prioritize on annotated junctions
|
||||
./minimap2 -cx asm5 asm1.fa asm2.fa > aln.paf # intra-species asm-to-asm alignment
|
||||
./minimap2 -x ava-pb reads.fa reads.fa > overlaps.paf # PacBio read overlap
|
||||
./minimap2 -x ava-ont reads.fa reads.fa > overlaps.paf # Nanopore read overlap
|
||||
# man page for detailed command line options
|
||||
man ./minimap2.1
|
||||
```
|
||||
|
||||
## Table of Contents
|
||||
|
||||
- [Getting Started](#started)
|
||||
@@ -71,8 +147,8 @@ Detailed evaluations are available from the [minimap2 paper][doi] or the
|
||||
Minimap2 is optimized for x86-64 CPUs. You can acquire precompiled binaries from
|
||||
the [release page][release] with:
|
||||
```sh
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.17/minimap2-2.17_x64-linux.tar.bz2 | tar -jxvf -
|
||||
./minimap2-2.17_x64-linux/minimap2
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.22/minimap2-2.22_x64-linux.tar.bz2 | tar -jxvf -
|
||||
./minimap2-2.22_x64-linux/minimap2
|
||||
```
|
||||
If you want to compile from the source, you need to have a C compiler, GNU make
|
||||
and zlib development files installed. Then type `make` in the source code
|
||||
@@ -80,13 +156,20 @@ directory to compile. If you see compilation errors, try `make sse2only=1`
|
||||
to disable SSE4 code, which will make minimap2 slightly slower.
|
||||
|
||||
Minimap2 also works with ARM CPUs supporting the NEON instruction sets. To
|
||||
compile for 32 bit ARM architectures (such as ARMv7), use `make arm_neon=1`. To compile for for 64 bit ARM architectures (such as ARMv8), use `make arm_neon=1 aarch64=1`.
|
||||
compile for 32 bit ARM architectures (such as ARMv7), use `make arm_neon=1`. To
|
||||
compile for for 64 bit ARM architectures (such as ARMv8), use `make arm_neon=1
|
||||
aarch64=1`.
|
||||
|
||||
Minimap2 can use [SIMD Everywhere (SIMDe)][simde] library for porting
|
||||
implementation to the different SIMD instruction sets. To compile using SIMDe,
|
||||
use `make -f Makefile.simde`. To compile for ARM CPUs, use `Makefile.simde`
|
||||
with the ARM related command lines given above.
|
||||
|
||||
### <a name="general"></a>General usage
|
||||
|
||||
Without any options, minimap2 takes a reference database and a query sequence
|
||||
file as input and produce approximate mapping, without base-level alignment
|
||||
(i.e. no CIGAR), in the [PAF format][paf]:
|
||||
(i.e. coordinates are only approximate and no CIGAR in output), in the [PAF format][paf]:
|
||||
```sh
|
||||
minimap2 ref.fa query.fq > approx-mapping.paf
|
||||
```
|
||||
@@ -127,13 +210,13 @@ parameters at the same time. The default setting is the same as `map-ont`.
|
||||
#### <a name="map-long-genomic"></a>Map long noisy genomic reads
|
||||
|
||||
```sh
|
||||
minimap2 -ax map-pb ref.fa pacbio-reads.fq > aln.sam # for PacBio subreads
|
||||
minimap2 -ax map-pb ref.fa pacbio-reads.fq > aln.sam # for PacBio CLR reads
|
||||
minimap2 -ax map-ont ref.fa ont-reads.fq > aln.sam # for Oxford Nanopore reads
|
||||
```
|
||||
The difference between `map-pb` and `map-ont` is that `map-pb` uses
|
||||
homopolymer-compressed (HPC) minimizers as seeds, while `map-ont` uses ordinary
|
||||
minimizers as seeds. Emperical evaluation suggests HPC minimizers improve
|
||||
performance and sensitivity when aligning PacBio reads, but hurt when aligning
|
||||
performance and sensitivity when aligning PacBio CLR reads, but hurt when aligning
|
||||
Nanopore reads.
|
||||
|
||||
#### <a name="map-long-splice"></a>Map long mRNA/cDNA reads
|
||||
@@ -178,10 +261,23 @@ This is because SIRV does not honor the evolutionarily conservative splicing
|
||||
signal. If you are studying SIRV, you may apply `--splice-flank=no` to let
|
||||
minimap2 only model GT..AG, ignoring the additional base.
|
||||
|
||||
Since v2.17, minimap2 can optionally take annotated genes as input and
|
||||
prioritize on annotated splice junctions. To use this feature, you can
|
||||
```sh
|
||||
paftools.js gff2bed anno.gff > anno.bed
|
||||
minimap2 -ax splice --junc-bed anno.bed ref.fa query.fa > aln.sam
|
||||
```
|
||||
Here, `anno.gff` is the gene annotation in the GTF or GFF3 format (`gff2bed`
|
||||
automatically tests the format). The output of `gff2bed` is in the 12-column
|
||||
BED format, or the BED12 format. With the `--junc-bed` option, minimap2 adds a
|
||||
bonus score (tuned by `--junc-bonus`) if an aligned junction matches a junction
|
||||
in the annotation. Option `--junc-bed` also takes 5-column BED, including the
|
||||
strand field. In this case, each line indicates an oriented junction.
|
||||
|
||||
#### <a name="long-overlap"></a>Find overlaps between long reads
|
||||
|
||||
```sh
|
||||
minimap2 -x ava-pb reads.fq reads.fq > ovlp.paf # PacBio read overlap
|
||||
minimap2 -x ava-pb reads.fq reads.fq > ovlp.paf # PacBio CLR read overlap
|
||||
minimap2 -x ava-ont reads.fq reads.fq > ovlp.paf # Oxford Nanopore read overlap
|
||||
```
|
||||
Similarly, `ava-pb` uses HPC minimizers while `ava-ont` uses ordinary
|
||||
@@ -228,7 +324,7 @@ To avoid this issue, you can add option `-L` at the minimap2 command line.
|
||||
This option moves a long CIGAR to the `CG` tag and leaves a fully clipped CIGAR
|
||||
at the SAM CIGAR column. Current tools that don't read CIGAR (e.g. merging and
|
||||
sorting) still work with such BAM records; tools that read CIGAR will
|
||||
effectively ignore these records. It has been decided that future tools will
|
||||
effectively ignore these records. It has been decided that future tools
|
||||
will seamlessly recognize long-cigar records generated by option `-L`.
|
||||
|
||||
**TL;DR**: if you work with ultra-long reads and use tools that only process
|
||||
@@ -249,7 +345,7 @@ CGATCGATAAATAGAGTAG---GAATAGCA
|
||||
CGATCG---AATAGAGTAGGTCGAATtGCA
|
||||
```
|
||||
is represented as `:6-ata:10+gtc:4*at:3`, where `:[0-9]+` represents an
|
||||
identical block, `-ata` represents a deltion, `+gtc` an insertion and `*at`
|
||||
identical block, `-ata` represents a deletion, `+gtc` an insertion and `*at`
|
||||
indicates reference base `a` is substituted with a query base `t`. It is
|
||||
similar to the `MD` SAM tag but is standalone and easier to parse.
|
||||
|
||||
@@ -376,3 +472,5 @@ mappy` or [from BioConda][mappyconda] via `conda install -c bioconda mappy`.
|
||||
[manpage]: https://lh3.github.io/minimap2/minimap2.html
|
||||
[manpage-cs]: https://lh3.github.io/minimap2/minimap2.html#10
|
||||
[doi]: https://doi.org/10.1093/bioinformatics/bty191
|
||||
[smide]: https://github.com/nemequ/simde
|
||||
[unimap]: https://github.com/lh3/unimap
|
||||
|
||||
@@ -5,6 +5,13 @@
|
||||
#include "minimap.h"
|
||||
#include "mmpriv.h"
|
||||
#include "ksw2.h"
|
||||
#include "ksw2_extd2_avx.h"
|
||||
#include <x86intrin.h>
|
||||
extern uint64_t avg;
|
||||
extern uint64_t alignment_time;
|
||||
extern void *km1;
|
||||
extern uint64_t km_size;// = 500000000; // 500 MB
|
||||
extern int km_top;
|
||||
|
||||
static void ksw_gen_simple_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t sc_ambi)
|
||||
{
|
||||
@@ -38,8 +45,8 @@ static inline void update_max_zdrop(int32_t score, int i, int j, int32_t *max, i
|
||||
int z = *max - score - diff * e;
|
||||
if (z > *max_zdrop) {
|
||||
*max_zdrop = z;
|
||||
pos[0][0] = *max_i, pos[0][1] = i + 1;
|
||||
pos[1][0] = *max_j, pos[1][1] = j + 1;
|
||||
pos[0][0] = *max_i, pos[0][1] = i;
|
||||
pos[1][0] = *max_j, pos[1][1] = j;
|
||||
}
|
||||
} else *max = score, *max_i = i, *max_j = j;
|
||||
}
|
||||
@@ -53,16 +60,16 @@ static int mm_test_zdrop(void *km, const mm_mapopt_t *opt, const uint8_t *qseq,
|
||||
// find the score and the region where score drops most along diagonal
|
||||
for (k = 0, score = 0; k < n_cigar; ++k) {
|
||||
uint32_t l, op = cigar[k]&0xf, len = cigar[k]>>4;
|
||||
if (op == 0) {
|
||||
if (op == MM_CIGAR_MATCH) {
|
||||
for (l = 0; l < len; ++l) {
|
||||
score += mat[tseq[i + l] * 5 + qseq[j + l]];
|
||||
update_max_zdrop(score, i+l, j+l, &max, &max_i, &max_j, opt->e, &max_zdrop, pos);
|
||||
}
|
||||
i += len, j += len;
|
||||
} else if (op == 1 || op == 2 || op == 3) {
|
||||
} else if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL || op == MM_CIGAR_N_SKIP) {
|
||||
score -= opt->q + opt->e * len;
|
||||
if (op == 1) j += len; // insertion
|
||||
else i += len; // deletion
|
||||
if (op == MM_CIGAR_INS) j += len;
|
||||
else i += len;
|
||||
update_max_zdrop(score, i, j, &max, &max_i, &max_j, opt->e, &max_zdrop, pos);
|
||||
}
|
||||
}
|
||||
@@ -98,12 +105,12 @@ static void mm_fix_cigar(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq,
|
||||
for (k = 0; k < p->n_cigar; ++k) { // indel left alignment
|
||||
uint32_t op = p->cigar[k]&0xf, len = p->cigar[k]>>4;
|
||||
if (len == 0) to_shrink = 1;
|
||||
if (op == 0) {
|
||||
if (op == MM_CIGAR_MATCH) {
|
||||
toff += len, qoff += len;
|
||||
} else if (op == 1 || op == 2) { // insertion or deletion
|
||||
} else if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL) {
|
||||
if (k > 0 && k < p->n_cigar - 1 && (p->cigar[k-1]&0xf) == 0 && (p->cigar[k+1]&0xf) == 0) {
|
||||
int l, prev_len = p->cigar[k-1] >> 4;
|
||||
if (op == 1) {
|
||||
if (op == MM_CIGAR_INS) {
|
||||
for (l = 0; l < prev_len; ++l)
|
||||
if (qseq[qoff - 1 - l] != qseq[qoff + len - 1 - l])
|
||||
break;
|
||||
@@ -116,9 +123,9 @@ static void mm_fix_cigar(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq,
|
||||
p->cigar[k-1] -= l<<4, p->cigar[k+1] += l<<4, qoff -= l, toff -= l;
|
||||
if (l == prev_len) to_shrink = 1;
|
||||
}
|
||||
if (op == 1) qoff += len;
|
||||
if (op == MM_CIGAR_INS) qoff += len;
|
||||
else toff += len;
|
||||
} else if (op == 3) {
|
||||
} else if (op == MM_CIGAR_N_SKIP) {
|
||||
toff += len;
|
||||
}
|
||||
}
|
||||
@@ -128,13 +135,13 @@ static void mm_fix_cigar(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq,
|
||||
uint32_t l, s[3] = {0,0,0};
|
||||
for (l = k; l < p->n_cigar; ++l) { // count number of adjacent I and D
|
||||
uint32_t op = p->cigar[l]&0xf;
|
||||
if (op == 1 || op == 2 || p->cigar[l]>>4 == 0)
|
||||
if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL || p->cigar[l]>>4 == 0)
|
||||
s[op] += p->cigar[l] >> 4;
|
||||
else break;
|
||||
}
|
||||
if (s[1] > 0 && s[2] > 0 && l - k > 2) { // turn to a single I and a single D
|
||||
p->cigar[k] = s[1]<<4|1;
|
||||
p->cigar[k+1] = s[2]<<4|2;
|
||||
p->cigar[k] = s[1]<<4|MM_CIGAR_INS;
|
||||
p->cigar[k+1] = s[2]<<4|MM_CIGAR_DEL;
|
||||
for (k += 2; k < l; ++k)
|
||||
p->cigar[k] &= 0xf;
|
||||
to_shrink = 1;
|
||||
@@ -154,9 +161,9 @@ static void mm_fix_cigar(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq,
|
||||
else p->cigar[k+1] += p->cigar[k]>>4<<4; // add length to the next CIGAR operator
|
||||
p->n_cigar = l;
|
||||
}
|
||||
if ((p->cigar[0]&0xf) == 1 || (p->cigar[0]&0xf) == 2) { // get rid of leading I or D
|
||||
if ((p->cigar[0]&0xf) == MM_CIGAR_INS || (p->cigar[0]&0xf) == MM_CIGAR_DEL) { // get rid of leading I or D
|
||||
int32_t l = p->cigar[0] >> 4;
|
||||
if ((p->cigar[0]&0xf) == 1) {
|
||||
if ((p->cigar[0]&0xf) == MM_CIGAR_INS) {
|
||||
if (r->rev) r->qe -= l;
|
||||
else r->qs += l;
|
||||
*qshift = l;
|
||||
@@ -174,7 +181,7 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
||||
if (r->p == 0) return;
|
||||
for (k = 0; k < r->p->n_cigar; ++k) {
|
||||
uint32_t op = r->p->cigar[k]&0xf, len = r->p->cigar[k]>>4;
|
||||
if (op == 0) {
|
||||
if (op == MM_CIGAR_MATCH) {
|
||||
while (len > 0) {
|
||||
for (l = 0; l < len && qseq[qoff + l] == tseq[toff + l]; ++l) {} // run of "="; TODO: N<=>N is converted to "="
|
||||
if (l > 0) { ++n_EQX; len -= l; toff += l; qoff += l; }
|
||||
@@ -183,11 +190,11 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
||||
if (l > 0) { ++n_EQX; len -= l; toff += l; qoff += l; }
|
||||
}
|
||||
++n_M;
|
||||
} else if (op == 1) { // insertion
|
||||
} else if (op == MM_CIGAR_INS) {
|
||||
qoff += len;
|
||||
} else if (op == 2) { // deletion
|
||||
} else if (op == MM_CIGAR_DEL) {
|
||||
toff += len;
|
||||
} else if (op == 3) { // intron
|
||||
} else if (op == MM_CIGAR_N_SKIP) {
|
||||
toff += len;
|
||||
}
|
||||
}
|
||||
@@ -195,7 +202,7 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
||||
if (n_EQX == n_M) {
|
||||
for (k = 0; k < r->p->n_cigar; ++k) {
|
||||
uint32_t op = r->p->cigar[k]&0xf, len = r->p->cigar[k]>>4;
|
||||
if (op == 0) r->p->cigar[k] = len << 4 | 7;
|
||||
if (op == MM_CIGAR_MATCH) r->p->cigar[k] = len << 4 | MM_CIGAR_EQ_MATCH;
|
||||
}
|
||||
return;
|
||||
}
|
||||
@@ -209,25 +216,25 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
||||
toff = qoff = m = 0;
|
||||
for (k = 0; k < r->p->n_cigar; ++k) {
|
||||
uint32_t op = r->p->cigar[k]&0xf, len = r->p->cigar[k]>>4;
|
||||
if (op == 0) { // match/mismatch
|
||||
if (op == MM_CIGAR_MATCH) {
|
||||
while (len > 0) {
|
||||
// match
|
||||
for (l = 0; l < len && qseq[qoff + l] == tseq[toff + l]; ++l) {}
|
||||
if (l > 0) p->cigar[m++] = l << 4 | 7;
|
||||
if (l > 0) p->cigar[m++] = l << 4 | MM_CIGAR_EQ_MATCH;
|
||||
len -= l;
|
||||
toff += l, qoff += l;
|
||||
// mismatch
|
||||
for (l = 0; l < len && qseq[qoff + l] != tseq[toff + l]; ++l) {}
|
||||
if (l > 0) p->cigar[m++] = l << 4 | 8;
|
||||
if (l > 0) p->cigar[m++] = l << 4 | MM_CIGAR_X_MISMATCH;
|
||||
len -= l;
|
||||
toff += l, qoff += l;
|
||||
}
|
||||
continue;
|
||||
} else if (op == 1) { // insertion
|
||||
} else if (op == MM_CIGAR_INS) {
|
||||
qoff += len;
|
||||
} else if (op == 2) { // deletion
|
||||
} else if (op == MM_CIGAR_DEL) {
|
||||
toff += len;
|
||||
} else if (op == 3) { // intron
|
||||
} else if (op == MM_CIGAR_N_SKIP) {
|
||||
toff += len;
|
||||
}
|
||||
p->cigar[m++] = r->p->cigar[k];
|
||||
@@ -237,10 +244,11 @@ static void mm_update_cigar_eqx(mm_reg1_t *r, const uint8_t *qseq, const uint8_t
|
||||
r->p = p;
|
||||
}
|
||||
|
||||
static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq, const int8_t *mat, int8_t q, int8_t e, int is_eqx)
|
||||
static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *tseq, const int8_t *mat, int8_t q, int8_t e, int is_eqx, int log_gap)
|
||||
{
|
||||
uint32_t k, l;
|
||||
int32_t s = 0, max = 0, qshift, tshift, toff = 0, qoff = 0;
|
||||
int32_t qshift, tshift, toff = 0, qoff = 0;
|
||||
double s = 0.0, max = 0.0;
|
||||
mm_extra_t *p = r->p;
|
||||
if (p == 0) return;
|
||||
mm_fix_cigar(r, qseq, tseq, &qshift, &tshift);
|
||||
@@ -248,7 +256,7 @@ static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *ts
|
||||
r->blen = r->mlen = 0;
|
||||
for (k = 0; k < p->n_cigar; ++k) {
|
||||
uint32_t op = p->cigar[k]&0xf, len = p->cigar[k]>>4;
|
||||
if (op == 0) { // match/mismatch
|
||||
if (op == MM_CIGAR_MATCH) {
|
||||
int n_ambi = 0, n_diff = 0;
|
||||
for (l = 0; l < len; ++l) {
|
||||
int cq = qseq[qoff + l], ct = tseq[toff + l];
|
||||
@@ -260,27 +268,29 @@ static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *ts
|
||||
}
|
||||
r->blen += len - n_ambi, r->mlen += len - (n_ambi + n_diff), p->n_ambi += n_ambi;
|
||||
toff += len, qoff += len;
|
||||
} else if (op == 1) { // insertion
|
||||
} else if (op == MM_CIGAR_INS) {
|
||||
int n_ambi = 0;
|
||||
for (l = 0; l < len; ++l)
|
||||
if (qseq[qoff + l] > 3) ++n_ambi;
|
||||
r->blen += len - n_ambi, p->n_ambi += n_ambi;
|
||||
s -= q + e * len;
|
||||
if (log_gap) s -= q + (double)e * mg_log2(1.0 + len);
|
||||
else s -= q + e;
|
||||
if (s < 0) s = 0;
|
||||
qoff += len;
|
||||
} else if (op == 2) { // deletion
|
||||
} else if (op == MM_CIGAR_DEL) {
|
||||
int n_ambi = 0;
|
||||
for (l = 0; l < len; ++l)
|
||||
if (tseq[toff + l] > 3) ++n_ambi;
|
||||
r->blen += len - n_ambi, p->n_ambi += n_ambi;
|
||||
s -= q + e * len;
|
||||
if (log_gap) s -= q + (double)e * mg_log2(1.0 + len);
|
||||
else s -= q + e;
|
||||
if (s < 0) s = 0;
|
||||
toff += len;
|
||||
} else if (op == 3) { // intron
|
||||
} else if (op == MM_CIGAR_N_SKIP) {
|
||||
toff += len;
|
||||
}
|
||||
}
|
||||
p->dp_max = max;
|
||||
p->dp_max = (int32_t)(max + .499);
|
||||
assert(qoff == r->qe - r->qs && toff == r->re - r->rs);
|
||||
if (is_eqx) mm_update_cigar_eqx(r, qseq, tseq); // NB: it has to be called here as changes to qseq and tseq are not returned
|
||||
}
|
||||
@@ -310,6 +320,7 @@ static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, uint32_t *cigar) //
|
||||
}
|
||||
}
|
||||
|
||||
#if 0
|
||||
static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc, const int8_t *mat, int w, int end_bonus, int zdrop, int flag, ksw_extz_t *ez)
|
||||
{
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
||||
@@ -333,11 +344,69 @@ static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint
|
||||
int i;
|
||||
fprintf(stderr, "score=%d, cigar=", ez->score);
|
||||
for (i = 0; i < ez->n_cigar; ++i)
|
||||
fprintf(stderr, "%d%c", ez->cigar[i]>>4, "MIDN"[ez->cigar[i]&0xf]);
|
||||
fprintf(stderr, "%d%c", ez->cigar[i]>>4, MM_CIGAR_STR[ez->cigar[i]&0xf]);
|
||||
fprintf(stderr, "\n");
|
||||
}
|
||||
}
|
||||
#endif
|
||||
#if 1
|
||||
static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc, const int8_t *mat, int w, int end_bonus, int zdrop, int flag, ksw_extz_t *ez)
|
||||
{
|
||||
#ifdef MANUAL_PROFILING
|
||||
uint64_t align_start = __rdtsc();
|
||||
#endif
|
||||
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
||||
int i;
|
||||
fprintf(stderr, "===> q=(%d,%d), e=(%d,%d), bw=%d, flag=%d, zdrop=%d <===\n", opt->q, opt->q2, opt->e, opt->e2, w, flag, opt->zdrop);
|
||||
for (i = 0; i < tlen; ++i) fputc("ACGTN"[tseq[i]], stderr);
|
||||
fputc('\n', stderr);
|
||||
for (i = 0; i < qlen; ++i) fputc("ACGTN"[qseq[i]], stderr);
|
||||
fputc('\n', stderr);
|
||||
}
|
||||
if (opt->max_sw_mat > 0 && (int64_t)tlen * qlen > opt->max_sw_mat) {
|
||||
ksw_reset_extz(ez);
|
||||
ez->zdropped = 1;
|
||||
} else if (opt->flag & MM_F_SPLICE)
|
||||
ksw_exts2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->noncan, zdrop, opt->junc_bonus, flag, junc, ez);
|
||||
else if (opt->q == opt->q2 && opt->e == opt->e2)
|
||||
ksw_extz2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, w, zdrop, end_bonus, flag, ez);
|
||||
else{
|
||||
#if defined (ALIGN_AVX) && (defined(__AVX512BW__) || (defined(__AVX2__) && defined(APPLY_AVX2)))
|
||||
#ifdef __AVX512BW__
|
||||
|
||||
ksw_extd2_avx512(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, flag, ez);
|
||||
#elif __AVX2__
|
||||
avg = 0;
|
||||
// uint64_t *ptr_km = (uint64_t *) km1;
|
||||
// for(uint64_t itr = 0; itr < km_size/512; itr++){
|
||||
// avg+=ptr_km[itr];
|
||||
// }
|
||||
|
||||
//#ifdef MANUAL_PROFILING
|
||||
// uint64_t align_start = __rdtsc();
|
||||
//#endif
|
||||
ksw_extd2_avx2(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, flag, ez);
|
||||
//#ifdef MANUAL_PROFILING
|
||||
// alignment_time += (__rdtsc() - align_start);
|
||||
//#endif
|
||||
#endif
|
||||
#else
|
||||
ksw_extd2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, flag, ez);
|
||||
#endif
|
||||
}
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
||||
int i;
|
||||
fprintf(stderr, "score=%d, cigar=", ez->score);
|
||||
for (i = 0; i < ez->n_cigar; ++i)
|
||||
fprintf(stderr, "%d%c", ez->cigar[i]>>4, "MIDN"[ez->cigar[i]&0xf]);
|
||||
fprintf(stderr, "\n");
|
||||
}
|
||||
#ifdef MANUAL_PROFILING
|
||||
alignment_time += (__rdtsc() - align_start);
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
static inline int mm_get_hplen_back(const mm_idx_t *mi, uint32_t rid, uint32_t x)
|
||||
{
|
||||
int64_t i, off0 = mi->seq[rid].offset, off = off0 + x;
|
||||
@@ -533,8 +602,13 @@ static int mm_seed_ext_score(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
||||
re = re + ext_len < (int32_t)mi->seq[rid].len? re + ext_len : mi->seq[rid].len;
|
||||
qe = qe + ext_len < qlen? qe + ext_len : qlen;
|
||||
tseq = (uint8_t*)kmalloc(km, re - rs);
|
||||
mm_idx_getseq(mi, rid, rs, re, tseq);
|
||||
qseq = qseq0[a->x>>63] + qs;
|
||||
if (opt->flag & MM_F_QSTRAND) {
|
||||
qseq = qseq0[0] + qs;
|
||||
mm_idx_getseq2(mi, a->x>>63, rid, rs, re, tseq);
|
||||
} else {
|
||||
qseq = qseq0[a->x>>63] + qs;
|
||||
mm_idx_getseq(mi, rid, rs, re, tseq);
|
||||
}
|
||||
qp = ksw_ll_qinit(km, 2, qe - qs, qseq, 5, mat);
|
||||
score = ksw_ll_i16(qp, re - rs, tseq, opt->q, opt->e, &q_off, &t_off);
|
||||
kfree(km, tseq);
|
||||
@@ -567,7 +641,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
int is_sr = !!(opt->flag & MM_F_SR), is_splice = !!(opt->flag & MM_F_SPLICE);
|
||||
int32_t rid = a[r->as].x<<1>>33, rev = a[r->as].x>>63, as1, cnt1;
|
||||
uint8_t *tseq, *qseq, *junc;
|
||||
int32_t i, l, bw, dropped = 0, extra_flag = 0, rs0, re0, qs0, qe0;
|
||||
int32_t i, l, bw, bw_long, dropped = 0, extra_flag = 0, rs0, re0, qs0, qe0;
|
||||
int32_t rs, re, qs, qe;
|
||||
int32_t rs1, qs1, re1, qe1;
|
||||
int8_t mat[25];
|
||||
@@ -578,6 +652,8 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
if (r->cnt == 0) return;
|
||||
ksw_gen_simple_mat(5, mat, opt->a, opt->b, opt->sc_ambi);
|
||||
bw = (int)(opt->bw * 1.5 + 1.);
|
||||
bw_long = (int)(opt->bw_long * 1.5 + 1.);
|
||||
if (bw_long < bw) bw_long = bw;
|
||||
|
||||
if (is_sr && !(mi->flag & MM_I_HPC)) {
|
||||
mm_max_stretch(r, a, &as1, &cnt1);
|
||||
@@ -688,8 +764,13 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
junc = (uint8_t*)kmalloc(km, re0 - rs0);
|
||||
|
||||
if (qs > 0 && rs > 0) { // left extension; probably the condition can be changed to "qs > qs0 && rs > rs0"
|
||||
qseq = &qseq0[rev][qs0];
|
||||
mm_idx_getseq(mi, rid, rs0, rs, tseq);
|
||||
if (opt->flag & MM_F_QSTRAND) {
|
||||
qseq = &qseq0[0][qs0];
|
||||
mm_idx_getseq2(mi, rev, rid, rs0, rs, tseq);
|
||||
} else {
|
||||
qseq = &qseq0[rev][qs0];
|
||||
mm_idx_getseq(mi, rid, rs0, rs, tseq);
|
||||
}
|
||||
mm_idx_bed_junc(mi, rid, rs0, rs, junc);
|
||||
mm_seq_rev(qs - qs0, qseq);
|
||||
mm_seq_rev(rs - rs0, tseq);
|
||||
@@ -714,12 +795,17 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
} else mm_adjust_minier(mi, qseq0, &a[as1 + i], &re, &qe);
|
||||
re1 = re, qe1 = qe;
|
||||
if (i == cnt1 - 1 || (a[as1+i].y&MM_SEED_LONG_JOIN) || (qe - qs >= opt->min_ksw_len && re - rs >= opt->min_ksw_len)) {
|
||||
int j, bw1 = bw, zdrop_code;
|
||||
int j, bw1 = bw_long, zdrop_code;
|
||||
if (a[as1+i].y & MM_SEED_LONG_JOIN)
|
||||
bw1 = qe - qs > re - rs? qe - qs : re - rs;
|
||||
// perform alignment
|
||||
qseq = &qseq0[rev][qs];
|
||||
mm_idx_getseq(mi, rid, rs, re, tseq);
|
||||
if (opt->flag & MM_F_QSTRAND) {
|
||||
qseq = &qseq0[0][qs];
|
||||
mm_idx_getseq2(mi, rev, rid, rs, re, tseq);
|
||||
} else {
|
||||
qseq = &qseq0[rev][qs];
|
||||
mm_idx_getseq(mi, rid, rs, re, tseq);
|
||||
}
|
||||
mm_idx_bed_junc(mi, rid, rs, re, junc);
|
||||
if (is_sr) { // perform ungapped alignment
|
||||
assert(qe - qs == re - rs);
|
||||
@@ -728,7 +814,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
if (qseq[j] >= 4 || tseq[j] >= 4) ez->score += opt->e2;
|
||||
else ez->score += qseq[j] == tseq[j]? opt->a : -opt->b;
|
||||
}
|
||||
ez->cigar = ksw_push_cigar(km, &ez->n_cigar, &ez->m_cigar, ez->cigar, 0, qe - qs);
|
||||
ez->cigar = ksw_push_cigar(km, &ez->n_cigar, &ez->m_cigar, ez->cigar, MM_CIGAR_MATCH, qe - qs);
|
||||
} else { // perform normal gapped alignment
|
||||
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, opt->zdrop, extra_flag|KSW_EZ_APPROX_MAX, ez); // first pass: with approximate Z-drop
|
||||
}
|
||||
@@ -739,6 +825,13 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
if (ez->n_cigar > 0)
|
||||
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
||||
if (ez->zdropped) { // truncated by Z-drop; TODO: sometimes Z-drop kicks in because the next seed placement is wrong. This can be fixed in principle.
|
||||
if (!r->p) {
|
||||
assert(ez->n_cigar == 0);
|
||||
uint32_t capacity = sizeof(mm_extra_t)/4;
|
||||
kroundup32(capacity);
|
||||
r->p = (mm_extra_t*)calloc(capacity, 4);
|
||||
r->p->capacity = capacity;
|
||||
}
|
||||
for (j = i - 1; j >= 0; --j)
|
||||
if ((int32_t)a[as1 + j].x <= rs + ez->max_t)
|
||||
break;
|
||||
@@ -748,7 +841,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
re1 = rs + (ez->max_t + 1);
|
||||
qe1 = qs + (ez->max_q + 1);
|
||||
if (cnt1 - (j + 1) >= opt->min_cnt) {
|
||||
mm_split_reg(r, r2, as1 + j + 1 - r->as, qlen, a);
|
||||
mm_split_reg(r, r2, as1 + j + 1 - r->as, qlen, a, !!(opt->flag&MM_F_QSTRAND));
|
||||
if (zdrop_code == 2) r2->split_inv = 1;
|
||||
}
|
||||
break;
|
||||
@@ -758,8 +851,13 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
}
|
||||
|
||||
if (!dropped && qe < qe0 && re < re0) { // right extension
|
||||
qseq = &qseq0[rev][qe];
|
||||
mm_idx_getseq(mi, rid, re, re0, tseq);
|
||||
if (opt->flag & MM_F_QSTRAND) {
|
||||
qseq = &qseq0[0][qe];
|
||||
mm_idx_getseq2(mi, rev, rid, re, re0, tseq);
|
||||
} else {
|
||||
qseq = &qseq0[rev][qe];
|
||||
mm_idx_getseq(mi, rid, re, re0, tseq);
|
||||
}
|
||||
mm_idx_bed_junc(mi, rid, re, re0, junc);
|
||||
mm_align_pair(km, opt, qe0 - qe, qseq, re0 - re, tseq, junc, mat, bw, opt->end_bonus, opt->zdrop, extra_flag|KSW_EZ_EXTZ_ONLY, ez);
|
||||
if (ez->n_cigar > 0) {
|
||||
@@ -772,13 +870,19 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
assert(qe1 <= qlen);
|
||||
|
||||
r->rs = rs1, r->re = re1;
|
||||
if (rev) r->qs = qlen - qe1, r->qe = qlen - qs1;
|
||||
else r->qs = qs1, r->qe = qe1;
|
||||
if (!rev || (opt->flag & MM_F_QSTRAND)) r->qs = qs1, r->qe = qe1;
|
||||
else r->qs = qlen - qe1, r->qe = qlen - qs1;
|
||||
|
||||
assert(re1 - rs1 <= re0 - rs0);
|
||||
if (r->p) {
|
||||
mm_idx_getseq(mi, rid, rs1, re1, tseq);
|
||||
mm_update_extra(r, &qseq0[r->rev][qs1], tseq, mat, opt->q, opt->e, opt->flag & MM_F_EQX);
|
||||
if (opt->flag & MM_F_QSTRAND) {
|
||||
mm_idx_getseq2(mi, r->rev, rid, rs1, re1, tseq);
|
||||
qseq = &qseq0[0][qs1];
|
||||
} else {
|
||||
mm_idx_getseq(mi, rid, rs1, re1, tseq);
|
||||
qseq = &qseq0[r->rev][qs1];
|
||||
}
|
||||
mm_update_extra(r, qseq, tseq, mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & MM_F_SR));
|
||||
if (rev && r->p->trans_strand)
|
||||
r->p->trans_strand ^= 3; // flip to the read strand
|
||||
}
|
||||
@@ -788,7 +892,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
}
|
||||
|
||||
static int mm_align1_inv(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, uint8_t *qseq0[2], const mm_reg1_t *r1, const mm_reg1_t *r2, mm_reg1_t *r_inv, ksw_extz_t *ez)
|
||||
{
|
||||
{ // NB: this doesn't work with the qstrand mode
|
||||
int tl, ql, score, ret = 0, q_off, t_off;
|
||||
uint8_t *tseq, *qseq;
|
||||
int8_t mat[25];
|
||||
@@ -837,7 +941,7 @@ static int mm_align1_inv(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, i
|
||||
}
|
||||
r_inv->rs = r1->re + t_off;
|
||||
r_inv->re = r_inv->rs + ez->max_t + 1;
|
||||
mm_update_extra(r_inv, &qseq[q_off], &tseq[t_off], mat, opt->q, opt->e, opt->flag & MM_F_EQX);
|
||||
mm_update_extra(r_inv, &qseq[q_off], &tseq[t_off], mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & MM_F_SR));
|
||||
ret = 1;
|
||||
end_align1_inv:
|
||||
kfree(km, tseq);
|
||||
@@ -854,6 +958,71 @@ static inline mm_reg1_t *mm_insert_reg(const mm_reg1_t *r, int i, int *n_regs, m
|
||||
return regs;
|
||||
}
|
||||
|
||||
static inline void mm_count_gaps(const mm_reg1_t *r, int32_t *n_gap_, int32_t *n_gapo_)
|
||||
{
|
||||
uint32_t i;
|
||||
int32_t n_gapo = 0, n_gap = 0;
|
||||
*n_gap_ = *n_gapo_ = -1;
|
||||
if (r->p == 0) return;
|
||||
for (i = 0; i < r->p->n_cigar; ++i) {
|
||||
int32_t op = r->p->cigar[i] & 0xf, len = r->p->cigar[i] >> 4;
|
||||
if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL)
|
||||
++n_gapo, n_gap += len;
|
||||
}
|
||||
*n_gap_ = n_gap, *n_gapo_ = n_gapo;
|
||||
}
|
||||
|
||||
double mm_event_identity(const mm_reg1_t *r)
|
||||
{
|
||||
int32_t n_gap, n_gapo;
|
||||
if (r->p == 0) return -1.0f;
|
||||
mm_count_gaps(r, &n_gap, &n_gapo);
|
||||
return (double)r->mlen / (r->blen + r->p->n_ambi - n_gap + n_gapo);
|
||||
}
|
||||
|
||||
static int32_t mm_recal_max_dp(const mm_reg1_t *r, double b2, int32_t match_sc)
|
||||
{
|
||||
uint32_t i;
|
||||
int32_t n_gap = 0, n_gapo = 0, n_mis;
|
||||
double gap_cost = 0.0;
|
||||
if (r->p == 0) return -1;
|
||||
for (i = 0; i < r->p->n_cigar; ++i) {
|
||||
int32_t op = r->p->cigar[i] & 0xf, len = r->p->cigar[i] >> 4;
|
||||
if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL) {
|
||||
gap_cost += b2 + (double)mg_log2(1.0 + len);
|
||||
++n_gapo, n_gap += len;
|
||||
}
|
||||
}
|
||||
n_mis = r->blen + r->p->n_ambi - r->mlen - n_gap;
|
||||
return (int32_t)(match_sc * (r->mlen - b2 * n_mis - gap_cost) + .499);
|
||||
}
|
||||
|
||||
void mm_update_dp_max(int qlen, int n_regs, mm_reg1_t *regs, float frac, int a, int b)
|
||||
{
|
||||
int32_t max = -1, max2 = -1, i, max_i = -1;
|
||||
double div, b2;
|
||||
if (n_regs < 2) return;
|
||||
for (i = 0; i < n_regs; ++i) {
|
||||
mm_reg1_t *r = ®s[i];
|
||||
if (r->p == 0) continue;
|
||||
if (r->p->dp_max > max) max2 = max, max = r->p->dp_max, max_i = i;
|
||||
else if (r->p->dp_max > max2) max2 = r->p->dp_max;
|
||||
}
|
||||
if (max_i < 0 || max < 0 || max2 < 0) return;
|
||||
if (regs[max_i].qe - regs[max_i].qs < (double)qlen * frac) return;
|
||||
if (max2 < (double)max * frac) return;
|
||||
div = 1. - mm_event_identity(®s[max_i]);
|
||||
if (div < 0.02) div = 0.02;
|
||||
b2 = 0.5 / div; // max value: 25
|
||||
if (b2 * a < b) b2 = (double)a / b;
|
||||
for (i = 0; i < n_regs; ++i) {
|
||||
mm_reg1_t *r = ®s[i];
|
||||
if (r->p == 0) continue;
|
||||
r->p->dp_max = mm_recal_max_dp(r, b2, a);
|
||||
if (r->p->dp_max < 0) r->p->dp_max = 0;
|
||||
}
|
||||
}
|
||||
|
||||
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a)
|
||||
{
|
||||
extern unsigned char seq_nt4_table[256];
|
||||
@@ -897,7 +1066,7 @@ mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
||||
regs[i].p->trans_strand = opt->flag&MM_F_SPLICE_FOR? 1 : 2;
|
||||
}
|
||||
if (r2.cnt > 0) regs = mm_insert_reg(&r2, i, &n_regs, regs);
|
||||
if (i > 0 && regs[i].split_inv) {
|
||||
if (i > 0 && regs[i].split_inv && !(opt->flag & MM_F_NO_INV)) {
|
||||
if (mm_align1_inv(km, opt, mi, qlen, qseq0, ®s[i-1], ®s[i], &r2, &ez)) {
|
||||
regs = mm_insert_reg(&r2, i, &n_regs, regs);
|
||||
++i; // skip the inserted INV alignment
|
||||
@@ -908,6 +1077,10 @@ mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
||||
kfree(km, qseq0[0]);
|
||||
kfree(km, ez.cigar);
|
||||
mm_filter_regs(opt, qlen, n_regs_, regs);
|
||||
mm_hit_sort(km, n_regs_, regs);
|
||||
if (!(opt->flag&MM_F_SR) && !opt->split_prefix && qlen >= opt->rank_min_len) {
|
||||
mm_update_dp_max(qlen, *n_regs_, regs, opt->rank_frac, opt->a, opt->b);
|
||||
mm_filter_regs(opt, qlen, n_regs_, regs);
|
||||
}
|
||||
mm_hit_sort(km, n_regs_, regs, opt->alt_drop);
|
||||
return regs;
|
||||
}
|
||||
|
||||
@@ -77,7 +77,7 @@ static inline void kseq2bseq(kseq_t *ks, mm_bseq1_t *s, int with_qual, int with_
|
||||
s->l_seq = ks->seq.l;
|
||||
}
|
||||
|
||||
mm_bseq1_t *mm_bseq_read3(mm_bseq_file_t *fp, int chunk_size, int with_qual, int with_comment, int frag_mode, int *n_)
|
||||
mm_bseq1_t *mm_bseq_read3(mm_bseq_file_t *fp, int64_t chunk_size, int with_qual, int with_comment, int frag_mode, int *n_)
|
||||
{
|
||||
int64_t size = 0;
|
||||
int ret;
|
||||
@@ -99,7 +99,7 @@ mm_bseq1_t *mm_bseq_read3(mm_bseq_file_t *fp, int chunk_size, int with_qual, int
|
||||
size += s->l_seq;
|
||||
if (size >= chunk_size) {
|
||||
if (frag_mode && a.a[a.n-1].l_seq < CHECK_PAIR_THRES) {
|
||||
while (kseq_read(ks) >= 0) {
|
||||
while ((ret = kseq_read(ks)) >= 0) {
|
||||
kseq2bseq(ks, &fp->s, with_qual, with_comment);
|
||||
if (mm_qname_same(fp->s.name, a.a[a.n-1].name)) {
|
||||
kv_push(mm_bseq1_t, 0, a, fp->s);
|
||||
@@ -110,23 +110,25 @@ mm_bseq1_t *mm_bseq_read3(mm_bseq_file_t *fp, int chunk_size, int with_qual, int
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (ret < -1)
|
||||
fprintf(stderr, "[WARNING]\033[1;31m wrong FASTA/FASTQ record. Continue anyway.\033[0m\n");
|
||||
if (ret < -1) {
|
||||
if (a.n) fprintf(stderr, "[WARNING]\033[1;31m failed to parse the FASTA/FASTQ record next to '%s'. Continue anyway.\033[0m\n", a.a[a.n-1].name);
|
||||
else fprintf(stderr, "[WARNING]\033[1;31m failed to parse the first FASTA/FASTQ record. Continue anyway.\033[0m\n");
|
||||
}
|
||||
*n_ = a.n;
|
||||
return a.a;
|
||||
}
|
||||
|
||||
mm_bseq1_t *mm_bseq_read2(mm_bseq_file_t *fp, int chunk_size, int with_qual, int frag_mode, int *n_)
|
||||
mm_bseq1_t *mm_bseq_read2(mm_bseq_file_t *fp, int64_t chunk_size, int with_qual, int frag_mode, int *n_)
|
||||
{
|
||||
return mm_bseq_read3(fp, chunk_size, with_qual, 0, frag_mode, n_);
|
||||
}
|
||||
|
||||
mm_bseq1_t *mm_bseq_read(mm_bseq_file_t *fp, int chunk_size, int with_qual, int *n_)
|
||||
mm_bseq1_t *mm_bseq_read(mm_bseq_file_t *fp, int64_t chunk_size, int with_qual, int *n_)
|
||||
{
|
||||
return mm_bseq_read2(fp, chunk_size, with_qual, 0, n_);
|
||||
}
|
||||
|
||||
mm_bseq1_t *mm_bseq_read_frag2(int n_fp, mm_bseq_file_t **fp, int chunk_size, int with_qual, int with_comment, int *n_)
|
||||
mm_bseq1_t *mm_bseq_read_frag2(int n_fp, mm_bseq_file_t **fp, int64_t chunk_size, int with_qual, int with_comment, int *n_)
|
||||
{
|
||||
int i;
|
||||
int64_t size = 0;
|
||||
@@ -156,7 +158,7 @@ mm_bseq1_t *mm_bseq_read_frag2(int n_fp, mm_bseq_file_t **fp, int chunk_size, in
|
||||
return a.a;
|
||||
}
|
||||
|
||||
mm_bseq1_t *mm_bseq_read_frag(int n_fp, mm_bseq_file_t **fp, int chunk_size, int with_qual, int *n_)
|
||||
mm_bseq1_t *mm_bseq_read_frag(int n_fp, mm_bseq_file_t **fp, int64_t chunk_size, int with_qual, int *n_)
|
||||
{
|
||||
return mm_bseq_read_frag2(n_fp, fp, chunk_size, with_qual, 0, n_);
|
||||
}
|
||||
|
||||
@@ -18,11 +18,11 @@ typedef struct {
|
||||
|
||||
mm_bseq_file_t *mm_bseq_open(const char *fn);
|
||||
void mm_bseq_close(mm_bseq_file_t *fp);
|
||||
mm_bseq1_t *mm_bseq_read3(mm_bseq_file_t *fp, int chunk_size, int with_qual, int with_comment, int frag_mode, int *n_);
|
||||
mm_bseq1_t *mm_bseq_read2(mm_bseq_file_t *fp, int chunk_size, int with_qual, int frag_mode, int *n_);
|
||||
mm_bseq1_t *mm_bseq_read(mm_bseq_file_t *fp, int chunk_size, int with_qual, int *n_);
|
||||
mm_bseq1_t *mm_bseq_read_frag2(int n_fp, mm_bseq_file_t **fp, int chunk_size, int with_qual, int with_comment, int *n_);
|
||||
mm_bseq1_t *mm_bseq_read_frag(int n_fp, mm_bseq_file_t **fp, int chunk_size, int with_qual, int *n_);
|
||||
mm_bseq1_t *mm_bseq_read3(mm_bseq_file_t *fp, int64_t chunk_size, int with_qual, int with_comment, int frag_mode, int *n_);
|
||||
mm_bseq1_t *mm_bseq_read2(mm_bseq_file_t *fp, int64_t chunk_size, int with_qual, int frag_mode, int *n_);
|
||||
mm_bseq1_t *mm_bseq_read(mm_bseq_file_t *fp, int64_t chunk_size, int with_qual, int *n_);
|
||||
mm_bseq1_t *mm_bseq_read_frag2(int n_fp, mm_bseq_file_t **fp, int64_t chunk_size, int with_qual, int with_comment, int *n_);
|
||||
mm_bseq1_t *mm_bseq_read_frag(int n_fp, mm_bseq_file_t **fp, int64_t chunk_size, int with_qual, int *n_);
|
||||
int mm_bseq_eof(mm_bseq_file_t *fp);
|
||||
|
||||
extern unsigned char seq_nt4_table[256];
|
||||
|
||||
Executable
+16
@@ -0,0 +1,16 @@
|
||||
ref_data=$1
|
||||
preset=$2
|
||||
|
||||
make clean && make lhash_index=1
|
||||
touch temp_read.fastq
|
||||
./minimap2 -ax $2 $1 temp_read.fastq >/dev/null
|
||||
|
||||
kv_file=$1"_"$2"_minimizers_key_value_sorted"
|
||||
|
||||
full_path=`readlink -f $kv_file`
|
||||
|
||||
cd ./ext/TAL
|
||||
make lisa_hash
|
||||
./build-lisa-hash-index $full_path
|
||||
|
||||
rm ../../temp_read.fastq
|
||||
@@ -1,162 +0,0 @@
|
||||
#include <stdint.h>
|
||||
#include <string.h>
|
||||
#include <stdio.h>
|
||||
#include "minimap.h"
|
||||
#include "mmpriv.h"
|
||||
#include "kalloc.h"
|
||||
|
||||
static const char LogTable256[256] = {
|
||||
#define LT(n) n, n, n, n, n, n, n, n, n, n, n, n, n, n, n, n
|
||||
-1, 0, 1, 1, 2, 2, 2, 2, 3, 3, 3, 3, 3, 3, 3, 3,
|
||||
LT(4), LT(5), LT(5), LT(6), LT(6), LT(6), LT(6),
|
||||
LT(7), LT(7), LT(7), LT(7), LT(7), LT(7), LT(7), LT(7)
|
||||
};
|
||||
|
||||
static inline int ilog2_32(uint32_t v)
|
||||
{
|
||||
uint32_t t, tt;
|
||||
if ((tt = v>>16)) return (t = tt>>8) ? 24 + LogTable256[t] : 16 + LogTable256[tt];
|
||||
return (t = v>>8) ? 8 + LogTable256[t] : LogTable256[v];
|
||||
}
|
||||
|
||||
mm128_t *mm_chain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||
{ // TODO: make sure this works when n has more than 32 bits
|
||||
int32_t k, *f, *p, *t, *v, n_u, n_v;
|
||||
int64_t i, j, st = 0;
|
||||
uint64_t *u, *u2, sum_qspan = 0;
|
||||
float avg_qspan;
|
||||
mm128_t *b, *w;
|
||||
|
||||
if (_u) *_u = 0, *n_u_ = 0;
|
||||
if (n == 0 || a == 0) {
|
||||
kfree(km, a);
|
||||
return 0;
|
||||
}
|
||||
f = (int32_t*)kmalloc(km, n * 4);
|
||||
p = (int32_t*)kmalloc(km, n * 4);
|
||||
t = (int32_t*)kmalloc(km, n * 4);
|
||||
v = (int32_t*)kmalloc(km, n * 4);
|
||||
memset(t, 0, n * 4);
|
||||
|
||||
for (i = 0; i < n; ++i) sum_qspan += a[i].y>>32&0xff;
|
||||
avg_qspan = (float)sum_qspan / n;
|
||||
|
||||
// fill the score and backtrack arrays
|
||||
for (i = 0; i < n; ++i) {
|
||||
uint64_t ri = a[i].x;
|
||||
int64_t max_j = -1;
|
||||
int32_t qi = (int32_t)a[i].y, q_span = a[i].y>>32&0xff; // NB: only 8 bits of span is used!!!
|
||||
int32_t max_f = q_span, n_skip = 0, min_d;
|
||||
int32_t sidi = (a[i].y & MM_SEED_SEG_MASK) >> MM_SEED_SEG_SHIFT;
|
||||
while (st < i && ri > a[st].x + max_dist_x) ++st;
|
||||
if (i - st > max_iter) st = i - max_iter;
|
||||
for (j = i - 1; j >= st; --j) {
|
||||
int64_t dr = ri - a[j].x;
|
||||
int32_t dq = qi - (int32_t)a[j].y, dd, sc, log_dd;
|
||||
int32_t sidj = (a[j].y & MM_SEED_SEG_MASK) >> MM_SEED_SEG_SHIFT;
|
||||
if ((sidi == sidj && dr == 0) || dq <= 0) continue; // don't skip if an anchor is used by multiple segments; see below
|
||||
if ((sidi == sidj && dq > max_dist_y) || dq > max_dist_x) continue;
|
||||
dd = dr > dq? dr - dq : dq - dr;
|
||||
if (sidi == sidj && dd > bw) continue;
|
||||
if (n_segs > 1 && !is_cdna && sidi == sidj && dr > max_dist_y) continue;
|
||||
min_d = dq < dr? dq : dr;
|
||||
sc = min_d > q_span? q_span : dq < dr? dq : dr;
|
||||
log_dd = dd? ilog2_32(dd) : 0;
|
||||
if (is_cdna || sidi != sidj) {
|
||||
int c_log, c_lin;
|
||||
c_lin = (int)(dd * .01 * avg_qspan);
|
||||
c_log = log_dd;
|
||||
if (sidi != sidj && dr == 0) ++sc; // possibly due to overlapping paired ends; give a minor bonus
|
||||
else if (dr > dq || sidi != sidj) sc -= c_lin < c_log? c_lin : c_log;
|
||||
else sc -= c_lin + (c_log>>1);
|
||||
} else sc -= (int)(dd * .01 * avg_qspan) + (log_dd>>1);
|
||||
sc += f[j];
|
||||
if (sc > max_f) {
|
||||
max_f = sc, max_j = j;
|
||||
if (n_skip > 0) --n_skip;
|
||||
} else if (t[j] == i) {
|
||||
if (++n_skip > max_skip)
|
||||
break;
|
||||
}
|
||||
if (p[j] >= 0) t[p[j]] = i;
|
||||
}
|
||||
f[i] = max_f, p[i] = max_j;
|
||||
v[i] = max_j >= 0 && v[max_j] > max_f? v[max_j] : max_f; // v[] keeps the peak score up to i; f[] is the score ending at i, not always the peak
|
||||
}
|
||||
|
||||
// find the ending positions of chains
|
||||
memset(t, 0, n * 4);
|
||||
for (i = 0; i < n; ++i)
|
||||
if (p[i] >= 0) t[p[i]] = 1;
|
||||
for (i = n_u = 0; i < n; ++i)
|
||||
if (t[i] == 0 && v[i] >= min_sc)
|
||||
++n_u;
|
||||
if (n_u == 0) {
|
||||
kfree(km, a); kfree(km, f); kfree(km, p); kfree(km, t); kfree(km, v);
|
||||
return 0;
|
||||
}
|
||||
u = (uint64_t*)kmalloc(km, n_u * 8);
|
||||
for (i = n_u = 0; i < n; ++i) {
|
||||
if (t[i] == 0 && v[i] >= min_sc) {
|
||||
j = i;
|
||||
while (j >= 0 && f[j] < v[j]) j = p[j]; // find the peak that maximizes f[]
|
||||
if (j < 0) j = i; // TODO: this should really be assert(j>=0)
|
||||
u[n_u++] = (uint64_t)f[j] << 32 | j;
|
||||
}
|
||||
}
|
||||
radix_sort_64(u, u + n_u);
|
||||
for (i = 0; i < n_u>>1; ++i) { // reverse, s.t. the highest scoring chain is the first
|
||||
uint64_t t = u[i];
|
||||
u[i] = u[n_u - i - 1], u[n_u - i - 1] = t;
|
||||
}
|
||||
|
||||
// backtrack
|
||||
memset(t, 0, n * 4);
|
||||
for (i = n_v = k = 0; i < n_u; ++i) { // starting from the highest score
|
||||
int32_t n_v0 = n_v, k0 = k;
|
||||
j = (int32_t)u[i];
|
||||
do {
|
||||
v[n_v++] = j;
|
||||
t[j] = 1;
|
||||
j = p[j];
|
||||
} while (j >= 0 && t[j] == 0);
|
||||
if (j < 0) {
|
||||
if (n_v - n_v0 >= min_cnt) u[k++] = u[i]>>32<<32 | (n_v - n_v0);
|
||||
} else if ((int32_t)(u[i]>>32) - f[j] >= min_sc) {
|
||||
if (n_v - n_v0 >= min_cnt) u[k++] = ((u[i]>>32) - f[j]) << 32 | (n_v - n_v0);
|
||||
}
|
||||
if (k0 == k) n_v = n_v0; // no new chain added, reset
|
||||
}
|
||||
*n_u_ = n_u = k, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||
|
||||
// free temporary arrays
|
||||
kfree(km, f); kfree(km, p); kfree(km, t);
|
||||
|
||||
// write the result to b[]
|
||||
b = (mm128_t*)kmalloc(km, n_v * sizeof(mm128_t));
|
||||
for (i = 0, k = 0; i < n_u; ++i) {
|
||||
int32_t k0 = k, ni = (int32_t)u[i];
|
||||
for (j = 0; j < ni; ++j)
|
||||
b[k] = a[v[k0 + (ni - j - 1)]], ++k;
|
||||
}
|
||||
kfree(km, v);
|
||||
|
||||
// sort u[] and a[] by a[].x, such that adjacent chains may be joined (required by mm_join_long)
|
||||
w = (mm128_t*)kmalloc(km, n_u * sizeof(mm128_t));
|
||||
for (i = k = 0; i < n_u; ++i) {
|
||||
w[i].x = b[k].x, w[i].y = (uint64_t)k<<32|i;
|
||||
k += (int32_t)u[i];
|
||||
}
|
||||
radix_sort_128x(w, w + n_u);
|
||||
u2 = (uint64_t*)kmalloc(km, n_u * 8);
|
||||
for (i = k = 0; i < n_u; ++i) {
|
||||
int32_t j = (int32_t)w[i].y, n = (int32_t)u[j];
|
||||
u2[i] = u[j];
|
||||
memcpy(&a[k], &b[w[i].y>>32], n * sizeof(mm128_t));
|
||||
k += n;
|
||||
}
|
||||
if (n_u) memcpy(u, u2, n_u * 8);
|
||||
if (k) memcpy(b, a, k * sizeof(mm128_t)); // write _a_ to _b_ and deallocate _a_ because _a_ is oversized, sometimes a lot
|
||||
kfree(km, a); kfree(km, w); kfree(km, u2);
|
||||
return b;
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
## Contributor Code of Conduct
|
||||
|
||||
As contributors and maintainers of this project, we pledge to respect all
|
||||
people who contribute through reporting issues, posting feature requests,
|
||||
updating documentation, submitting pull requests or patches, and other
|
||||
activities.
|
||||
|
||||
We are committed to making participation in this project a harassment-free
|
||||
experience for everyone, regardless of level of experience, gender, gender
|
||||
identity and expression, sexual orientation, disability, personal appearance,
|
||||
body size, race, age, or religion.
|
||||
|
||||
Examples of unacceptable behavior by participants include the use of sexual
|
||||
language or imagery, derogatory comments or personal attacks, trolling, public
|
||||
or private harassment, insults, or other unprofessional conduct.
|
||||
|
||||
Project maintainers have the right and responsibility to remove, edit, or
|
||||
reject comments, commits, code, wiki edits, issues, and other contributions
|
||||
that are not aligned to this Code of Conduct. Project maintainers or
|
||||
contributors who do not follow the Code of Conduct may be removed from the
|
||||
project team.
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported by opening an issue or contacting the maintainer via email.
|
||||
|
||||
This Code of Conduct is adapted from the [Contributor Covenant][cc], [version
|
||||
1.0.0][v1].
|
||||
|
||||
[cc]: http://contributor-covenant.org/
|
||||
[v1]: http://contributor-covenant.org/version/1/0/0/
|
||||
+6
-6
@@ -31,8 +31,8 @@ To acquire the data used in this cookbook and to install minimap2 and paftools,
|
||||
please follow the command lines below:
|
||||
```sh
|
||||
# install minimap2 executables
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.17/minimap2-2.17_x64-linux.tar.bz2 | tar jxf -
|
||||
cp minimap2-2.17_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.22/minimap2-2.22_x64-linux.tar.bz2 | tar jxf -
|
||||
cp minimap2-2.22_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
||||
export PATH="$PATH:"`pwd` # put the current directory on PATH
|
||||
# download example datasets
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.10/cookbook-data.tgz | tar zxf -
|
||||
@@ -80,12 +80,12 @@ where a `U`-line gives the number of unmapped reads (for SAM input only); a
|
||||
5. Accumulative number of mappings
|
||||
|
||||
For `paftools.js mapeval` to work, you need to encode the true read positions
|
||||
in read names in the right format. For [PBSIM][pbsim] and [mason2][mason2], we
|
||||
in read names in the right format. For [pbsim2][pbsim] and [mason2][mason2], we
|
||||
provide scripts to generate the right format. Simulated reads in this cookbook
|
||||
were created with the following command lines:
|
||||
```sh
|
||||
# in PBSIM source code directory:
|
||||
src/pbsim ../ecoli_ref.fa --depth 1 --sample-fastq sample/sample.fastq
|
||||
# in the pbsim2 source code directory:
|
||||
src/pbsim --depth 1 --length-min 5000 --length-mean 20000 --accuracy-mean 0.95 --hmm_model data/R94.model ../ecoli_ref.fa
|
||||
paftools.js pbsim2fq ../ecoli_ref.fa.fai sd_0001.maf > ../ecoli_pbsim.fa
|
||||
|
||||
# mason2 simulation
|
||||
@@ -237,7 +237,7 @@ with `-x ava-pb` (99% vs 93% with `-x ava-ont`).
|
||||
|
||||
|
||||
|
||||
[pbsim]: https://github.com/pfaucon/PBSIM-PacBio-Simulator
|
||||
[pbsim]: https://github.com/yukiteruono/pbsim2
|
||||
[mason2]: https://github.com/seqan/seqan/tree/master/apps/mason2
|
||||
[paf]: https://github.com/lh3/miniasm/blob/master/PAF.md
|
||||
[v2.10]: https://github.com/lh3/minimap2/releases/tag/v2.10
|
||||
|
||||
@@ -59,6 +59,6 @@ void mm_est_err(const mm_idx_t *mi, int qlen, int n_regs, mm_reg1_t *regs, const
|
||||
n_tot = en - st + 1;
|
||||
if (r->qs > avg_k && r->rs > avg_k) ++n_tot;
|
||||
if (qlen - r->qs > avg_k && l_ref - r->re > avg_k) ++n_tot;
|
||||
r->div = logf((float)n_tot / n_match) / avg_k;
|
||||
r->div = n_match >= n_tot? 0.0f : (float)(1.0 - pow((double)n_match / n_tot, 1.0 / avg_k));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -47,7 +47,7 @@ int main(int argc, char *argv[])
|
||||
printf("%s\t%d\t%d\t%d\t%c\t", ks->name.s, ks->seq.l, r->qs, r->qe, "+-"[r->rev]);
|
||||
printf("%s\t%d\t%d\t%d\t%d\t%d\t%d\tcg:Z:", mi->seq[r->rid].name, mi->seq[r->rid].len, r->rs, r->re, r->mlen, r->blen, r->mapq);
|
||||
for (i = 0; i < r->p->n_cigar; ++i) // IMPORTANT: this gives the CIGAR in the aligned regions. NO soft/hard clippings!
|
||||
printf("%d%c", r->p->cigar[i]>>4, "MIDNSH"[r->p->cigar[i]&0xf]);
|
||||
printf("%d%c", r->p->cigar[i]>>4, MM_CIGAR_STR[r->p->cigar[i]&0xf]);
|
||||
putchar('\n');
|
||||
free(r->p);
|
||||
}
|
||||
|
||||
Submodule
+1
Submodule ext/TAL added at 2a97815a5f
@@ -144,8 +144,8 @@ static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
||||
if (write_tag) mm_sprintf_lite(s, "\tcs:Z:");
|
||||
for (i = q_off = t_off = 0; i < (int)r->p->n_cigar; ++i) {
|
||||
int j, op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||
assert((op >= 0 && op <= 3) || op == 7 || op == 8);
|
||||
if (op == 0 || op == 7 || op == 8) { // match
|
||||
assert((op >= MM_CIGAR_MATCH && op <= MM_CIGAR_N_SKIP) || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH);
|
||||
if (op == MM_CIGAR_MATCH || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH) {
|
||||
int l_tmp = 0;
|
||||
for (j = 0; j < len; ++j) {
|
||||
if (qseq[q_off + j] != tseq[t_off + j]) {
|
||||
@@ -166,12 +166,12 @@ static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
||||
} else mm_sprintf_lite(s, ":%d", l_tmp);
|
||||
}
|
||||
q_off += len, t_off += len;
|
||||
} else if (op == 1) { // insertion to ref
|
||||
} else if (op == MM_CIGAR_INS) {
|
||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||
tmp[j] = "acgtn"[qseq[q_off + j]];
|
||||
mm_sprintf_lite(s, "+%s", tmp);
|
||||
q_off += len;
|
||||
} else if (op == 2) { // deletion from ref
|
||||
} else if (op == MM_CIGAR_DEL) {
|
||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||
tmp[j] = "acgtn"[tseq[t_off + j]];
|
||||
mm_sprintf_lite(s, "-%s", tmp);
|
||||
@@ -192,8 +192,8 @@ static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
||||
if (write_tag) mm_sprintf_lite(s, "\tMD:Z:");
|
||||
for (i = q_off = t_off = 0; i < (int)r->p->n_cigar; ++i) {
|
||||
int j, op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||
assert((op >= 0 && op <= 3) || op == 7 || op == 8);
|
||||
if (op == 0 || op == 7 || op == 8) { // match
|
||||
assert((op >= MM_CIGAR_MATCH && op <= MM_CIGAR_N_SKIP) || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH);
|
||||
if (op == MM_CIGAR_MATCH || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH) {
|
||||
for (j = 0; j < len; ++j) {
|
||||
if (qseq[q_off + j] != tseq[t_off + j]) {
|
||||
mm_sprintf_lite(s, "%d%c", l_MD, "ACGTN"[tseq[t_off + j]]);
|
||||
@@ -201,15 +201,15 @@ static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
||||
} else ++l_MD;
|
||||
}
|
||||
q_off += len, t_off += len;
|
||||
} else if (op == 1) { // insertion to ref
|
||||
} else if (op == MM_CIGAR_INS) {
|
||||
q_off += len;
|
||||
} else if (op == 2) { // deletion from ref
|
||||
} else if (op == MM_CIGAR_DEL) {
|
||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||
tmp[j] = "ACGTN"[tseq[t_off + j]];
|
||||
mm_sprintf_lite(s, "%d^%s", l_MD, tmp);
|
||||
l_MD = 0;
|
||||
t_off += len;
|
||||
} else if (op == 3) { // reference skip
|
||||
} else if (op == MM_CIGAR_N_SKIP) {
|
||||
t_off += len;
|
||||
}
|
||||
}
|
||||
@@ -217,7 +217,7 @@ static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
||||
assert(t_off == r->re - r->rs && q_off == r->qe - r->qs);
|
||||
}
|
||||
|
||||
static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int no_iden, int is_MD, int write_tag)
|
||||
static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int no_iden, int is_MD, int write_tag, int is_qstrand)
|
||||
{
|
||||
extern unsigned char seq_nt4_table[256];
|
||||
int i;
|
||||
@@ -227,14 +227,20 @@ static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_
|
||||
qseq = (uint8_t*)kmalloc(km, r->qe - r->qs);
|
||||
tseq = (uint8_t*)kmalloc(km, r->re - r->rs);
|
||||
tmp = (char*)kmalloc(km, r->re - r->rs > r->qe - r->qs? r->re - r->rs + 1 : r->qe - r->qs + 1);
|
||||
mm_idx_getseq(mi, r->rid, r->rs, r->re, tseq);
|
||||
if (!r->rev) {
|
||||
if (is_qstrand) {
|
||||
mm_idx_getseq2(mi, r->rev, r->rid, r->rs, r->re, tseq);
|
||||
for (i = r->qs; i < r->qe; ++i)
|
||||
qseq[i - r->qs] = seq_nt4_table[(uint8_t)t->seq[i]];
|
||||
} else {
|
||||
for (i = r->qs; i < r->qe; ++i) {
|
||||
uint8_t c = seq_nt4_table[(uint8_t)t->seq[i]];
|
||||
qseq[r->qe - i - 1] = c >= 4? 4 : 3 - c;
|
||||
mm_idx_getseq(mi, r->rid, r->rs, r->re, tseq);
|
||||
if (!r->rev) {
|
||||
for (i = r->qs; i < r->qe; ++i)
|
||||
qseq[i - r->qs] = seq_nt4_table[(uint8_t)t->seq[i]];
|
||||
} else {
|
||||
for (i = r->qs; i < r->qe; ++i) {
|
||||
uint8_t c = seq_nt4_table[(uint8_t)t->seq[i]];
|
||||
qseq[r->qe - i - 1] = c >= 4? 4 : 3 - c;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (is_MD) write_MD_core(s, tseq, qseq, r, tmp, write_tag);
|
||||
@@ -242,14 +248,14 @@ static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_
|
||||
kfree(km, qseq); kfree(km, tseq); kfree(km, tmp);
|
||||
}
|
||||
|
||||
int mm_gen_cs_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int is_MD, int no_iden)
|
||||
int mm_gen_cs_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int is_MD, int no_iden, int is_qstrand)
|
||||
{
|
||||
mm_bseq1_t t;
|
||||
kstring_t str;
|
||||
str.s = *buf, str.l = 0, str.m = *max_len;
|
||||
t.l_seq = strlen(seq);
|
||||
t.seq = (char*)seq;
|
||||
write_cs_or_MD(km, &str, mi, &t, r, no_iden, is_MD, 0);
|
||||
write_cs_or_MD(km, &str, mi, &t, r, no_iden, is_MD, 0, is_qstrand);
|
||||
*max_len = str.m;
|
||||
*buf = str.s;
|
||||
return str.l;
|
||||
@@ -257,24 +263,12 @@ int mm_gen_cs_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, cons
|
||||
|
||||
int mm_gen_cs(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden)
|
||||
{
|
||||
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 0, no_iden);
|
||||
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 0, no_iden, 0);
|
||||
}
|
||||
|
||||
int mm_gen_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq)
|
||||
{
|
||||
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 1, 0);
|
||||
}
|
||||
|
||||
double mm_event_identity(const mm_reg1_t *r)
|
||||
{
|
||||
int32_t i, n_gapo = 0, n_gap = 0;
|
||||
if (r->p == 0) return -1.0f;
|
||||
for (i = 0; i < r->p->n_cigar; ++i) {
|
||||
int32_t op = r->p->cigar[i] & 0xf, len = r->p->cigar[i] >> 4;
|
||||
if (op == 1 || op == 2)
|
||||
++n_gapo, n_gap += len;
|
||||
}
|
||||
return (double)r->mlen / (r->blen - n_gap + n_gapo);
|
||||
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 1, 0, 0);
|
||||
}
|
||||
|
||||
static inline void write_tags(kstring_t *s, const mm_reg1_t *r)
|
||||
@@ -305,7 +299,7 @@ static inline void write_tags(kstring_t *s, const mm_reg1_t *r)
|
||||
if (r->split) mm_sprintf_lite(s, "\tzd:i:%d", r->split);
|
||||
}
|
||||
|
||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int opt_flag, int rep_len)
|
||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len)
|
||||
{
|
||||
s->l = 0;
|
||||
if (r == 0) {
|
||||
@@ -316,7 +310,11 @@ void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const
|
||||
mm_sprintf_lite(s, "%s\t%d\t%d\t%d\t%c\t", t->name, t->l_seq, r->qs, r->qe, "+-"[r->rev]);
|
||||
if (mi->seq[r->rid].name) mm_sprintf_lite(s, "%s", mi->seq[r->rid].name);
|
||||
else mm_sprintf_lite(s, "%d", r->rid);
|
||||
mm_sprintf_lite(s, "\t%d\t%d\t%d", mi->seq[r->rid].len, r->rs, r->re);
|
||||
mm_sprintf_lite(s, "\t%d", mi->seq[r->rid].len);
|
||||
if ((opt_flag & MM_F_QSTRAND) && r->rev)
|
||||
mm_sprintf_lite(s, "\t%d\t%d", mi->seq[r->rid].len - r->re, mi->seq[r->rid].len - r->rs);
|
||||
else
|
||||
mm_sprintf_lite(s, "\t%d\t%d", r->rs, r->re);
|
||||
mm_sprintf_lite(s, "\t%d\t%d", r->mlen, r->blen);
|
||||
mm_sprintf_lite(s, "\t%d", r->mapq);
|
||||
write_tags(s, r);
|
||||
@@ -325,15 +323,15 @@ void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const
|
||||
uint32_t k;
|
||||
mm_sprintf_lite(s, "\tcg:Z:");
|
||||
for (k = 0; k < r->p->n_cigar; ++k)
|
||||
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, "MIDNSHP=XB"[r->p->cigar[k]&0xf]);
|
||||
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, MM_CIGAR_STR[r->p->cigar[k]&0xf]);
|
||||
}
|
||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1);
|
||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1, !!(opt_flag&MM_F_QSTRAND));
|
||||
if ((opt_flag & MM_F_COPY_COMMENT) && t->comment)
|
||||
mm_sprintf_lite(s, "\t%s", t->comment);
|
||||
}
|
||||
|
||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int opt_flag)
|
||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag)
|
||||
{
|
||||
mm_write_paf3(s, mi, t, r, km, opt_flag, -1);
|
||||
}
|
||||
@@ -362,7 +360,7 @@ static inline const mm_reg1_t *get_sam_pri(int n_regs, const mm_reg1_t *regs)
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static void write_sam_cigar(kstring_t *s, int sam_flag, int in_tag, int qlen, const mm_reg1_t *r, int opt_flag)
|
||||
static void write_sam_cigar(kstring_t *s, int sam_flag, int in_tag, int qlen, const mm_reg1_t *r, int64_t opt_flag)
|
||||
{
|
||||
if (r->p == 0) {
|
||||
mm_sprintf_lite(s, "*");
|
||||
@@ -382,17 +380,17 @@ static void write_sam_cigar(kstring_t *s, int sam_flag, int in_tag, int qlen, co
|
||||
assert(clip_len[0] < qlen && clip_len[1] < qlen);
|
||||
if (clip_len[0]) mm_sprintf_lite(s, "%d%c", clip_len[0], clip_char);
|
||||
for (k = 0; k < r->p->n_cigar; ++k)
|
||||
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, "MIDNSHP=XB"[r->p->cigar[k]&0xf]);
|
||||
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, MM_CIGAR_STR[r->p->cigar[k]&0xf]);
|
||||
if (clip_len[1]) mm_sprintf_lite(s, "%d%c", clip_len[1], clip_char);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int opt_flag, int rep_len)
|
||||
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int64_t opt_flag, int rep_len)
|
||||
{
|
||||
const int max_bam_cigar_op = 65535;
|
||||
int flag, n_regs = n_regss[seg_idx], cigar_in_tag = 0;
|
||||
int this_rid = -1, this_pos = -1, this_rev = 0;
|
||||
int this_rid = -1, this_pos = -1;
|
||||
const mm_reg1_t *regs = regss[seg_idx], *r_prev = NULL, *r_next;
|
||||
const mm_reg1_t *r = n_regs > 0 && reg_idx < n_regs && reg_idx >= 0? ®s[reg_idx] : NULL;
|
||||
|
||||
@@ -441,7 +439,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
||||
mm_sprintf_lite(s, "\t%s\t%d\t0\t*", mi->seq[this_rid].name, this_pos+1);
|
||||
} else mm_sprintf_lite(s, "\t*\t0\t0\t*");
|
||||
} else {
|
||||
this_rid = r->rid, this_pos = r->rs, this_rev = r->rev;
|
||||
this_rid = r->rid, this_pos = r->rs;
|
||||
mm_sprintf_lite(s, "\t%s\t%d\t%d\t", mi->seq[r->rid].name, r->rs+1, r->mapq);
|
||||
if ((opt_flag & MM_F_LONG_CIGAR) && r->p && r->p->n_cigar > max_bam_cigar_op - 2) {
|
||||
int n_cigar = r->p->n_cigar;
|
||||
@@ -535,7 +533,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
||||
}
|
||||
}
|
||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1);
|
||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1, 0);
|
||||
if (cigar_in_tag)
|
||||
write_sam_cigar(s, flag, 1, t->l_seq, r, opt_flag);
|
||||
}
|
||||
@@ -547,7 +545,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
||||
s->s[s->l] = 0; // we always have room for an extra byte (see str_enlarge)
|
||||
}
|
||||
|
||||
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int opt_flag)
|
||||
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int64_t opt_flag)
|
||||
{
|
||||
mm_write_sam3(s, mi, t, seg_idx, reg_idx, n_seg, n_regss, regss, km, opt_flag, -1);
|
||||
}
|
||||
|
||||
@@ -20,14 +20,14 @@ static inline void mm_cal_fuzzy_len(mm_reg1_t *r, const mm128_t *a)
|
||||
}
|
||||
}
|
||||
|
||||
static inline void mm_reg_set_coor(mm_reg1_t *r, int32_t qlen, const mm128_t *a)
|
||||
static inline void mm_reg_set_coor(mm_reg1_t *r, int32_t qlen, const mm128_t *a, int is_qstrand)
|
||||
{ // NB: r->as and r->cnt MUST BE set correctly for this function to work
|
||||
int32_t k = r->as, q_span = (int32_t)(a[k].y>>32&0xff);
|
||||
r->rev = a[k].x>>63;
|
||||
r->rid = a[k].x<<1>>33;
|
||||
r->rs = (int32_t)a[k].x + 1 > q_span? (int32_t)a[k].x + 1 - q_span : 0; // NB: target span may be shorter, so this test is necessary
|
||||
r->re = (int32_t)a[k + r->cnt - 1].x + 1;
|
||||
if (!r->rev) {
|
||||
if (!r->rev || is_qstrand) {
|
||||
r->qs = (int32_t)a[k].y + 1 - q_span;
|
||||
r->qe = (int32_t)a[k + r->cnt - 1].y + 1;
|
||||
} else {
|
||||
@@ -49,7 +49,7 @@ static inline uint64_t hash64(uint64_t key)
|
||||
return key;
|
||||
}
|
||||
|
||||
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a) // convert chains to hits
|
||||
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a, int is_qstrand) // convert chains to hits
|
||||
{
|
||||
mm128_t *z, tmp;
|
||||
mm_reg1_t *r;
|
||||
@@ -81,13 +81,29 @@ mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u,
|
||||
ri->cnt = (int32_t)z[i].y;
|
||||
ri->as = z[i].y >> 32;
|
||||
ri->div = -1.0f;
|
||||
mm_reg_set_coor(ri, qlen, a);
|
||||
mm_reg_set_coor(ri, qlen, a, is_qstrand);
|
||||
}
|
||||
kfree(km, z);
|
||||
return r;
|
||||
}
|
||||
|
||||
void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a)
|
||||
void mm_mark_alt(const mm_idx_t *mi, int n, mm_reg1_t *r)
|
||||
{
|
||||
int i;
|
||||
if (mi->n_alt == 0) return;
|
||||
for (i = 0; i < n; ++i)
|
||||
if (mi->seq[r[i].rid].is_alt)
|
||||
r[i].is_alt = 1;
|
||||
}
|
||||
|
||||
static inline int mm_alt_score(int score, float alt_diff_frac)
|
||||
{
|
||||
if (score < 0) return score;
|
||||
score = (int)(score * (1.0 - alt_diff_frac) + .499);
|
||||
return score > 0? score : 1;
|
||||
}
|
||||
|
||||
void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a, int is_qstrand)
|
||||
{
|
||||
if (n <= 0 || n >= r->cnt) return;
|
||||
*r2 = *r;
|
||||
@@ -99,14 +115,14 @@ void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a)
|
||||
r2->score = (int32_t)(r->score * ((float)r2->cnt / r->cnt) + .499);
|
||||
r2->as = r->as + n;
|
||||
if (r->parent == r->id) r2->parent = MM_PARENT_TMP_PRI;
|
||||
mm_reg_set_coor(r2, qlen, a);
|
||||
mm_reg_set_coor(r2, qlen, a, is_qstrand);
|
||||
r->cnt -= r2->cnt;
|
||||
r->score -= r2->score;
|
||||
mm_reg_set_coor(r, qlen, a);
|
||||
mm_reg_set_coor(r, qlen, a, is_qstrand);
|
||||
r->split |= 1, r2->split |= 2;
|
||||
}
|
||||
|
||||
void mm_set_parent(void *km, float mask_level, int n, mm_reg1_t *r, int sub_diff, int hard_mask_level) // and compute mm_reg1_t::subsc
|
||||
void mm_set_parent(void *km, float mask_level, int mask_len, int n, mm_reg1_t *r, int sub_diff, int hard_mask_level, float alt_diff_frac) // and compute mm_reg1_t::subsc
|
||||
{
|
||||
int i, j, k, *w;
|
||||
uint64_t *cov;
|
||||
@@ -146,13 +162,16 @@ skip_uncov:
|
||||
min = ej - sj < ei - si? ej - sj : ei - si;
|
||||
max = ej - sj > ei - si? ej - sj : ei - si;
|
||||
ol = si < sj? (ei < sj? 0 : ei < ej? ei - sj : ej - sj) : (ej < si? 0 : ej < ei? ej - si : ei - si); // overlap length; TODO: this can be simplified
|
||||
if ((float)ol / min - (float)uncov_len / max > mask_level) {
|
||||
int cnt_sub = 0;
|
||||
if ((float)ol / min - (float)uncov_len / max > mask_level && uncov_len <= mask_len) { // then this is a secondary hit
|
||||
int cnt_sub = 0, sci = ri->score;
|
||||
ri->parent = rp->parent;
|
||||
rp->subsc = rp->subsc > ri->score? rp->subsc : ri->score;
|
||||
if (!rp->is_alt && ri->is_alt) sci = mm_alt_score(sci, alt_diff_frac);
|
||||
rp->subsc = rp->subsc > sci? rp->subsc : sci;
|
||||
if (ri->cnt >= rp->cnt) cnt_sub = 1;
|
||||
if (rp->p && ri->p && (rp->rid != ri->rid || rp->rs != ri->rs || rp->re != ri->re || ol != min)) { // the last condition excludes identical hits after DP
|
||||
rp->p->dp_max2 = rp->p->dp_max2 > ri->p->dp_max? rp->p->dp_max2 : ri->p->dp_max;
|
||||
sci = ri->p->dp_max;
|
||||
if (!rp->is_alt && ri->is_alt) sci = mm_alt_score(sci, alt_diff_frac);
|
||||
rp->p->dp_max2 = rp->p->dp_max2 > sci? rp->p->dp_max2 : sci;
|
||||
if (rp->p->dp_max - ri->p->dp_max <= sub_diff) cnt_sub = 1;
|
||||
}
|
||||
if (cnt_sub) ++rp->n_sub;
|
||||
@@ -166,7 +185,7 @@ set_parent_test:
|
||||
kfree(km, w);
|
||||
}
|
||||
|
||||
void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r)
|
||||
void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r, float alt_diff_frac)
|
||||
{
|
||||
int32_t i, n_aux, n = *n_regs, has_cigar = 0, no_cigar = 0;
|
||||
mm128_t *aux;
|
||||
@@ -177,13 +196,11 @@ void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r)
|
||||
t = (mm_reg1_t*)kmalloc(km, n * sizeof(mm_reg1_t));
|
||||
for (i = n_aux = 0; i < n; ++i) {
|
||||
if (r[i].inv || r[i].cnt > 0) { // squeeze out elements with cnt==0 (soft deleted)
|
||||
if (r[i].p) {
|
||||
aux[n_aux].x = (uint64_t)r[i].p->dp_max << 32 | r[i].hash;
|
||||
has_cigar = 1;
|
||||
} else {
|
||||
aux[n_aux].x = (uint64_t)r[i].score << 32 | r[i].hash;
|
||||
no_cigar = 1;
|
||||
}
|
||||
int score;
|
||||
if (r[i].p) score = r[i].p->dp_max, has_cigar = 1;
|
||||
else score = r[i].score, no_cigar = 1;
|
||||
if (r[i].is_alt) score = mm_alt_score(score, alt_diff_frac);
|
||||
aux[n_aux].x = (uint64_t)score << 32 | r[i].hash;
|
||||
aux[n_aux++].y = i;
|
||||
} else if (r[i].p) {
|
||||
free(r[i].p);
|
||||
@@ -295,64 +312,6 @@ int mm_squeeze_a(void *km, int n_regs, mm_reg1_t *regs, mm128_t *a)
|
||||
return as;
|
||||
}
|
||||
|
||||
void mm_join_long(void *km, const mm_mapopt_t *opt, int qlen, int *n_regs_, mm_reg1_t *regs, mm128_t *a)
|
||||
{
|
||||
int i, n_aux, n_regs = *n_regs_, n_drop = 0;
|
||||
uint64_t *aux;
|
||||
|
||||
if (n_regs < 2) return; // nothing to join
|
||||
mm_squeeze_a(km, n_regs, regs, a);
|
||||
|
||||
aux = (uint64_t*)kmalloc(km, n_regs * 8);
|
||||
for (i = n_aux = 0; i < n_regs; ++i)
|
||||
if (regs[i].parent == i || regs[i].parent < 0)
|
||||
aux[n_aux++] = (uint64_t)regs[i].as << 32 | i;
|
||||
radix_sort_64(aux, aux + n_aux);
|
||||
|
||||
for (i = n_aux - 1; i >= 1; --i) {
|
||||
mm_reg1_t *r0 = ®s[(int32_t)aux[i-1]], *r1 = ®s[(int32_t)aux[i]];
|
||||
mm128_t *a0e, *a1s;
|
||||
int max_gap, min_gap, sc_thres, min_flank_len;
|
||||
|
||||
// test
|
||||
if (r0->as + r0->cnt != r1->as) continue; // not adjacent in a[]
|
||||
if (r0->rid != r1->rid || r0->rev != r1->rev) continue; // make sure on the same target and strand
|
||||
a0e = &a[r0->as + r0->cnt - 1];
|
||||
a1s = &a[r1->as];
|
||||
if (a1s->x <= a0e->x || (int32_t)a1s->y <= (int32_t)a0e->y) continue; // keep colinearity
|
||||
max_gap = min_gap = (int32_t)a1s->y - (int32_t)a0e->y;
|
||||
max_gap = a0e->x + max_gap > a1s->x? max_gap : a1s->x - a0e->x;
|
||||
min_gap = a0e->x + min_gap < a1s->x? min_gap : a1s->x - a0e->x;
|
||||
if (max_gap > opt->max_join_long || min_gap > opt->max_join_short) continue;
|
||||
sc_thres = (int)((float)opt->min_join_flank_sc / opt->max_join_long * max_gap + .499);
|
||||
if (r0->score < sc_thres || r1->score < sc_thres) continue; // require good flanking chains
|
||||
min_flank_len = (int)(max_gap * opt->min_join_flank_ratio);
|
||||
if (r0->re - r0->rs < min_flank_len || r0->qe - r0->qs < min_flank_len) continue; // require enough flanking length
|
||||
if (r1->re - r1->rs < min_flank_len || r1->qe - r1->qs < min_flank_len) continue;
|
||||
|
||||
// all conditions satisfied; join
|
||||
a[r1->as].y |= MM_SEED_LONG_JOIN;
|
||||
r0->cnt += r1->cnt, r0->score += r1->score;
|
||||
mm_reg_set_coor(r0, qlen, a);
|
||||
r1->cnt = 0;
|
||||
r1->parent = r0->id;
|
||||
++n_drop;
|
||||
}
|
||||
kfree(km, aux);
|
||||
|
||||
if (n_drop > 0) { // then fix the hits hierarchy
|
||||
for (i = 0; i < n_regs; ++i) { // adjust the mm_reg1_t::parent
|
||||
mm_reg1_t *r = ®s[i];
|
||||
if (r->parent >= 0 && r->id != r->parent) { // fix for secondary hits only
|
||||
if (regs[r->parent].parent >= 0 && regs[r->parent].parent != r->parent)
|
||||
r->parent = regs[r->parent].parent;
|
||||
}
|
||||
}
|
||||
mm_filter_regs(opt, qlen, n_regs_, regs);
|
||||
mm_sync_regs(km, *n_regs_, regs);
|
||||
}
|
||||
}
|
||||
|
||||
mm_seg_t *mm_seg_gen(void *km, uint32_t hash, int n_segs, const int *qlens, int n_regs0, const mm_reg1_t *regs0, int *n_regs, mm_reg1_t **regs, const mm128_t *a)
|
||||
{
|
||||
int s, i, j, acc_qlen[MM_MAX_SEG+1], qlen_sum = 0;
|
||||
@@ -399,7 +358,7 @@ mm_seg_t *mm_seg_gen(void *km, uint32_t hash, int n_segs, const int *qlens, int
|
||||
}
|
||||
}
|
||||
for (s = 0; s < n_segs; ++s) {
|
||||
regs[s] = mm_gen_regs(km, hash, qlens[s], seg[s].n_u, seg[s].u, seg[s].a);
|
||||
regs[s] = mm_gen_regs(km, hash, qlens[s], seg[s].n_u, seg[s].u, seg[s].a, 0);
|
||||
n_regs[s] = seg[s].n_u;
|
||||
for (i = 0; i < n_regs[s]; ++i) {
|
||||
regs[s][i].seg_split = 1;
|
||||
|
||||
@@ -14,6 +14,21 @@
|
||||
#include "mmpriv.h"
|
||||
#include "kvec.h"
|
||||
#include "khash.h"
|
||||
#include <map>
|
||||
#include <fstream>
|
||||
#include <vector>
|
||||
#include <algorithm>
|
||||
#include <x86intrin.h>
|
||||
using namespace std;
|
||||
|
||||
|
||||
extern uint64_t minimizer_lookup_time, alignment_time, dp_time, rmq_time, rmq_t1, rmq_t2, rmq_t3, rmq_t4;
|
||||
#ifdef LISA_HASH
|
||||
#include "lisa_hash.h"
|
||||
extern lisa_hash<uint64_t, uint64_t> *lh;
|
||||
#endif
|
||||
|
||||
|
||||
|
||||
#define idx_hash(a) ((a)>>1)
|
||||
#define idx_eq(a, b) ((a)>>1 == (b)>>1)
|
||||
@@ -52,9 +67,43 @@ mm_idx_t *mm_idx_init(int w, int k, int b, int flag)
|
||||
if (!(mm_dbg_flag & 1)) mi->km = km_init();
|
||||
return mi;
|
||||
}
|
||||
void mm_idx_destroy_mm_hash(mm_idx_t *mi)
|
||||
{
|
||||
//fprintf(stderr, "mm_destroy_hash\n");
|
||||
uint32_t i;
|
||||
if (mi == 0) return;
|
||||
if (mi->h) kh_destroy(str, (khash_t(str)*)mi->h);
|
||||
if (mi->B) {
|
||||
for (i = 0; i < 1U<<mi->b; ++i) {
|
||||
free(mi->B[i].p);
|
||||
free(mi->B[i].a.a);
|
||||
kh_destroy(idx, (idxhash_t*)mi->B[i].h);
|
||||
}
|
||||
}
|
||||
}
|
||||
void mm_idx_destroy_seq(mm_idx_t *mi)
|
||||
{
|
||||
//fprintf(stderr, "mm_destroy_seq\n");
|
||||
|
||||
uint32_t i;
|
||||
if (mi == 0) return;
|
||||
if (mi->I) {
|
||||
for (i = 0; i < mi->n_seq; ++i)
|
||||
free(mi->I[i].a);
|
||||
free(mi->I);
|
||||
}
|
||||
if (!mi->km) {
|
||||
for (i = 0; i < mi->n_seq; ++i)
|
||||
free(mi->seq[i].name);
|
||||
free(mi->seq);
|
||||
} else km_destroy(mi->km);
|
||||
free(mi->B); free(mi->S); free(mi);
|
||||
}
|
||||
|
||||
|
||||
void mm_idx_destroy(mm_idx_t *mi)
|
||||
{
|
||||
|
||||
uint32_t i;
|
||||
if (mi == 0) return;
|
||||
if (mi->h) kh_destroy(str, (khash_t(str)*)mi->h);
|
||||
@@ -96,6 +145,317 @@ const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n)
|
||||
return &b->p[kh_val(h, k)>>32];
|
||||
}
|
||||
}
|
||||
//Output minimap2's hash table entries
|
||||
class hash_entry {
|
||||
public:
|
||||
uint64_t key;
|
||||
uint64_t n;
|
||||
uint64_t *p;
|
||||
hash_entry(uint64_t k, uint64_t n_, uint64_t *p_){
|
||||
key = k;
|
||||
n = n_;
|
||||
p = p_;
|
||||
}
|
||||
|
||||
};
|
||||
bool key_sort( hash_entry i1, hash_entry i2)
|
||||
{
|
||||
return (i1.key < i2.key);
|
||||
}
|
||||
|
||||
#if 0
|
||||
void mm_idx_load_key_value_lisa(const char* f_name, const mm_idx_t *mi)
|
||||
{
|
||||
uint64_t tic = __rdtsc();
|
||||
std::vector<hash_entry> v_hash;
|
||||
|
||||
//ofstream f(f_name);
|
||||
fprintf(stderr, "Building sorted key-val map\n");
|
||||
|
||||
uint32_t i,j;
|
||||
uint64_t num_values = 0;
|
||||
for (i = 0; i < 1U<<mi->b; ++i) {
|
||||
|
||||
|
||||
//fprintf(stderr, "BucketID %lu \n", i);
|
||||
idxhash_t *h = (idxhash_t*)mi->B[i].h;
|
||||
khint_t k;
|
||||
if (h == 0) continue;
|
||||
for (k = 0; k < kh_end(h); ++k){
|
||||
if (kh_exist(h, k)) {
|
||||
uint64_t key = kh_key(h, k), bucket_id = i;
|
||||
key = key>>1;
|
||||
|
||||
key = key<<mi->b | bucket_id;
|
||||
|
||||
if(kh_key(h, k)&1)
|
||||
{
|
||||
//print key value
|
||||
//fprintf(stderr, "%llu %llu %llu\n", key, kh_val(h, k), 0);
|
||||
v_hash.push_back(hash_entry(key, kh_val(h, k), NULL));
|
||||
}
|
||||
else
|
||||
{ // print key
|
||||
uint32_t n = (uint32_t)kh_val(h, k);
|
||||
//fprintf(stderr, "%llu %llu %llu ", key, kh_val(h, k), n);
|
||||
// for 0 to lsb 32 val
|
||||
// print b->p[msb 32 of val]
|
||||
|
||||
v_hash.push_back(hash_entry(key, n, &mi->B[i].p[(kh_val(h, k)>>32) + 0]));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
sort(v_hash.begin(), v_hash.end(), key_sort);
|
||||
fprintf(stderr, "Sorted map building time = %lld \n", __rdtsc() - tic);
|
||||
fprintf(stderr, "Storing hash to %s \n", f_name);
|
||||
tic = __rdtsc();
|
||||
|
||||
int64_t itr_p = 0;
|
||||
for( int i = 0; i < v_hash.size(); i++){
|
||||
|
||||
if(v_hash[i].p == NULL){
|
||||
//f<<v_hash[i].key << " "<<1<<"\n"<<v_hash[i].n<<" \n";
|
||||
|
||||
lh->p[itr_p++] = v_hash[i].n;
|
||||
continue;
|
||||
}
|
||||
|
||||
//f<<v_hash[i].key << " "<<v_hash[i].n<<endl;
|
||||
|
||||
for(int j = 0; j < v_hash[i].n; j++){
|
||||
// f<<v_hash[i].p[j]<<" ";
|
||||
lh->p[itr_p++] = v_hash[i].p[j];
|
||||
num_values++;
|
||||
}
|
||||
//f<<endl;
|
||||
|
||||
}
|
||||
|
||||
//f.close();
|
||||
|
||||
string size_file_name = (string) f_name + "_size";
|
||||
ofstream size_f(size_file_name);
|
||||
size_f<<v_hash.size()<<" "<<num_values;
|
||||
size_f.close();
|
||||
|
||||
string prefix = (string)f_name + "_keys";
|
||||
string keys_bin_file_name = prefix + ".uint64";
|
||||
ofstream wf(keys_bin_file_name, ios::out | ios::binary);
|
||||
wf.write((char*)&key_list[0], (key_list.size())*sizeof(uint64_t));
|
||||
wf.close();
|
||||
|
||||
key_list.clear();
|
||||
|
||||
m.clear();
|
||||
v_hash.clear();
|
||||
|
||||
fprintf(stderr, "Index store File IO time %lld \n", __rdtsc() - tic);
|
||||
|
||||
}
|
||||
#endif
|
||||
|
||||
void mm_idx_dump_hash(const char* f_name, const mm_idx_t *mi)
|
||||
{
|
||||
uint64_t tic = __rdtsc();
|
||||
//std::map<uint64_t, vector<uint64_t>> m;
|
||||
std::vector<hash_entry> v_hash;
|
||||
|
||||
//ofstream f(f_name);
|
||||
fprintf(stderr, "Building sorted key-val map\n");
|
||||
|
||||
uint32_t i,j;
|
||||
uint64_t num_values = 0;
|
||||
for (i = 0; i < 1U<<mi->b; ++i) {
|
||||
|
||||
|
||||
//fprintf(stderr, "BucketID %lu \n", i);
|
||||
idxhash_t *h = (idxhash_t*)mi->B[i].h;
|
||||
khint_t k;
|
||||
if (h == 0) continue;
|
||||
for (k = 0; k < kh_end(h); ++k){
|
||||
if (kh_exist(h, k)) {
|
||||
uint64_t key = kh_key(h, k), bucket_id = i;
|
||||
key = key>>1;
|
||||
|
||||
key = key<<mi->b | bucket_id;
|
||||
|
||||
if(kh_key(h, k)&1)
|
||||
{
|
||||
//print key value
|
||||
//fprintf(stderr, "%llu %llu %llu\n", key, kh_val(h, k), 0);
|
||||
//m[key].push_back(kh_val(h, k));
|
||||
v_hash.push_back(hash_entry(key, kh_val(h, k), NULL));
|
||||
}
|
||||
else
|
||||
{ // print key
|
||||
uint32_t n = (uint32_t)kh_val(h, k);
|
||||
//fprintf(stderr, "%llu %llu %llu ", key, kh_val(h, k), n);
|
||||
// for 0 to lsb 32 val
|
||||
// print b->p[msb 32 of val]
|
||||
|
||||
v_hash.push_back(hash_entry(key, n, &mi->B[i].p[(kh_val(h, k)>>32) + 0]));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
sort(v_hash.begin(), v_hash.end(), key_sort);
|
||||
fprintf(stderr, "Sorted map building time = %lld \n", __rdtsc() - tic);
|
||||
fprintf(stderr, "Storing hash to %s \n", f_name);
|
||||
tic = __rdtsc();
|
||||
|
||||
vector<uint64_t> key_list;
|
||||
vector<uint64_t> val_list;
|
||||
vector<uint64_t> p_list;
|
||||
/*
|
||||
key_list.push_back(m.size());
|
||||
for(auto k : m){
|
||||
key_list.push_back(k.first);
|
||||
f<<k.first << " "<<k.second.size()<<endl;
|
||||
for(int j = 0; j < k.second.size(); j++){
|
||||
f<<k.second[j]<<" ";
|
||||
num_values++;
|
||||
}
|
||||
f<<endl;
|
||||
}
|
||||
*/
|
||||
|
||||
|
||||
key_list.push_back(v_hash.size());
|
||||
int64_t itr_p = 0;
|
||||
uint64_t sum_pos = 0;
|
||||
string f1_name = (string)f_name + "_pos_bin";
|
||||
string f2_name = (string)f_name + "_val_bin";
|
||||
ofstream f1(f1_name, ios::out | ios::binary);
|
||||
ofstream f2(f2_name, ios::out | ios::binary);
|
||||
for( int i = 0; i < v_hash.size(); i++){
|
||||
key_list.push_back(v_hash[i].key);
|
||||
if(v_hash[i].p == NULL){
|
||||
//f<<v_hash[i].key << " "<<1<<"\n"<<v_hash[i].n<<" \n";
|
||||
val_list.push_back(sum_pos<<32|(uint64_t)1);
|
||||
sum_pos+=1;
|
||||
p_list.push_back(v_hash[i].n);
|
||||
num_values++;
|
||||
continue;
|
||||
}
|
||||
|
||||
//f<<v_hash[i].key << " "<<v_hash[i].n<<endl;
|
||||
val_list.push_back(sum_pos<<32|(uint64_t)v_hash[i].n);
|
||||
sum_pos+=v_hash[i].n;
|
||||
|
||||
num_values+=v_hash[i].n;
|
||||
|
||||
|
||||
for(int j = 0; j < v_hash[i].n; j++){
|
||||
//f<<v_hash[i].p[j]<<" ";
|
||||
p_list.push_back(v_hash[i].p[j]);
|
||||
}
|
||||
// f<<endl;
|
||||
|
||||
}
|
||||
f1.write((char*)&val_list[0], (val_list.size())*sizeof(uint64_t));
|
||||
f2.write((char*)&p_list[0], (p_list.size())*sizeof(uint64_t));
|
||||
f1.close();
|
||||
f2.close();
|
||||
fprintf(stderr, "Index sorted SoA time %lld \n", __rdtsc() - tic);
|
||||
|
||||
//f.close();
|
||||
|
||||
string size_file_name = (string) f_name + "_size";
|
||||
ofstream size_f(size_file_name);
|
||||
size_f<<v_hash.size()<<" "<<num_values;
|
||||
size_f.close();
|
||||
|
||||
string prefix = (string)f_name + "_keys";
|
||||
string keys_bin_file_name = prefix + ".uint64";
|
||||
ofstream wf(keys_bin_file_name, ios::out | ios::binary);
|
||||
wf.write((char*)&key_list[0], (key_list.size())*sizeof(uint64_t));
|
||||
wf.close();
|
||||
|
||||
key_list.clear();
|
||||
|
||||
//m.clear();
|
||||
v_hash.clear();
|
||||
|
||||
fprintf(stderr, "Index store File IO time %lld \n", __rdtsc() - tic);
|
||||
|
||||
}
|
||||
void mm_idx_dump_hash_1(const char* f_name, const mm_idx_t *mi)
|
||||
{
|
||||
uint64_t tic = __rdtsc();
|
||||
std::map<uint64_t, vector<uint64_t>> m;
|
||||
|
||||
ofstream f(f_name);
|
||||
fprintf(stderr, "Building sorted key-val map\n");
|
||||
|
||||
uint32_t i,j;
|
||||
uint64_t num_values = 0;
|
||||
for (i = 0; i < 1U<<mi->b; ++i) {
|
||||
|
||||
|
||||
//fprintf(stderr, "BucketID %lu \n", i);
|
||||
idxhash_t *h = (idxhash_t*)mi->B[i].h;
|
||||
khint_t k;
|
||||
if (h == 0) continue;
|
||||
for (k = 0; k < kh_end(h); ++k){
|
||||
if (kh_exist(h, k)) {
|
||||
uint64_t key = kh_key(h, k), bucket_id = i;
|
||||
key = key>>1;
|
||||
|
||||
key = key<<mi->b | bucket_id;
|
||||
|
||||
if(kh_key(h, k)&1)
|
||||
{
|
||||
//print key value
|
||||
//fprintf(stderr, "%llu %llu %llu\n", key, kh_val(h, k), 0);
|
||||
m[key].push_back(kh_val(h, k));
|
||||
}
|
||||
else
|
||||
{ // print key
|
||||
uint32_t n = (uint32_t)kh_val(h, k);
|
||||
//fprintf(stderr, "%llu %llu %llu ", key, kh_val(h, k), n);
|
||||
// for 0 to lsb 32 val
|
||||
// print b->p[msb 32 of val]
|
||||
for(j = 0; j < n; j++)
|
||||
{
|
||||
//fprintf(stderr, "%llu ", mi->B[i].p[(kh_val(h, k)>>32) + j]);
|
||||
m[key].push_back(mi->B[i].p[(kh_val(h, k)>>32) + j]);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
fprintf(stderr, "Sorted map building time = %lld \n", __rdtsc() - tic);
|
||||
fprintf(stderr, "Storing hash to %s \n", f_name);
|
||||
tic = __rdtsc();
|
||||
vector<uint64_t> key_list;
|
||||
key_list.push_back(m.size());
|
||||
for(auto k : m){
|
||||
key_list.push_back(k.first);
|
||||
f<<k.first << " "<<k.second.size()<<endl;
|
||||
for(int j = 0; j < k.second.size(); j++){
|
||||
f<<k.second[j]<<" ";
|
||||
num_values++;
|
||||
}
|
||||
f<<endl;
|
||||
}
|
||||
f.close();
|
||||
string size_file_name = (string) f_name + "_size";
|
||||
ofstream size_f(size_file_name);
|
||||
size_f<<m.size()<<" "<<num_values;
|
||||
size_f.close();
|
||||
|
||||
string prefix = (string)f_name + "_keys";
|
||||
string keys_bin_file_name = prefix + ".uint64";
|
||||
ofstream wf(keys_bin_file_name, ios::out | ios::binary);
|
||||
wf.write((char*)&key_list[0], (key_list.size())*sizeof(uint64_t));
|
||||
wf.close();
|
||||
|
||||
key_list.clear();
|
||||
m.clear();
|
||||
fprintf(stderr, "Index store File IO time %lld \n", __rdtsc() - tic);
|
||||
|
||||
}
|
||||
|
||||
void mm_idx_stat(const mm_idx_t *mi)
|
||||
{
|
||||
@@ -119,6 +479,7 @@ void mm_idx_stat(const mm_idx_t *mi)
|
||||
}
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] distinct minimizers: %d (%.2f%% are singletons); average occurrences: %.3lf; average spacing: %.3lf; total length: %ld\n",
|
||||
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), n, 100.0*n1/n, (double)sum / n, (double)len / sum, (long)len);
|
||||
fprintf(stderr, "minimizer-lookup: %lld dp: %lld rmq: %lld rmq_t1: %lld rmq_t2: %lld rmq_t3: %lld rmq_t4: %lld alignment: %lld \n", minimizer_lookup_time, dp_time, rmq_time, rmq_t1, rmq_t2, rmq_t3, rmq_t4, alignment_time);
|
||||
}
|
||||
|
||||
int mm_idx_index_name(mm_idx_t *mi)
|
||||
@@ -161,6 +522,28 @@ int mm_idx_getseq(const mm_idx_t *mi, uint32_t rid, uint32_t st, uint32_t en, ui
|
||||
return en - st;
|
||||
}
|
||||
|
||||
int mm_idx_getseq_rev(const mm_idx_t *mi, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq)
|
||||
{
|
||||
uint64_t i, st1, en1;
|
||||
const mm_idx_seq_t *s;
|
||||
if (rid >= mi->n_seq || st >= mi->seq[rid].len) return -1;
|
||||
s = &mi->seq[rid];
|
||||
if (en > s->len) en = s->len;
|
||||
st1 = s->offset + (s->len - en);
|
||||
en1 = s->offset + (s->len - st);
|
||||
for (i = st1; i < en1; ++i) {
|
||||
uint8_t c = mm_seq4_get(mi->S, i);
|
||||
seq[en1 - i - 1] = c < 4? 3 - c : c;
|
||||
}
|
||||
return en - st;
|
||||
}
|
||||
|
||||
int mm_idx_getseq2(const mm_idx_t *mi, int is_rev, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq)
|
||||
{
|
||||
if (is_rev) return mm_idx_getseq_rev(mi, rid, st, en, seq);
|
||||
else return mm_idx_getseq(mi, rid, st, en, seq);
|
||||
}
|
||||
|
||||
int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f)
|
||||
{
|
||||
int i;
|
||||
@@ -316,6 +699,7 @@ static void *worker_pipeline(void *shared, int step, void *in)
|
||||
} else seq->name = 0;
|
||||
seq->len = s->seq[i].l_seq;
|
||||
seq->offset = p->sum_len;
|
||||
seq->is_alt = 0;
|
||||
// copy the sequence
|
||||
if (!(p->mi->flag & MM_I_NO_SEQ)) {
|
||||
for (j = 0; j < seq->len; ++j) { // TODO: this is not the fastest way, but let's first see if speed matters here
|
||||
@@ -414,6 +798,7 @@ mm_idx_t *mm_idx_str(int w, int k, int is_hpc, int bucket_bits, int n, const cha
|
||||
}
|
||||
p->offset = sum_len;
|
||||
p->len = strlen(s);
|
||||
p->is_alt = 0;
|
||||
for (j = 0; j < p->len; ++j) {
|
||||
int c = seq_nt4_table[(uint8_t)s[j]];
|
||||
uint64_t o = sum_len + j;
|
||||
@@ -500,6 +885,7 @@ mm_idx_t *mm_idx_load(FILE *fp)
|
||||
}
|
||||
fread(&s->len, 4, 1, fp);
|
||||
s->offset = sum_len;
|
||||
s->is_alt = 0;
|
||||
sum_len += s->len;
|
||||
}
|
||||
for (i = 0; i < 1<<mi->b; ++i) {
|
||||
@@ -607,6 +993,30 @@ int mm_idx_reader_eof(const mm_idx_reader_t *r) // TODO: in extremely rare cases
|
||||
#include "kseq.h"
|
||||
KSTREAM_DECLARE(gzFile, gzread)
|
||||
|
||||
int mm_idx_alt_read(mm_idx_t *mi, const char *fn)
|
||||
{
|
||||
int n_alt = 0;
|
||||
gzFile fp;
|
||||
kstream_t *ks;
|
||||
kstring_t str = {0,0,0};
|
||||
fp = fn && strcmp(fn, "-")? gzopen(fn, "r") : gzdopen(fileno(stdin), "r");
|
||||
if (fp == 0) return -1;
|
||||
ks = ks_init(fp);
|
||||
if (mi->h == 0) mm_idx_index_name(mi);
|
||||
while (ks_getuntil(ks, KS_SEP_LINE, &str, 0) >= 0) {
|
||||
char *p;
|
||||
int id;
|
||||
for (p = str.s; *p && !isspace(*p); ++p) { }
|
||||
*p = 0;
|
||||
id = mm_idx_name2id(mi, str.s);
|
||||
if (id >= 0) mi->seq[id].is_alt = 1, ++n_alt;
|
||||
}
|
||||
mi->n_alt = n_alt;
|
||||
if (mm_verbose >= 3)
|
||||
fprintf(stderr, "[M::%s] found %d ALT contigs\n", __func__, n_alt);
|
||||
return n_alt;
|
||||
}
|
||||
|
||||
#define sort_key_bed(a) ((a).st)
|
||||
KRADIX_SORT_INIT(bed, mm_idx_intv1_t, sort_key_bed, 4)
|
||||
|
||||
@@ -627,7 +1037,7 @@ mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc
|
||||
char *p, *q, *bl, *bs;
|
||||
int32_t i, id = -1, n_blk = 0;
|
||||
for (p = q = str.s, i = 0;; ++p) {
|
||||
if (*p == 0 || isspace(*p)) {
|
||||
if (*p == 0 || *p == '\t') {
|
||||
int32_t c = *p;
|
||||
*p = 0;
|
||||
if (i == 0) { // chr
|
||||
|
||||
@@ -34,4 +34,43 @@ void km_stat(const void *_km, km_stat_t *s);
|
||||
KREALLOC((km), (a), (m)); \
|
||||
} while (0)
|
||||
|
||||
#ifndef klib_unused
|
||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
||||
#define klib_unused __attribute__ ((__unused__))
|
||||
#else
|
||||
#define klib_unused
|
||||
#endif
|
||||
#endif /* klib_unused */
|
||||
|
||||
#define KALLOC_POOL_INIT2(SCOPE, name, kmptype_t) \
|
||||
typedef struct { \
|
||||
size_t cnt, n, max; \
|
||||
kmptype_t **buf; \
|
||||
void *km; \
|
||||
} kmp_##name##_t; \
|
||||
SCOPE kmp_##name##_t *kmp_init_##name(void *km) { \
|
||||
kmp_##name##_t *mp; \
|
||||
KCALLOC(km, mp, 1); \
|
||||
mp->km = km; \
|
||||
return mp; \
|
||||
} \
|
||||
SCOPE void kmp_destroy_##name(kmp_##name##_t *mp) { \
|
||||
size_t k; \
|
||||
for (k = 0; k < mp->n; ++k) kfree(mp->km, mp->buf[k]); \
|
||||
kfree(mp->km, mp->buf); kfree(mp->km, mp); \
|
||||
} \
|
||||
SCOPE kmptype_t *kmp_alloc_##name(kmp_##name##_t *mp) { \
|
||||
++mp->cnt; \
|
||||
if (mp->n == 0) return (kmptype_t*)kcalloc(mp->km, 1, sizeof(kmptype_t)); \
|
||||
return mp->buf[--mp->n]; \
|
||||
} \
|
||||
SCOPE void kmp_free_##name(kmp_##name##_t *mp, kmptype_t *p) { \
|
||||
--mp->cnt; \
|
||||
if (mp->n == mp->max) KEXPAND(mp->km, mp->buf, mp->max); \
|
||||
mp->buf[mp->n++] = p; \
|
||||
}
|
||||
|
||||
#define KALLOC_POOL_INIT(name, kmptype_t) \
|
||||
KALLOC_POOL_INIT2(static inline klib_unused, name, kmptype_t)
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,474 @@
|
||||
/* The MIT License
|
||||
|
||||
Copyright (c) 2019 by Attractive Chaos <attractor@live.co.uk>
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
*/
|
||||
|
||||
/* An example:
|
||||
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <stdlib.h>
|
||||
#include "krmq.h"
|
||||
|
||||
struct my_node {
|
||||
char key;
|
||||
KRMQ_HEAD(struct my_node) head;
|
||||
};
|
||||
#define my_cmp(p, q) (((q)->key < (p)->key) - ((p)->key < (q)->key))
|
||||
KRMQ_INIT(my, struct my_node, head, my_cmp)
|
||||
|
||||
int main(void) {
|
||||
const char *str = "MNOLKQOPHIA"; // from wiki, except a duplicate
|
||||
struct my_node *root = 0;
|
||||
int i, l = strlen(str);
|
||||
for (i = 0; i < l; ++i) { // insert in the input order
|
||||
struct my_node *q, *p = malloc(sizeof(*p));
|
||||
p->key = str[i];
|
||||
q = krmq_insert(my, &root, p, 0);
|
||||
if (p != q) free(p); // if already present, free
|
||||
}
|
||||
krmq_itr_t(my) itr;
|
||||
krmq_itr_first(my, root, &itr); // place at first
|
||||
do { // traverse
|
||||
const struct my_node *p = krmq_at(&itr);
|
||||
putchar(p->key);
|
||||
free((void*)p); // free node
|
||||
} while (krmq_itr_next(my, &itr));
|
||||
putchar('\n');
|
||||
return 0;
|
||||
}
|
||||
*/
|
||||
|
||||
#ifndef KRMQ_H
|
||||
#define KRMQ_H
|
||||
|
||||
#ifdef __STRICT_ANSI__
|
||||
#define inline __inline__
|
||||
#endif
|
||||
|
||||
#define KRMQ_MAX_DEPTH 64
|
||||
|
||||
#define krmq_size(head, p) ((p)? (p)->head.size : 0)
|
||||
#define krmq_size_child(head, q, i) ((q)->head.p[(i)]? (q)->head.p[(i)]->head.size : 0)
|
||||
|
||||
#define KRMQ_HEAD(__type) \
|
||||
struct { \
|
||||
__type *p[2], *s; \
|
||||
signed char balance; /* balance factor */ \
|
||||
unsigned size; /* #elements in subtree */ \
|
||||
}
|
||||
|
||||
#define __KRMQ_FIND(suf, __scope, __type, __head, __cmp) \
|
||||
__scope __type *krmq_find_##suf(const __type *root, const __type *x, unsigned *cnt_) { \
|
||||
const __type *p = root; \
|
||||
unsigned cnt = 0; \
|
||||
while (p != 0) { \
|
||||
int cmp; \
|
||||
cmp = __cmp(x, p); \
|
||||
if (cmp >= 0) cnt += krmq_size_child(__head, p, 0) + 1; \
|
||||
if (cmp < 0) p = p->__head.p[0]; \
|
||||
else if (cmp > 0) p = p->__head.p[1]; \
|
||||
else break; \
|
||||
} \
|
||||
if (cnt_) *cnt_ = cnt; \
|
||||
return (__type*)p; \
|
||||
} \
|
||||
__scope __type *krmq_interval_##suf(const __type *root, const __type *x, __type **lower, __type **upper) { \
|
||||
const __type *p = root, *l = 0, *u = 0; \
|
||||
while (p != 0) { \
|
||||
int cmp; \
|
||||
cmp = __cmp(x, p); \
|
||||
if (cmp < 0) u = p, p = p->__head.p[0]; \
|
||||
else if (cmp > 0) l = p, p = p->__head.p[1]; \
|
||||
else { l = u = p; break; } \
|
||||
} \
|
||||
if (lower) *lower = (__type*)l; \
|
||||
if (upper) *upper = (__type*)u; \
|
||||
return (__type*)p; \
|
||||
}
|
||||
|
||||
#define __KRMQ_RMQ(suf, __scope, __type, __head, __cmp, __lt2) \
|
||||
__scope __type *krmq_rmq_##suf(const __type *root, const __type *lo, const __type *up) { /* CLOSED interval */ \
|
||||
const __type *p = root, *path[2][KRMQ_MAX_DEPTH], *min; \
|
||||
int plen[2] = {0, 0}, pcmp[2][KRMQ_MAX_DEPTH], i, cmp, lca; \
|
||||
if (root == 0) return 0; \
|
||||
while (p) { \
|
||||
cmp = __cmp(lo, p); \
|
||||
path[0][plen[0]] = p, pcmp[0][plen[0]++] = cmp; \
|
||||
if (cmp < 0) p = p->__head.p[0]; \
|
||||
else if (cmp > 0) p = p->__head.p[1]; \
|
||||
else break; \
|
||||
} \
|
||||
p = root; \
|
||||
while (p) { \
|
||||
cmp = __cmp(up, p); \
|
||||
path[1][plen[1]] = p, pcmp[1][plen[1]++] = cmp; \
|
||||
if (cmp < 0) p = p->__head.p[0]; \
|
||||
else if (cmp > 0) p = p->__head.p[1]; \
|
||||
else break; \
|
||||
} \
|
||||
for (i = 0; i < plen[0] && i < plen[1]; ++i) /* find the LCA */ \
|
||||
if (path[0][i] == path[1][i] && pcmp[0][i] <= 0 && pcmp[1][i] >= 0) \
|
||||
break; \
|
||||
if (i == plen[0] || i == plen[1]) return 0; /* no elements in the closed interval */ \
|
||||
lca = i, min = path[0][lca]; \
|
||||
for (i = lca + 1; i < plen[0]; ++i) { \
|
||||
if (pcmp[0][i] <= 0) { \
|
||||
if (__lt2(path[0][i], min)) min = path[0][i]; \
|
||||
if (path[0][i]->__head.p[1] && __lt2(path[0][i]->__head.p[1]->__head.s, min)) \
|
||||
min = path[0][i]->__head.p[1]->__head.s; \
|
||||
} \
|
||||
} \
|
||||
for (i = lca + 1; i < plen[1]; ++i) { \
|
||||
if (pcmp[1][i] >= 0) { \
|
||||
if (__lt2(path[1][i], min)) min = path[1][i]; \
|
||||
if (path[1][i]->__head.p[0] && __lt2(path[1][i]->__head.p[0]->__head.s, min)) \
|
||||
min = path[1][i]->__head.p[0]->__head.s; \
|
||||
} \
|
||||
} \
|
||||
return (__type*)min; \
|
||||
}
|
||||
|
||||
#define __KRMQ_ROTATE(suf, __type, __head, __lt2) \
|
||||
/* */ \
|
||||
static inline void krmq_update_min_##suf(__type *p, const __type *q, const __type *r) { \
|
||||
p->__head.s = !q || __lt2(p, q->__head.s)? p : q->__head.s; \
|
||||
p->__head.s = !r || __lt2(p->__head.s, r->__head.s)? p->__head.s : r->__head.s; \
|
||||
} \
|
||||
/* one rotation: (a,(b,c)q)p => ((a,b)p,c)q */ \
|
||||
static inline __type *krmq_rotate1_##suf(__type *p, int dir) { /* dir=0 to left; dir=1 to right */ \
|
||||
int opp = 1 - dir; /* opposite direction */ \
|
||||
__type *q = p->__head.p[opp], *s = p->__head.s; \
|
||||
unsigned size_p = p->__head.size; \
|
||||
p->__head.size -= q->__head.size - krmq_size_child(__head, q, dir); \
|
||||
q->__head.size = size_p; \
|
||||
krmq_update_min_##suf(p, p->__head.p[dir], q->__head.p[dir]); \
|
||||
q->__head.s = s; \
|
||||
p->__head.p[opp] = q->__head.p[dir]; \
|
||||
q->__head.p[dir] = p; \
|
||||
return q; \
|
||||
} \
|
||||
/* two consecutive rotations: (a,((b,c)r,d)q)p => ((a,b)p,(c,d)q)r */ \
|
||||
static inline __type *krmq_rotate2_##suf(__type *p, int dir) { \
|
||||
int b1, opp = 1 - dir; \
|
||||
__type *q = p->__head.p[opp], *r = q->__head.p[dir], *s = p->__head.s; \
|
||||
unsigned size_x_dir = krmq_size_child(__head, r, dir); \
|
||||
r->__head.size = p->__head.size; \
|
||||
p->__head.size -= q->__head.size - size_x_dir; \
|
||||
q->__head.size -= size_x_dir + 1; \
|
||||
krmq_update_min_##suf(p, p->__head.p[dir], r->__head.p[dir]); \
|
||||
krmq_update_min_##suf(q, q->__head.p[opp], r->__head.p[opp]); \
|
||||
r->__head.s = s; \
|
||||
p->__head.p[opp] = r->__head.p[dir]; \
|
||||
r->__head.p[dir] = p; \
|
||||
q->__head.p[dir] = r->__head.p[opp]; \
|
||||
r->__head.p[opp] = q; \
|
||||
b1 = dir == 0? +1 : -1; \
|
||||
if (r->__head.balance == b1) q->__head.balance = 0, p->__head.balance = -b1; \
|
||||
else if (r->__head.balance == 0) q->__head.balance = p->__head.balance = 0; \
|
||||
else q->__head.balance = b1, p->__head.balance = 0; \
|
||||
r->__head.balance = 0; \
|
||||
return r; \
|
||||
}
|
||||
|
||||
#define __KRMQ_INSERT(suf, __scope, __type, __head, __cmp, __lt2) \
|
||||
__scope __type *krmq_insert_##suf(__type **root_, __type *x, unsigned *cnt_) { \
|
||||
unsigned char stack[KRMQ_MAX_DEPTH]; \
|
||||
__type *path[KRMQ_MAX_DEPTH]; \
|
||||
__type *bp, *bq; \
|
||||
__type *p, *q, *r = 0; /* _r_ is potentially the new root */ \
|
||||
int i, which = 0, top, b1, path_len; \
|
||||
unsigned cnt = 0; \
|
||||
bp = *root_, bq = 0; \
|
||||
/* find the insertion location */ \
|
||||
for (p = bp, q = bq, top = path_len = 0; p; q = p, p = p->__head.p[which]) { \
|
||||
int cmp; \
|
||||
cmp = __cmp(x, p); \
|
||||
if (cmp >= 0) cnt += krmq_size_child(__head, p, 0) + 1; \
|
||||
if (cmp == 0) { \
|
||||
if (cnt_) *cnt_ = cnt; \
|
||||
return p; \
|
||||
} \
|
||||
if (p->__head.balance != 0) \
|
||||
bq = q, bp = p, top = 0; \
|
||||
stack[top++] = which = (cmp > 0); \
|
||||
path[path_len++] = p; \
|
||||
} \
|
||||
if (cnt_) *cnt_ = cnt; \
|
||||
x->__head.balance = 0, x->__head.size = 1, x->__head.p[0] = x->__head.p[1] = 0, x->__head.s = x; \
|
||||
if (q == 0) *root_ = x; \
|
||||
else q->__head.p[which] = x; \
|
||||
if (bp == 0) return x; \
|
||||
for (i = 0; i < path_len; ++i) ++path[i]->__head.size; \
|
||||
for (i = path_len - 1; i >= 0; --i) { \
|
||||
krmq_update_min_##suf(path[i], path[i]->__head.p[0], path[i]->__head.p[1]); \
|
||||
if (path[i]->__head.s != x) break; \
|
||||
} \
|
||||
for (p = bp, top = 0; p != x; p = p->__head.p[stack[top]], ++top) /* update balance factors */ \
|
||||
if (stack[top] == 0) --p->__head.balance; \
|
||||
else ++p->__head.balance; \
|
||||
if (bp->__head.balance > -2 && bp->__head.balance < 2) return x; /* no re-balance needed */ \
|
||||
/* re-balance */ \
|
||||
which = (bp->__head.balance < 0); \
|
||||
b1 = which == 0? +1 : -1; \
|
||||
q = bp->__head.p[1 - which]; \
|
||||
if (q->__head.balance == b1) { \
|
||||
r = krmq_rotate1_##suf(bp, which); \
|
||||
q->__head.balance = bp->__head.balance = 0; \
|
||||
} else r = krmq_rotate2_##suf(bp, which); \
|
||||
if (bq == 0) *root_ = r; \
|
||||
else bq->__head.p[bp != bq->__head.p[0]] = r; \
|
||||
return x; \
|
||||
}
|
||||
|
||||
#define __KRMQ_ERASE(suf, __scope, __type, __head, __cmp, __lt2) \
|
||||
__scope __type *krmq_erase_##suf(__type **root_, const __type *x, unsigned *cnt_) { \
|
||||
__type *p, *path[KRMQ_MAX_DEPTH], fake; \
|
||||
unsigned char dir[KRMQ_MAX_DEPTH]; \
|
||||
int i, d = 0, cmp; \
|
||||
unsigned cnt = 0; \
|
||||
fake = **root_, fake.__head.p[0] = *root_, fake.__head.p[1] = 0; \
|
||||
if (cnt_) *cnt_ = 0; \
|
||||
if (x) { \
|
||||
for (cmp = -1, p = &fake; cmp; cmp = __cmp(x, p)) { \
|
||||
int which = (cmp > 0); \
|
||||
if (cmp > 0) cnt += krmq_size_child(__head, p, 0) + 1; \
|
||||
dir[d] = which; \
|
||||
path[d++] = p; \
|
||||
p = p->__head.p[which]; \
|
||||
if (p == 0) { \
|
||||
if (cnt_) *cnt_ = 0; \
|
||||
return 0; \
|
||||
} \
|
||||
} \
|
||||
cnt += krmq_size_child(__head, p, 0) + 1; /* because p==x is not counted */ \
|
||||
} else { \
|
||||
for (p = &fake, cnt = 1; p; p = p->__head.p[0]) \
|
||||
dir[d] = 0, path[d++] = p; \
|
||||
p = path[--d]; \
|
||||
} \
|
||||
if (cnt_) *cnt_ = cnt; \
|
||||
for (i = 1; i < d; ++i) --path[i]->__head.size; \
|
||||
if (p->__head.p[1] == 0) { /* ((1,.)2,3)4 => (1,3)4; p=2 */ \
|
||||
path[d-1]->__head.p[dir[d-1]] = p->__head.p[0]; \
|
||||
} else { \
|
||||
__type *q = p->__head.p[1]; \
|
||||
if (q->__head.p[0] == 0) { /* ((1,2)3,4)5 => ((1)2,4)5; p=3,q=2 */ \
|
||||
q->__head.p[0] = p->__head.p[0]; \
|
||||
q->__head.balance = p->__head.balance; \
|
||||
path[d-1]->__head.p[dir[d-1]] = q; \
|
||||
path[d] = q, dir[d++] = 1; \
|
||||
q->__head.size = p->__head.size - 1; \
|
||||
} else { /* ((1,((.,2)3,4)5)6,7)8 => ((1,(2,4)5)3,7)8; p=6 */ \
|
||||
__type *r; \
|
||||
int e = d++; /* backup _d_ */\
|
||||
for (;;) { \
|
||||
dir[d] = 0; \
|
||||
path[d++] = q; \
|
||||
r = q->__head.p[0]; \
|
||||
if (r->__head.p[0] == 0) break; \
|
||||
q = r; \
|
||||
} \
|
||||
r->__head.p[0] = p->__head.p[0]; \
|
||||
q->__head.p[0] = r->__head.p[1]; \
|
||||
r->__head.p[1] = p->__head.p[1]; \
|
||||
r->__head.balance = p->__head.balance; \
|
||||
path[e-1]->__head.p[dir[e-1]] = r; \
|
||||
path[e] = r, dir[e] = 1; \
|
||||
for (i = e + 1; i < d; ++i) --path[i]->__head.size; \
|
||||
r->__head.size = p->__head.size - 1; \
|
||||
} \
|
||||
} \
|
||||
for (i = d - 1; i >= 0; --i) /* not sure why adding condition "path[i]->__head.s==p" doesn't work */ \
|
||||
krmq_update_min_##suf(path[i], path[i]->__head.p[0], path[i]->__head.p[1]); \
|
||||
while (--d > 0) { \
|
||||
__type *q = path[d]; \
|
||||
int which, other, b1 = 1, b2 = 2; \
|
||||
which = dir[d], other = 1 - which; \
|
||||
if (which) b1 = -b1, b2 = -b2; \
|
||||
q->__head.balance += b1; \
|
||||
if (q->__head.balance == b1) break; \
|
||||
else if (q->__head.balance == b2) { \
|
||||
__type *r = q->__head.p[other]; \
|
||||
if (r->__head.balance == -b1) { \
|
||||
path[d-1]->__head.p[dir[d-1]] = krmq_rotate2_##suf(q, which); \
|
||||
} else { \
|
||||
path[d-1]->__head.p[dir[d-1]] = krmq_rotate1_##suf(q, which); \
|
||||
if (r->__head.balance == 0) { \
|
||||
r->__head.balance = -b1; \
|
||||
q->__head.balance = b1; \
|
||||
break; \
|
||||
} else r->__head.balance = q->__head.balance = 0; \
|
||||
} \
|
||||
} \
|
||||
} \
|
||||
*root_ = fake.__head.p[0]; \
|
||||
return p; \
|
||||
}
|
||||
|
||||
#define krmq_free(__type, __head, __root, __free) do { \
|
||||
__type *_p, *_q; \
|
||||
for (_p = __root; _p; _p = _q) { \
|
||||
if (_p->__head.p[0] == 0) { \
|
||||
_q = _p->__head.p[1]; \
|
||||
__free(_p); \
|
||||
} else { \
|
||||
_q = _p->__head.p[0]; \
|
||||
_p->__head.p[0] = _q->__head.p[1]; \
|
||||
_q->__head.p[1] = _p; \
|
||||
} \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
#define __KRMQ_ITR(suf, __scope, __type, __head, __cmp) \
|
||||
struct krmq_itr_##suf { \
|
||||
const __type *stack[KRMQ_MAX_DEPTH], **top; \
|
||||
}; \
|
||||
__scope void krmq_itr_first_##suf(const __type *root, struct krmq_itr_##suf *itr) { \
|
||||
const __type *p; \
|
||||
for (itr->top = itr->stack - 1, p = root; p; p = p->__head.p[0]) \
|
||||
*++itr->top = p; \
|
||||
} \
|
||||
__scope int krmq_itr_find_##suf(const __type *root, const __type *x, struct krmq_itr_##suf *itr) { \
|
||||
const __type *p = root; \
|
||||
itr->top = itr->stack - 1; \
|
||||
while (p != 0) { \
|
||||
int cmp; \
|
||||
*++itr->top = p; \
|
||||
cmp = __cmp(x, p); \
|
||||
if (cmp < 0) p = p->__head.p[0]; \
|
||||
else if (cmp > 0) p = p->__head.p[1]; \
|
||||
else break; \
|
||||
} \
|
||||
return p? 1 : 0; \
|
||||
} \
|
||||
__scope int krmq_itr_next_bidir_##suf(struct krmq_itr_##suf *itr, int dir) { \
|
||||
const __type *p; \
|
||||
if (itr->top < itr->stack) return 0; \
|
||||
dir = !!dir; \
|
||||
p = (*itr->top)->__head.p[dir]; \
|
||||
if (p) { /* go down */ \
|
||||
for (; p; p = p->__head.p[!dir]) \
|
||||
*++itr->top = p; \
|
||||
return 1; \
|
||||
} else { /* go up */ \
|
||||
do { \
|
||||
p = *itr->top--; \
|
||||
} while (itr->top >= itr->stack && p == (*itr->top)->__head.p[dir]); \
|
||||
return itr->top < itr->stack? 0 : 1; \
|
||||
} \
|
||||
} \
|
||||
|
||||
/**
|
||||
* Insert a node to the tree
|
||||
*
|
||||
* @param suf name suffix used in KRMQ_INIT()
|
||||
* @param proot pointer to the root of the tree (in/out: root may change)
|
||||
* @param x node to insert (in)
|
||||
* @param cnt number of nodes smaller than or equal to _x_; can be NULL (out)
|
||||
*
|
||||
* @return _x_ if not present in the tree, or the node equal to x.
|
||||
*/
|
||||
#define krmq_insert(suf, proot, x, cnt) krmq_insert_##suf(proot, x, cnt)
|
||||
|
||||
/**
|
||||
* Find a node in the tree
|
||||
*
|
||||
* @param suf name suffix used in KRMQ_INIT()
|
||||
* @param root root of the tree
|
||||
* @param x node value to find (in)
|
||||
* @param cnt number of nodes smaller than or equal to _x_; can be NULL (out)
|
||||
*
|
||||
* @return node equal to _x_ if present, or NULL if absent
|
||||
*/
|
||||
#define krmq_find(suf, root, x, cnt) krmq_find_##suf(root, x, cnt)
|
||||
#define krmq_interval(suf, root, x, lower, upper) krmq_interval_##suf(root, x, lower, upper)
|
||||
#define krmq_rmq(suf, root, lo, up) krmq_rmq_##suf(root, lo, up)
|
||||
|
||||
/**
|
||||
* Delete a node from the tree
|
||||
*
|
||||
* @param suf name suffix used in KRMQ_INIT()
|
||||
* @param proot pointer to the root of the tree (in/out: root may change)
|
||||
* @param x node value to delete; if NULL, delete the first node (in)
|
||||
*
|
||||
* @return node removed from the tree if present, or NULL if absent
|
||||
*/
|
||||
#define krmq_erase(suf, proot, x, cnt) krmq_erase_##suf(proot, x, cnt)
|
||||
#define krmq_erase_first(suf, proot) krmq_erase_##suf(proot, 0, 0)
|
||||
|
||||
#define krmq_itr_t(suf) struct krmq_itr_##suf
|
||||
|
||||
/**
|
||||
* Place the iterator at the smallest object
|
||||
*
|
||||
* @param suf name suffix used in KRMQ_INIT()
|
||||
* @param root root of the tree
|
||||
* @param itr iterator
|
||||
*/
|
||||
#define krmq_itr_first(suf, root, itr) krmq_itr_first_##suf(root, itr)
|
||||
|
||||
/**
|
||||
* Place the iterator at the object equal to or greater than the query
|
||||
*
|
||||
* @param suf name suffix used in KRMQ_INIT()
|
||||
* @param root root of the tree
|
||||
* @param x query (in)
|
||||
* @param itr iterator (out)
|
||||
*
|
||||
* @return 1 if find; 0 otherwise. krmq_at(itr) is NULL if and only if query is
|
||||
* larger than all objects in the tree
|
||||
*/
|
||||
#define krmq_itr_find(suf, root, x, itr) krmq_itr_find_##suf(root, x, itr)
|
||||
|
||||
/**
|
||||
* Move to the next object in order
|
||||
*
|
||||
* @param itr iterator (modified)
|
||||
*
|
||||
* @return 1 if there is a next object; 0 otherwise
|
||||
*/
|
||||
#define krmq_itr_next(suf, itr) krmq_itr_next_bidir_##suf(itr, 1)
|
||||
#define krmq_itr_prev(suf, itr) krmq_itr_next_bidir_##suf(itr, 0)
|
||||
|
||||
/**
|
||||
* Return the pointer at the iterator
|
||||
*
|
||||
* @param itr iterator
|
||||
*
|
||||
* @return pointer if present; NULL otherwise
|
||||
*/
|
||||
#define krmq_at(itr) ((itr)->top < (itr)->stack? 0 : *(itr)->top)
|
||||
|
||||
#define KRMQ_INIT2(suf, __scope, __type, __head, __cmp, __lt2) \
|
||||
__KRMQ_FIND(suf, __scope, __type, __head, __cmp) \
|
||||
__KRMQ_RMQ(suf, __scope, __type, __head, __cmp, __lt2) \
|
||||
__KRMQ_ROTATE(suf, __type, __head, __lt2) \
|
||||
__KRMQ_INSERT(suf, __scope, __type, __head, __cmp, __lt2) \
|
||||
__KRMQ_ERASE(suf, __scope, __type, __head, __cmp, __lt2) \
|
||||
__KRMQ_ITR(suf, __scope, __type, __head, __cmp)
|
||||
|
||||
#define KRMQ_INIT(suf, __type, __head, __cmp, __lt2) \
|
||||
KRMQ_INIT2(suf,, __type, __head, __cmp, __lt2)
|
||||
|
||||
#endif
|
||||
@@ -89,7 +89,7 @@
|
||||
#ifndef KSTRING_T
|
||||
#define KSTRING_T kstring_t
|
||||
typedef struct __kstring_t {
|
||||
unsigned l, m;
|
||||
size_t l, m;
|
||||
char *s;
|
||||
} kstring_t;
|
||||
#endif
|
||||
|
||||
@@ -16,6 +16,13 @@
|
||||
#define KSW_EZ_SPLICE_REV 0x200
|
||||
#define KSW_EZ_SPLICE_FLANK 0x400
|
||||
|
||||
// The subset of CIGAR operators used by ksw code.
|
||||
// Use MM_CIGAR_* from minimap.h if you need the full list.
|
||||
#define KSW_CIGAR_MATCH 0
|
||||
#define KSW_CIGAR_INS 1
|
||||
#define KSW_CIGAR_DEL 2
|
||||
#define KSW_CIGAR_N_SKIP 3
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
@@ -137,13 +144,13 @@ static inline void ksw_backtrack(void *km, int is_rot, int is_rev, int min_intro
|
||||
else if (!(tmp >> (state + 2) & 1)) state = 0; // if requesting other states, _state_ stays the same if it is a continuation; otherwise, set to H
|
||||
if (state == 0) state = tmp & 7; // TODO: probably this line can be merged into the "else if" line right above; not 100% sure
|
||||
if (force_state >= 0) state = force_state;
|
||||
if (state == 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 0, 1), --i, --j; // match
|
||||
else if (state == 1 || (state == 3 && min_intron_len <= 0)) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 2, 1), --i; // deletion
|
||||
else if (state == 3 && min_intron_len > 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 3, 1), --i; // intron
|
||||
else cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, 1), --j; // insertion
|
||||
if (state == 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_MATCH, 1), --i, --j;
|
||||
else if (state == 1 || (state == 3 && min_intron_len <= 0)) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_DEL, 1), --i;
|
||||
else if (state == 3 && min_intron_len > 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_N_SKIP, 1), --i;
|
||||
else cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_INS, 1), --j;
|
||||
}
|
||||
if (i >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, min_intron_len > 0 && i >= min_intron_len? 3 : 2, i + 1); // first deletion
|
||||
if (j >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, j + 1); // first insertion
|
||||
if (i >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, min_intron_len > 0 && i >= min_intron_len? KSW_CIGAR_N_SKIP : KSW_CIGAR_DEL, i + 1); // first deletion
|
||||
if (j >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, KSW_CIGAR_INS, j + 1); // first insertion
|
||||
if (!is_rev)
|
||||
for (i = 0; i < n_cigar>>1; ++i) // reverse CIGAR
|
||||
tmp = cigar[i], cigar[i] = cigar[n_cigar-1-i], cigar[n_cigar-1-i] = tmp;
|
||||
|
||||
+2319
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,42 @@
|
||||
/* The MIT License
|
||||
|
||||
Copyright (c) 2018- Dana-Farber Cancer Institute
|
||||
2017-2018 Broad Institute, Inc.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
Modified Copyright (C) 2021 Intel Corporation
|
||||
Contacts: Saurabh Kalikar <saurabh.kalikar@intel.com>;
|
||||
Vasimuddin Md <vasimuddin.md@intel.com>; Sanchit Misra <sanchit.misra@intel.com>;
|
||||
Chirag Jain <chirag@iisc.ac.in>; Heng Li <hli@jimmy.harvard.edu>
|
||||
*/
|
||||
#include <string.h>
|
||||
#include <stdio.h>
|
||||
#include <assert.h>
|
||||
#include "ksw2.h"
|
||||
#include <immintrin.h>
|
||||
#include <x86intrin.h>
|
||||
#include <smmintrin.h>
|
||||
#include <emmintrin.h>
|
||||
void ksw_extd2_avx512(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int8_t q2, int8_t e2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||
|
||||
void ksw_extd2_avx2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int8_t q2, int8_t e2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||
@@ -4,15 +4,23 @@
|
||||
#include "ksw2.h"
|
||||
|
||||
#ifdef __SSE2__
|
||||
#ifdef USE_SIMDE
|
||||
#include <simde/x86/sse2.h>
|
||||
#else
|
||||
#include <emmintrin.h>
|
||||
#endif
|
||||
|
||||
#ifdef KSW_SSE2_ONLY
|
||||
#undef __SSE4_1__
|
||||
#endif
|
||||
|
||||
#ifdef __SSE4_1__
|
||||
#ifdef USE_SIMDE
|
||||
#include <simde/x86/sse4.1.h>
|
||||
#else
|
||||
#include <smmintrin.h>
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef KSW_CPU_DISPATCH
|
||||
#ifdef __SSE4_1__
|
||||
|
||||
+8
-1
@@ -4,15 +4,22 @@
|
||||
#include "ksw2.h"
|
||||
|
||||
#ifdef __SSE2__
|
||||
#ifdef USE_SIMDE
|
||||
#include <simde/x86/sse2.h>
|
||||
#else
|
||||
#include <emmintrin.h>
|
||||
|
||||
#endif
|
||||
#ifdef KSW_SSE2_ONLY
|
||||
#undef __SSE4_1__
|
||||
#endif
|
||||
|
||||
#ifdef __SSE4_1__
|
||||
#ifdef USE_SIMDE
|
||||
#include <simde/x86/sse4.1.h>
|
||||
#else
|
||||
#include <smmintrin.h>
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef KSW_CPU_DISPATCH
|
||||
#ifdef __SSE4_1__
|
||||
|
||||
@@ -3,15 +3,23 @@
|
||||
#include "ksw2.h"
|
||||
|
||||
#ifdef __SSE2__
|
||||
#ifdef USE_SIMDE
|
||||
#include <simde/x86/sse2.h>
|
||||
#else
|
||||
#include <emmintrin.h>
|
||||
#endif
|
||||
|
||||
#ifdef KSW_SSE2_ONLY
|
||||
#undef __SSE4_1__
|
||||
#endif
|
||||
|
||||
#ifdef __SSE4_1__
|
||||
#ifdef USE_SIMDE
|
||||
#include <simde/x86/sse4.1.h>
|
||||
#else
|
||||
#include <smmintrin.h>
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#ifdef KSW_CPU_DISPATCH
|
||||
#ifdef __SSE4_1__
|
||||
|
||||
+6
-1
@@ -1,9 +1,14 @@
|
||||
#include <stdlib.h>
|
||||
#include <stdint.h>
|
||||
#include <string.h>
|
||||
#include <emmintrin.h>
|
||||
#include "ksw2.h"
|
||||
|
||||
#ifdef USE_SIMDE
|
||||
#include <simde/x86/sse2.h>
|
||||
#else
|
||||
#include <emmintrin.h>
|
||||
#endif
|
||||
|
||||
#ifdef __GNUC__
|
||||
#define LIKELY(x) __builtin_expect((x),1)
|
||||
#define UNLIKELY(x) __builtin_expect((x),0)
|
||||
|
||||
@@ -0,0 +1,521 @@
|
||||
#include <stdint.h>
|
||||
#include <string.h>
|
||||
#include <stdio.h>
|
||||
#include <assert.h>
|
||||
#include "mmpriv.h"
|
||||
#include "kalloc.h"
|
||||
#include "krmq.h"
|
||||
#include <x86intrin.h>
|
||||
//#include "simd_chain.h"
|
||||
//#include "parallel_chaining_32_bit.h"
|
||||
#include "parallel_chaining_v2_22.h"
|
||||
|
||||
#ifdef MANUAL_PROFILING
|
||||
extern uint64_t dp_time, rmq_time, rmq_t1, rmq_t2, rmq_t3, rmq_t4;
|
||||
#endif
|
||||
|
||||
extern bool enable_vect_dp_chaining;
|
||||
uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_t *p, int32_t *v, int32_t *t, int32_t min_cnt, int32_t min_sc, int32_t *n_u_, int32_t *n_v_)
|
||||
{
|
||||
mm128_t *z;
|
||||
uint64_t *u;
|
||||
int64_t i, k, n_z, n_v;
|
||||
int32_t n_u;
|
||||
|
||||
*n_u_ = *n_v_ = 0;
|
||||
for (i = 0, n_z = 0; i < n; ++i) // precompute n_z
|
||||
if (f[i] >= min_sc) ++n_z;
|
||||
if (n_z == 0) return 0;
|
||||
KMALLOC(km, z, n_z);
|
||||
for (i = 0, k = 0; i < n; ++i) // populate z[]
|
||||
if (f[i] >= min_sc) z[k].x = f[i], z[k++].y = i;
|
||||
radix_sort_128x(z, z + n_z);
|
||||
|
||||
memset(t, 0, n * 4);
|
||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // precompute n_u
|
||||
int64_t n_v0 = n_v;
|
||||
int32_t sc;
|
||||
for (i = z[k].y; i >= 0 && t[i] == 0; i = p[i])
|
||||
++n_v, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
++n_u;
|
||||
else n_v = n_v0;
|
||||
}
|
||||
KMALLOC(km, u, n_u);
|
||||
memset(t, 0, n * 4);
|
||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // populate u[]
|
||||
int64_t n_v0 = n_v;
|
||||
int32_t sc;
|
||||
for (i = z[k].y; i >= 0 && t[i] == 0; i = p[i])
|
||||
v[n_v++] = i, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
u[n_u++] = (uint64_t)sc << 32 | (n_v - n_v0);
|
||||
else n_v = n_v0;
|
||||
}
|
||||
kfree(km, z);
|
||||
assert(n_v < INT32_MAX);
|
||||
*n_u_ = n_u, *n_v_ = n_v;
|
||||
return u;
|
||||
}
|
||||
|
||||
static mm128_t *compact_a(void *km, int32_t n_u, uint64_t *u, int32_t n_v, int32_t *v, mm128_t *a)
|
||||
{
|
||||
mm128_t *b, *w;
|
||||
uint64_t *u2;
|
||||
int64_t i, j, k;
|
||||
|
||||
// write the result to b[]
|
||||
KMALLOC(km, b, n_v);
|
||||
for (i = 0, k = 0; i < n_u; ++i) {
|
||||
int32_t k0 = k, ni = (int32_t)u[i];
|
||||
for (j = 0; j < ni; ++j)
|
||||
b[k++] = a[v[k0 + (ni - j - 1)]];
|
||||
}
|
||||
kfree(km, v);
|
||||
|
||||
// sort u[] and a[] by the target position, such that adjacent chains may be joined
|
||||
KMALLOC(km, w, n_u);
|
||||
for (i = k = 0; i < n_u; ++i) {
|
||||
w[i].x = b[k].x, w[i].y = (uint64_t)k<<32|i;
|
||||
k += (int32_t)u[i];
|
||||
}
|
||||
radix_sort_128x(w, w + n_u);
|
||||
KMALLOC(km, u2, n_u);
|
||||
for (i = k = 0; i < n_u; ++i) {
|
||||
int32_t j = (int32_t)w[i].y, n = (int32_t)u[j];
|
||||
u2[i] = u[j];
|
||||
memcpy(&a[k], &b[w[i].y>>32], n * sizeof(mm128_t));
|
||||
k += n;
|
||||
}
|
||||
memcpy(u, u2, n_u * 8);
|
||||
memcpy(b, a, k * sizeof(mm128_t)); // write _a_ to _b_ and deallocate _a_ because _a_ is oversized, sometimes a lot
|
||||
kfree(km, a); kfree(km, w); kfree(km, u2);
|
||||
return b;
|
||||
}
|
||||
|
||||
static inline int32_t comput_sc(const mm128_t *ai, const mm128_t *aj, int32_t max_dist_x, int32_t max_dist_y, int32_t bw, float chn_pen_gap, float chn_pen_skip, int is_cdna, int n_seg)
|
||||
{
|
||||
|
||||
uint64_t ai_x, ai_y, aj_x, aj_y;
|
||||
ai_x = ai->x; ai_y = ai->y; aj_x = aj->x; aj_y = aj->y;
|
||||
|
||||
#ifdef CHAIN_DEBUG
|
||||
int32_t sc_vect = obj.comput_sc_vectorized_avx2_caller(ai_x, ai_y, aj_x, aj_y, aj->y>>32&0xff);
|
||||
#endif
|
||||
|
||||
//if (sc_vect == 0) return INT32_MIN;
|
||||
//else
|
||||
//return sc_vect;
|
||||
|
||||
//fprintf(stderr, "%lld %lld %lld %lld \n", ai_x, ai_y, aj_x, aj_y);
|
||||
//fprintf(stderr, "%lld %lld %lld %f %f %d %d\n", max_dist_x, max_dist_y, bw, chn_pen_gap, chn_pen_skip, is_cdna, n_seg);
|
||||
int32_t dq = (int32_t)ai_y - (int32_t)aj_y, dr, dd, dg, q_span, sc;
|
||||
int32_t sidi = (ai_y & MM_SEED_SEG_MASK) >> MM_SEED_SEG_SHIFT;
|
||||
int32_t sidj = (aj_y & MM_SEED_SEG_MASK) >> MM_SEED_SEG_SHIFT;
|
||||
if (dq <= 0 || dq > max_dist_x) {
|
||||
|
||||
#ifdef CHAIN_DEBUG
|
||||
if(INT32_MIN != sc_vect){
|
||||
//fprintf(stderr, "score mismatch %d -- %d", sc , sc_vect);
|
||||
fprintf(stderr, "int-min exit: %llu, %llu, %llu, %llu : %d -- %d\n", ai_x, ai_y, aj_x, aj_y, sc, sc_vect);
|
||||
}
|
||||
#endif
|
||||
return INT32_MIN;
|
||||
}
|
||||
dr = (int32_t)(ai_x - aj_x);
|
||||
if (sidi == sidj && (dr == 0 || dq > max_dist_y)) {
|
||||
|
||||
#ifdef CHAIN_DEBUG
|
||||
if(INT32_MIN != sc_vect){
|
||||
//fprintf(stderr, "score mismatch %d -- %d", sc , sc_vect);
|
||||
fprintf(stderr, "int-min exit: %llu, %llu, %llu, %llu : %d -- %d\n", ai_x, ai_y, aj_x, aj_y, sc, sc_vect);
|
||||
}
|
||||
#endif
|
||||
return INT32_MIN;
|
||||
}
|
||||
dd = dr > dq? dr - dq : dq - dr;
|
||||
if (sidi == sidj && dd > bw) {
|
||||
|
||||
#ifdef CHAIN_DEBUG
|
||||
if(INT32_MIN != sc_vect){
|
||||
//fprintf(stderr, "score mismatch %d -- %d", sc , sc_vect);
|
||||
fprintf(stderr, "int-min exit: %llu, %llu, %llu, %llu : %d -- %d\n", ai_x, ai_y, aj_x, aj_y, sc, sc_vect);
|
||||
}
|
||||
#endif
|
||||
return INT32_MIN;
|
||||
}
|
||||
if (n_seg > 1 && !is_cdna && sidi == sidj && dr > max_dist_y) {
|
||||
|
||||
#ifdef CHAIN_DEBUG
|
||||
if(INT32_MIN != sc_vect){
|
||||
//fprintf(stderr, "score mismatch %d -- %d", sc , sc_vect);
|
||||
fprintf(stderr, "int-min exit: %llu, %llu, %llu, %llu : %d -- %d\n", ai_x, ai_y, aj_x, aj_y, sc, sc_vect);
|
||||
}
|
||||
#endif
|
||||
return INT32_MIN;
|
||||
}
|
||||
dg = dr < dq? dr : dq;
|
||||
q_span = aj->y>>32&0xff;
|
||||
sc = q_span < dg? q_span : dg;
|
||||
if (dd || dg > q_span) {
|
||||
float lin_pen, log_pen;
|
||||
lin_pen = chn_pen_gap * (float)dd + chn_pen_skip * (float)dg;
|
||||
log_pen = dd >= 1? mg_log2(dd + 1) : 0.0f; // mg_log2() only works for dd>=2
|
||||
if (is_cdna || sidi != sidj) {
|
||||
if (sidi != sidj && dr == 0) ++sc; // possibly due to overlapping paired ends; give a minor bonus
|
||||
else if (dr > dq || sidi != sidj) sc -= (int)(lin_pen < log_pen? lin_pen : log_pen); // deletion or jump between paired ends
|
||||
else sc -= (int)(lin_pen + .5f * log_pen);
|
||||
} else sc -= (int)(lin_pen + .5f * log_pen);
|
||||
}
|
||||
#ifdef CHAIN_DEBUG
|
||||
|
||||
if(sc != sc_vect ){
|
||||
//fprintf(stderr, "score mismatch %d -- %d", sc , sc_vect);
|
||||
fprintf(stderr, "outer: %llu, %llu, %llu, %llu : %d -- %d\n", ai_x, ai_y, aj_x, aj_y, sc, sc_vect);
|
||||
}
|
||||
#endif
|
||||
return sc;
|
||||
}
|
||||
|
||||
/* Input:
|
||||
* a[].x: tid<<33 | rev<<32 | tpos
|
||||
* a[].y: flags<<40 | q_span<<32 | q_pos
|
||||
* Output:
|
||||
* n_u: #chains
|
||||
* u[]: score<<32 | #anchors (sum of lower 32 bits of u[] is the returned length of a[])
|
||||
* input a[] is deallocated on return
|
||||
*/
|
||||
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
int is_cdna, int n_seg, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||
{ // TODO: make sure this works when n has more than 32 bits
|
||||
///fprintf(stderr, "chaining called\n");
|
||||
|
||||
|
||||
|
||||
#ifdef MANUAL_PROFILING
|
||||
uint64_t align_start = __rdtsc();
|
||||
#endif
|
||||
|
||||
int32_t *f, *t, *v, *v_1, *p_1, n_u, n_v, mmax_f = 0;
|
||||
int64_t *p, i, j, max_ii, st = 0, n_iter = 0;
|
||||
uint64_t *u;
|
||||
uint32_t* f_1;
|
||||
if (_u) *_u = 0, *n_u_ = 0;
|
||||
if (n == 0 || a == 0) {
|
||||
kfree(km, a);
|
||||
return 0;
|
||||
}
|
||||
if (max_dist_x < bw) max_dist_x = bw;
|
||||
if (max_dist_y < bw && !is_cdna) max_dist_y = bw;
|
||||
KMALLOC(km, p, n);
|
||||
KMALLOC(km, p_1, n);
|
||||
KMALLOC(km, f, n);
|
||||
KMALLOC(km, f_1, n);
|
||||
KMALLOC(km, v, n);
|
||||
KMALLOC(km, v_1, n);
|
||||
KCALLOC(km, t, n);
|
||||
|
||||
//#ifdef PARALLEL_CHAINING
|
||||
if(enable_vect_dp_chaining){
|
||||
// Parallel chaining data-structures
|
||||
anchor_t* anchors = (anchor_t*)malloc(n* sizeof(anchor_t));
|
||||
for (i = 0; i < n; ++i) {
|
||||
uint64_t ri = a[i].x;
|
||||
int32_t qi = (int32_t)a[i].y, q_span = a[i].y>>32&0xff; // NB: only 8 bits of span is used!!!
|
||||
anchors[i].r = ri;
|
||||
anchors[i].q = qi;
|
||||
anchors[i].l = q_span;
|
||||
}
|
||||
num_bits_t *anchor_r, *anchor_q, *anchor_l;
|
||||
create_SoA_Anchors_32_bit(anchors, n, anchor_r, anchor_q, anchor_l);
|
||||
dp_chain obj(max_dist_x, max_dist_y, bw, max_skip, max_iter, min_cnt, min_sc, chn_pen_gap, chn_pen_skip, is_cdna, n_seg);
|
||||
|
||||
#ifdef PARALLEL_CHAINING
|
||||
obj.mm_dp_vectorized(n, &anchors[0], anchor_r, anchor_q, anchor_l, f_1, p_1, v_1, max_dist_x, max_dist_y, NULL, NULL);
|
||||
#endif
|
||||
// -16 is due to extra padding at the start of arrays
|
||||
anchor_r -= 16; anchor_q -= 16; anchor_l -= 16;
|
||||
free(anchor_r);
|
||||
free(anchor_q);
|
||||
free(anchor_l);
|
||||
free(anchors);
|
||||
for(int i = 0; i < n; i++){
|
||||
#if 1
|
||||
f[i] = f_1[i];
|
||||
p[i] = p_1[i];
|
||||
v[i] = v_1[i];
|
||||
#endif
|
||||
}
|
||||
|
||||
//
|
||||
} else {
|
||||
//#else
|
||||
|
||||
// fill the score and backtrack arrays
|
||||
for (i = 0, max_ii = -1; i < n; ++i) {
|
||||
int64_t max_j = -1, end_j;
|
||||
int32_t max_f = a[i].y>>32&0xff, n_skip = 0;
|
||||
while (st < i && (a[i].x>>32 != a[st].x>>32 || a[i].x > a[st].x + max_dist_x)) ++st;
|
||||
if (i - st > max_iter) st = i - max_iter;
|
||||
int my_cnt = 0;
|
||||
for (j = i - 1; j >= st; --j) {
|
||||
int32_t sc;
|
||||
sc = comput_sc(&a[i], &a[j], max_dist_x, max_dist_y, bw, chn_pen_gap, chn_pen_skip, is_cdna, n_seg);
|
||||
++n_iter;
|
||||
if (sc == INT32_MIN) continue;
|
||||
sc += f[j];
|
||||
if (sc > max_f) {
|
||||
max_f = sc, max_j = j;
|
||||
if (n_skip > 0) --n_skip;
|
||||
} else if (t[j] == (int32_t)i) {
|
||||
if (++n_skip > max_skip)
|
||||
break;
|
||||
}
|
||||
if (p[j] >= 0) t[p[j]] = i;
|
||||
}
|
||||
end_j = j;
|
||||
int debug_iter = 2057329;
|
||||
|
||||
if (max_ii < 0 || a[i].x - a[max_ii].x > (int64_t)max_dist_x) {
|
||||
int32_t max = INT32_MIN;
|
||||
max_ii = -1;
|
||||
for (j = i - 1; j >= st; --j) {
|
||||
if (max < (int32_t)f[j]) max = f[j], max_ii = j;
|
||||
}
|
||||
}
|
||||
if (max_ii >= 0 && max_ii < end_j) {
|
||||
int32_t tmp;
|
||||
tmp = comput_sc(&a[i], &a[max_ii], max_dist_x, max_dist_y, bw, chn_pen_gap, chn_pen_skip, is_cdna, n_seg);
|
||||
// if (i == debug_iter) fprintf(stderr, "mm2: endj: %d max_ii: %d max_f: %d tmp_score: %d \n", end_j, max_ii, max_f, tmp);
|
||||
|
||||
|
||||
if (tmp != INT32_MIN && max_f < tmp + f[max_ii]){
|
||||
max_f = tmp + f[max_ii], max_j = max_ii;
|
||||
}
|
||||
}
|
||||
f[i] = max_f, p[i] = max_j;
|
||||
v[i] = max_j >= 0 && v[max_j] > max_f? v[max_j] : max_f; // v[] keeps the peak score up to i; f[] is the score ending at i, not always the peak
|
||||
|
||||
if (max_ii < 0 || (a[i].x - a[max_ii].x <= (int64_t)max_dist_x && f[max_ii] < f[i]))
|
||||
max_ii = i;
|
||||
if (mmax_f < max_f) mmax_f = max_f;
|
||||
}
|
||||
|
||||
}
|
||||
//#endif
|
||||
|
||||
#ifdef CHAIN_DEBUG
|
||||
|
||||
for(int i = 0; i < n; i++){
|
||||
if(f[i] != f_1[i] || p[i] != p_1[i] || v[i] !=v_1[i])
|
||||
{
|
||||
fprintf(stderr, "i:%d %d %d %d %d %d %d\n",i, f[i], f_1[i], p[i], p_1[i], v[i], v_1[i] );
|
||||
}
|
||||
#if 0
|
||||
f[i] = f_1[i];
|
||||
p[i] = p_1[i];
|
||||
v[i] = v_1[i];
|
||||
#endif
|
||||
}
|
||||
#endif
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, &n_u, &n_v);
|
||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||
kfree(km, p); kfree(km, p_1); kfree(km, f); kfree(km, f_1); kfree(km, t); kfree(km, v_1);
|
||||
if (n_u == 0) {
|
||||
kfree(km, a); kfree(km, v);
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
#ifdef MANUAL_PROFILING
|
||||
dp_time += __rdtsc() - align_start;
|
||||
#endif
|
||||
return compact_a(km, n_u, u, n_v, v, a);
|
||||
}
|
||||
|
||||
typedef struct lc_elem_s {
|
||||
int32_t y;
|
||||
int64_t i;
|
||||
double pri;
|
||||
KRMQ_HEAD(struct lc_elem_s) head;
|
||||
} lc_elem_t;
|
||||
|
||||
#define lc_elem_cmp(a, b) ((a)->y < (b)->y? -1 : (a)->y > (b)->y? 1 : ((a)->i > (b)->i) - ((a)->i < (b)->i))
|
||||
#define lc_elem_lt2(a, b) ((a)->pri < (b)->pri)
|
||||
KRMQ_INIT(lc_elem, lc_elem_t, head, lc_elem_cmp, lc_elem_lt2)
|
||||
|
||||
KALLOC_POOL_INIT(rmq, lc_elem_t)
|
||||
|
||||
static inline int32_t comput_sc_simple(const mm128_t *ai, const mm128_t *aj, float chn_pen_gap, float chn_pen_skip, int32_t *exact, int32_t *width)
|
||||
{
|
||||
int32_t dq = (int32_t)ai->y - (int32_t)aj->y, dr, dd, dg, q_span, sc;
|
||||
dr = (int32_t)(ai->x - aj->x);
|
||||
*width = dd = dr > dq? dr - dq : dq - dr;
|
||||
dg = dr < dq? dr : dq;
|
||||
q_span = aj->y>>32&0xff;
|
||||
sc = q_span < dg? q_span : dg;
|
||||
if (exact) *exact = (dd == 0 && dg <= q_span);
|
||||
if (dd || dq > q_span) {
|
||||
float lin_pen, log_pen;
|
||||
lin_pen = chn_pen_gap * (float)dd + chn_pen_skip * (float)dg;
|
||||
log_pen = dd >= 1? mg_log2(dd + 1) : 0.0f; // mg_log2() only works for dd>=2
|
||||
sc -= (int)(lin_pen + .5f * log_pen);
|
||||
}
|
||||
return sc;
|
||||
}
|
||||
|
||||
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||
{
|
||||
#ifdef MANUAL_PROFILING
|
||||
uint64_t start = __rdtsc();
|
||||
#endif
|
||||
uint64_t tim;
|
||||
//fprintf(stderr, "rmq call \n");
|
||||
int32_t *f,*t, *v, n_u, n_v, mmax_f = 0, max_rmq_size = 0;
|
||||
int64_t *p, i, i0, st = 0, st_inner = 0, n_iter = 0;
|
||||
uint64_t *u;
|
||||
lc_elem_t *root = 0, *root_inner = 0;
|
||||
void *mem_mp = 0;
|
||||
kmp_rmq_t *mp;
|
||||
|
||||
if (_u) *_u = 0, *n_u_ = 0;
|
||||
if (n == 0 || a == 0) {
|
||||
kfree(km, a);
|
||||
return 0;
|
||||
}
|
||||
if (max_dist < bw) max_dist = bw;
|
||||
if (max_dist_inner <= 0 || max_dist_inner >= max_dist) max_dist_inner = 0;
|
||||
KMALLOC(km, p, n);
|
||||
KMALLOC(km, f, n);
|
||||
KCALLOC(km, t, n);
|
||||
KMALLOC(km, v, n);
|
||||
mem_mp = km_init2(km, 0x10000);
|
||||
mp = kmp_init_rmq(mem_mp);
|
||||
|
||||
// fill the score and backtrack arrays
|
||||
for (i = i0 = 0; i < n; ++i) {
|
||||
int64_t max_j = -1;
|
||||
int32_t q_span = a[i].y>>32&0xff, max_f = q_span;
|
||||
lc_elem_t s, *q, *r, lo, hi;
|
||||
// add in-range anchors
|
||||
#ifdef MANUAL_PROFILING_RMQ
|
||||
tim = __rdtsc();
|
||||
#endif
|
||||
if (i0 < i && a[i0].x != a[i].x) {
|
||||
int64_t j;
|
||||
for (j = i0; j < i; ++j) {
|
||||
q = kmp_alloc_rmq(mp);
|
||||
q->y = (int32_t)a[j].y, q->i = j, q->pri = -(f[j] + 0.5 * chn_pen_gap * ((int32_t)a[j].x + (int32_t)a[j].y));
|
||||
krmq_insert(lc_elem, &root, q, 0);
|
||||
if (max_dist_inner > 0) {
|
||||
r = kmp_alloc_rmq(mp);
|
||||
*r = *q;
|
||||
krmq_insert(lc_elem, &root_inner, r, 0);
|
||||
}
|
||||
}
|
||||
i0 = i;
|
||||
}
|
||||
#ifdef MANUAL_PROFILING_RMQ
|
||||
rmq_t1 += __rdtsc() - tim;
|
||||
#endif
|
||||
// get rid of active chains out of range
|
||||
#ifdef MANUAL_PROFILING_RMQ
|
||||
tim = __rdtsc();
|
||||
#endif
|
||||
while (st < i && (a[i].x>>32 != a[st].x>>32 || a[i].x > a[st].x + max_dist || krmq_size(head, root) > cap_rmq_size)) {
|
||||
s.y = (int32_t)a[st].y, s.i = st;
|
||||
if ((q = krmq_find(lc_elem, root, &s, 0)) != 0) {
|
||||
q = krmq_erase(lc_elem, &root, q, 0);
|
||||
kmp_free_rmq(mp, q);
|
||||
}
|
||||
++st;
|
||||
}
|
||||
#ifdef MANUAL_PROFILING_RMQ
|
||||
rmq_t2 += __rdtsc() - tim;
|
||||
#endif
|
||||
#ifdef MANUAL_PROFILING_RMQ
|
||||
tim = __rdtsc();
|
||||
#endif
|
||||
if (max_dist_inner > 0) { // similar to the block above, but applied to the inner tree
|
||||
while (st_inner < i && (a[i].x>>32 != a[st_inner].x>>32 || a[i].x > a[st_inner].x + max_dist_inner || krmq_size(head, root_inner) > cap_rmq_size)) {
|
||||
s.y = (int32_t)a[st_inner].y, s.i = st_inner;
|
||||
if ((q = krmq_find(lc_elem, root_inner, &s, 0)) != 0) {
|
||||
q = krmq_erase(lc_elem, &root_inner, q, 0);
|
||||
kmp_free_rmq(mp, q);
|
||||
}
|
||||
++st_inner;
|
||||
}
|
||||
}
|
||||
#ifdef MANUAL_PROFILING_RMQ
|
||||
rmq_t3 += __rdtsc() - tim;
|
||||
#endif
|
||||
// RMQ
|
||||
lo.i = INT32_MAX, lo.y = (int32_t)a[i].y - max_dist;
|
||||
hi.i = 0, hi.y = (int32_t)a[i].y;
|
||||
if ((q = krmq_rmq(lc_elem, root, &lo, &hi)) != 0) {
|
||||
int32_t sc, exact, width, n_skip = 0;
|
||||
int64_t j = q->i;
|
||||
assert(q->y >= lo.y && q->y <= hi.y);
|
||||
sc = f[j] + comput_sc_simple(&a[i], &a[j], chn_pen_gap, chn_pen_skip, &exact, &width);
|
||||
if (width <= bw && sc > max_f) max_f = sc, max_j = j;
|
||||
if (!exact && root_inner && (int32_t)a[i].y > 0) {
|
||||
lc_elem_t *lo, *hi;
|
||||
s.y = (int32_t)a[i].y - 1, s.i = n;
|
||||
krmq_interval(lc_elem, root_inner, &s, &lo, &hi);
|
||||
if (lo) {
|
||||
const lc_elem_t *q;
|
||||
int32_t width, n_rmq_iter = 0;
|
||||
krmq_itr_t(lc_elem) itr;
|
||||
krmq_itr_find(lc_elem, root_inner, lo, &itr);
|
||||
while ((q = krmq_at(&itr)) != 0) {
|
||||
#ifdef MANUAL_PROFILING_RMQ
|
||||
tim = __rdtsc();
|
||||
#endif
|
||||
if (q->y < (int32_t)a[i].y - max_dist_inner) break;
|
||||
++n_rmq_iter;
|
||||
j = q->i;
|
||||
sc = f[j] + comput_sc_simple(&a[i], &a[j], chn_pen_gap, chn_pen_skip, 0, &width);
|
||||
if (width <= bw) {
|
||||
if (sc > max_f) {
|
||||
max_f = sc, max_j = j;
|
||||
if (n_skip > 0) --n_skip;
|
||||
} else if (t[j] == (int32_t)i) {
|
||||
if (++n_skip > max_chn_skip)
|
||||
break;
|
||||
}
|
||||
if (p[j] >= 0) t[p[j]] = i;
|
||||
}
|
||||
if (!krmq_itr_prev(lc_elem, &itr)) break;
|
||||
#ifdef MANUAL_PROFILING_RMQ
|
||||
rmq_t4 += __rdtsc() - tim;
|
||||
#endif
|
||||
}
|
||||
n_iter += n_rmq_iter;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// set max
|
||||
assert(max_j < 0 || (a[max_j].x < a[i].x && (int32_t)a[max_j].y < (int32_t)a[i].y));
|
||||
f[i] = max_f, p[i] = max_j;
|
||||
v[i] = max_j >= 0 && v[max_j] > max_f? v[max_j] : max_f; // v[] keeps the peak score up to i; f[] is the score ending at i, not always the peak
|
||||
if (mmax_f < max_f) mmax_f = max_f;
|
||||
if (max_rmq_size < krmq_size(head, root)) max_rmq_size = krmq_size(head, root);
|
||||
}
|
||||
km_destroy(mem_mp);
|
||||
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, &n_u, &n_v);
|
||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||
kfree(km, p); kfree(km, f); kfree(km, t);
|
||||
if (n_u == 0) {
|
||||
kfree(km, a); kfree(km, v);
|
||||
return 0;
|
||||
}
|
||||
#ifdef MANUAL_PROFILING
|
||||
rmq_time += __rdtsc() - start;
|
||||
#endif
|
||||
return compact_a(km, n_u, u, n_v, v, a);
|
||||
}
|
||||
Submodule
+1
Submodule lib/simde added at b30129b3b4
@@ -6,8 +6,77 @@
|
||||
#include "minimap.h"
|
||||
#include "mmpriv.h"
|
||||
#include "ketopt.h"
|
||||
#include <x86intrin.h>
|
||||
#include <immintrin.h>
|
||||
#include <sys/time.h>
|
||||
#include <stdlib.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <string>
|
||||
#include <map>
|
||||
#include <errno.h>
|
||||
#include "bseq.h"
|
||||
#include "minimap.h"
|
||||
#include "mmpriv.h"
|
||||
#include "ketopt.h"
|
||||
|
||||
#define MM_VERSION "2.17-r963-dirty"
|
||||
//#include "profile.h"
|
||||
#include <stdint.h>
|
||||
#include <unistd.h>
|
||||
#include <x86intrin.h>
|
||||
|
||||
using namespace std;
|
||||
uint64_t avg;
|
||||
uint64_t minimizer_lookup_time, alignment_time, dp_time, rmq_time, rmq_t1, rmq_t2, rmq_t3, rmq_t4;
|
||||
|
||||
bool enable_vect_dp_chaining = false;
|
||||
|
||||
#ifdef LISA_HASH
|
||||
#include "lisa_hash.h"
|
||||
lisa_hash<uint64_t, uint64_t> *lh;
|
||||
#endif
|
||||
|
||||
// New memory allocation approach for alignment optimizations
|
||||
//
|
||||
void *km1;
|
||||
uint64_t km_size = 500000000; // 500 MB
|
||||
int km_top;
|
||||
/*
|
||||
void *kcalloc_(void* km, int count, int size)
|
||||
{
|
||||
assert(count*size < km_size);
|
||||
km_top += count*size + 1024;
|
||||
memset(km, 0, count * size);
|
||||
|
||||
// printf("km_top: %d\n", km_top);
|
||||
return km;
|
||||
}
|
||||
|
||||
void *kmalloc_(void* km, int count) {
|
||||
if(km_top + count >= km_size)
|
||||
printf("count: %d\n", count);
|
||||
assert(km_top + count < km_size);
|
||||
void *mem = (void*) ((int8_t*) km + km_top);
|
||||
km_top += count + 1024;
|
||||
// printf("km_top: %d\n", km_top);
|
||||
return mem;
|
||||
}
|
||||
|
||||
void kfree_all() { km_top = 0;}
|
||||
*/
|
||||
|
||||
// Memory for alignment end
|
||||
|
||||
|
||||
#ifndef __rdtsc
|
||||
#ifdef _rdtsc
|
||||
#define __rdtsc _rdtsc
|
||||
#else
|
||||
#define __rdtsc __builtin_ia32_rdtsc
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#define MM_VERSION "2.22-r1101"
|
||||
|
||||
#ifdef __linux__
|
||||
#include <sys/resource.h>
|
||||
@@ -67,6 +136,13 @@ static ko_longopt_t long_options[] = {
|
||||
{ "junc-bed", ko_required_argument, 340 },
|
||||
{ "junc-bonus", ko_required_argument, 341 },
|
||||
{ "sam-hit-only", ko_no_argument, 342 },
|
||||
{ "chain-gap-scale",ko_required_argument, 343 },
|
||||
{ "alt", ko_required_argument, 344 },
|
||||
{ "alt-drop", ko_required_argument, 345 },
|
||||
{ "mask-len", ko_required_argument, 346 },
|
||||
{ "rmq", ko_optional_argument, 347 },
|
||||
{ "qstrand", ko_no_argument, 348 },
|
||||
{ "cap-kalloc", ko_required_argument, 349 },
|
||||
{ "help", ko_no_argument, 'h' },
|
||||
{ "max-intron-len", ko_required_argument, 'G' },
|
||||
{ "version", ko_no_argument, 'V' },
|
||||
@@ -78,18 +154,24 @@ static ko_longopt_t long_options[] = {
|
||||
{ 0, 0, 0 }
|
||||
};
|
||||
|
||||
static inline int64_t mm_parse_num(const char *str)
|
||||
static inline int64_t mm_parse_num2(const char *str, char **q)
|
||||
{
|
||||
double x;
|
||||
char *p;
|
||||
x = strtod(str, &p);
|
||||
if (*p == 'G' || *p == 'g') x *= 1e9;
|
||||
else if (*p == 'M' || *p == 'm') x *= 1e6;
|
||||
else if (*p == 'K' || *p == 'k') x *= 1e3;
|
||||
if (*p == 'G' || *p == 'g') x *= 1e9, ++p;
|
||||
else if (*p == 'M' || *p == 'm') x *= 1e6, ++p;
|
||||
else if (*p == 'K' || *p == 'k') x *= 1e3, ++p;
|
||||
if (q) *q = p;
|
||||
return (int64_t)(x + .499);
|
||||
}
|
||||
|
||||
static inline void yes_or_no(mm_mapopt_t *opt, int flag, int long_idx, const char *arg, int yes_to_set)
|
||||
static inline int64_t mm_parse_num(const char *str)
|
||||
{
|
||||
return mm_parse_num2(str, 0);
|
||||
}
|
||||
|
||||
static inline void yes_or_no(mm_mapopt_t *opt, int64_t flag, int long_idx, const char *arg, int yes_to_set)
|
||||
{
|
||||
if (yes_to_set) {
|
||||
if (strcmp(arg, "yes") == 0 || strcmp(arg, "y") == 0) opt->flag |= flag;
|
||||
@@ -104,12 +186,18 @@ static inline void yes_or_no(mm_mapopt_t *opt, int flag, int long_idx, const cha
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
const char *opt_str = "2aSDw:k:K:t:r:f:Vv:g:G:I:d:XT:s:x:Hcp:M:n:z:A:B:O:E:m:N:Qu:R:hF:LC:yYPo:";
|
||||
// Memory allocation for alignment optimizations
|
||||
//km1 = calloc(km_size, 1); // 10 MB init contg. alloc
|
||||
#ifdef PARALLEL_CHAINING
|
||||
enable_vect_dp_chaining = true;
|
||||
#endif
|
||||
|
||||
const char *opt_str = "2aSDw:k:K:t:r:f:Vv:g:G:I:d:XT:s:x:Hcp:M:n:z:A:B:O:E:m:N:Qu:R:hF:LC:yYPo:e:U:";
|
||||
ketopt_t o = KETOPT_INIT;
|
||||
mm_mapopt_t opt;
|
||||
mm_idxopt_t ipt;
|
||||
int i, c, n_threads = 3, n_parts, old_best_n = -1;
|
||||
char *fnw = 0, *rg = 0, *junc_bed = 0, *s;
|
||||
char *fnw = 0, *rg = 0, *junc_bed = 0, *s, *alt_list = 0;
|
||||
FILE *fp_help = stderr;
|
||||
mm_idx_reader_t *idx_rdr;
|
||||
mm_idx_t *mi;
|
||||
@@ -118,9 +206,11 @@ int main(int argc, char *argv[])
|
||||
liftrlimit();
|
||||
mm_realtime0 = realtime();
|
||||
mm_set_opt(0, &ipt, &opt);
|
||||
string preset_arg = "";
|
||||
|
||||
while ((c = ketopt(&o, argc, argv, 1, opt_str, long_options)) >= 0) { // test command line options and apply option -x/preset first
|
||||
if (c == 'x') {
|
||||
preset_arg += (string) o.arg;
|
||||
if (mm_set_opt(o.arg, &ipt, &opt) < 0) {
|
||||
fprintf(stderr, "[ERROR] unknown preset '%s'\n", o.arg);
|
||||
return 1;
|
||||
@@ -140,7 +230,6 @@ int main(int argc, char *argv[])
|
||||
else if (c == 'k') ipt.k = atoi(o.arg);
|
||||
else if (c == 'H') ipt.flag |= MM_I_HPC;
|
||||
else if (c == 'd') fnw = o.arg; // the above are indexing related options, except -I
|
||||
else if (c == 'r') opt.bw = (int)mm_parse_num(o.arg);
|
||||
else if (c == 't') n_threads = atoi(o.arg);
|
||||
else if (c == 'v') mm_verbose = atoi(o.arg);
|
||||
else if (c == 'g') opt.max_gap = (int)mm_parse_num(o.arg);
|
||||
@@ -166,7 +255,8 @@ int main(int argc, char *argv[])
|
||||
else if (c == 's') opt.min_dp_max = atoi(o.arg);
|
||||
else if (c == 'C') opt.noncan = atoi(o.arg);
|
||||
else if (c == 'I') ipt.batch_size = mm_parse_num(o.arg);
|
||||
else if (c == 'K') opt.mini_batch_size = (int)mm_parse_num(o.arg);
|
||||
else if (c == 'K') opt.mini_batch_size = mm_parse_num(o.arg);
|
||||
else if (c == 'e') opt.occ_dist = mm_parse_num(o.arg);
|
||||
else if (c == 'R') rg = o.arg;
|
||||
else if (c == 'h') fp_help = stdout;
|
||||
else if (c == '2') opt.flag |= MM_F_2_IO_THREADS;
|
||||
@@ -199,7 +289,6 @@ int main(int argc, char *argv[])
|
||||
else if (c == 327) opt.max_clip_ratio = atof(o.arg); // --max-clip-ratio
|
||||
else if (c == 328) opt.min_mid_occ = atoi(o.arg); // --min-occ-floor
|
||||
else if (c == 329) opt.flag |= MM_F_OUT_MD; // --MD
|
||||
else if (c == 330) opt.min_join_flank_ratio = atof(o.arg); // --lj-min-ratio
|
||||
else if (c == 331) opt.sc_ambi = atoi(o.arg); // --score-N
|
||||
else if (c == 332) opt.flag |= MM_F_EQX; // --eqx
|
||||
else if (c == 333) opt.flag |= MM_F_PAF_NO_HIT; // --paf-no-hit
|
||||
@@ -211,7 +300,15 @@ int main(int argc, char *argv[])
|
||||
else if (c == 340) junc_bed = o.arg; // --junc-bed
|
||||
else if (c == 341) opt.junc_bonus = atoi(o.arg); // --junc-bonus
|
||||
else if (c == 342) opt.flag |= MM_F_SAM_HIT_ONLY; // --sam-hit-only
|
||||
else if (c == 314) { // --frag
|
||||
else if (c == 343) opt.chain_gap_scale = atof(o.arg); // --chain-gap-scale
|
||||
else if (c == 344) alt_list = o.arg; // --alt
|
||||
else if (c == 345) opt.alt_drop = atof(o.arg); // --alt-drop
|
||||
else if (c == 346) opt.mask_len = mm_parse_num(o.arg); // --mask-len
|
||||
else if (c == 348) opt.flag |= MM_F_QSTRAND | MM_F_NO_INV; // --qstrand
|
||||
else if (c == 349) opt.cap_kalloc = mm_parse_num(o.arg); // --cap-kalloc
|
||||
else if (c == 330) {
|
||||
fprintf(stderr, "[WARNING] \033[1;31m --lj-min-ratio has been deprecated.\033[0m\n");
|
||||
} else if (c == 314) { // --frag
|
||||
yes_or_no(&opt, MM_F_FRAG_MODE, o.longidx, o.arg, 1);
|
||||
} else if (c == 315) { // --secondary
|
||||
yes_or_no(&opt, MM_F_NO_PRINT_2ND, o.longidx, o.arg, 0);
|
||||
@@ -232,6 +329,8 @@ int main(int argc, char *argv[])
|
||||
yes_or_no(&opt, MM_F_HEAP_SORT, o.longidx, o.arg, 1);
|
||||
} else if (c == 326) { // --dual
|
||||
yes_or_no(&opt, MM_F_NO_DUAL, o.longidx, o.arg, 0);
|
||||
} else if (c == 347) { // --rmq
|
||||
yes_or_no(&opt, MM_F_RMQ, o.longidx, o.arg, 1);
|
||||
} else if (c == 'S') {
|
||||
opt.flag |= MM_F_OUT_CS | MM_F_CIGAR | MM_F_OUT_CS_LONG;
|
||||
if (mm_verbose >= 2)
|
||||
@@ -239,6 +338,12 @@ int main(int argc, char *argv[])
|
||||
} else if (c == 'V') {
|
||||
puts(MM_VERSION);
|
||||
return 0;
|
||||
} else if (c == 'r') {
|
||||
opt.bw = (int)mm_parse_num2(o.arg, &s);
|
||||
if (*s == ',') opt.bw_long = (int)mm_parse_num2(s + 1, &s);
|
||||
} else if (c == 'U') {
|
||||
opt.min_mid_occ = strtol(o.arg, &s, 10);
|
||||
if (*s == ',') opt.max_mid_occ = strtol(s + 1, &s, 10);
|
||||
} else if (c == 'f') {
|
||||
double x;
|
||||
char *p;
|
||||
@@ -293,7 +398,7 @@ int main(int argc, char *argv[])
|
||||
fprintf(fp_help, " -g NUM stop chain enlongation if there are no minimizers in INT-bp [%d]\n", opt.max_gap);
|
||||
fprintf(fp_help, " -G NUM max intron length (effective with -xsplice; changing -r) [200k]\n");
|
||||
fprintf(fp_help, " -F NUM max fragment length (effective with -xsr or in the fragment mode) [800]\n");
|
||||
fprintf(fp_help, " -r NUM bandwidth used in chaining and DP-based alignment [%d]\n", opt.bw);
|
||||
fprintf(fp_help, " -r NUM[,NUM] chaining/alignment bandwidth and long-join bandwidth [%d,%d]\n", opt.bw, opt.bw_long);
|
||||
fprintf(fp_help, " -n INT minimal number of minimizers on a chain [%d]\n", opt.min_cnt);
|
||||
fprintf(fp_help, " -m INT minimal chaining score (matching bases minus log gap penalty) [%d]\n", opt.min_chain_score);
|
||||
// fprintf(fp_help, " -T INT SDUST threshold; 0 to disable SDUST [%d]\n", opt.sdust_thres); // TODO: this option is never used; might be buggy
|
||||
@@ -302,7 +407,7 @@ int main(int argc, char *argv[])
|
||||
fprintf(fp_help, " -N INT retain at most INT secondary alignments [%d]\n", opt.best_n);
|
||||
fprintf(fp_help, " Alignment:\n");
|
||||
fprintf(fp_help, " -A INT matching score [%d]\n", opt.a);
|
||||
fprintf(fp_help, " -B INT mismatch penalty [%d]\n", opt.b);
|
||||
fprintf(fp_help, " -B INT mismatch penalty (larger value for lower divergence) [%d]\n", opt.b);
|
||||
fprintf(fp_help, " -O INT[,INT] gap open penalty [%d,%d]\n", opt.q, opt.q2);
|
||||
fprintf(fp_help, " -E INT[,INT] gap extension penalty; a k-long gap costs min{O1+k*E1,O2+k*E2} [%d,%d]\n", opt.e, opt.e2);
|
||||
fprintf(fp_help, " -z INT[,INT] Z-drop score and inversion Z-drop score [%d,%d]\n", opt.zdrop, opt.zdrop_inv);
|
||||
@@ -324,7 +429,8 @@ int main(int argc, char *argv[])
|
||||
fprintf(fp_help, " --version show version number\n");
|
||||
fprintf(fp_help, " Preset:\n");
|
||||
fprintf(fp_help, " -x STR preset (always applied before other options; see minimap2.1 for details) []\n");
|
||||
fprintf(fp_help, " - map-pb/map-ont - PacBio/Nanopore vs reference mapping\n");
|
||||
fprintf(fp_help, " - map-pb/map-ont - PacBio CLR/Nanopore vs reference mapping\n");
|
||||
fprintf(fp_help, " - map-hifi - PacBio HiFi reads vs reference mapping\n");
|
||||
fprintf(fp_help, " - ava-pb/ava-ont - PacBio/Nanopore read overlap\n");
|
||||
fprintf(fp_help, " - asm5/asm10/asm20 - asm-to-ref mapping, for ~0.1/1/5%% sequence divergence\n");
|
||||
fprintf(fp_help, " - splice/splice:hq - long-read/Pacbio-CCS spliced alignment\n");
|
||||
@@ -337,6 +443,7 @@ int main(int argc, char *argv[])
|
||||
fprintf(stderr, "[ERROR] incorrect input: in the sr mode, please specify no more than two query files.\n");
|
||||
return 1;
|
||||
}
|
||||
preset_arg = (string)argv[o.ind] + "_" + preset_arg + "_minimizers_key_value_sorted";
|
||||
idx_rdr = mm_idx_reader_open(argv[o.ind], &ipt, fnw);
|
||||
if (idx_rdr == 0) {
|
||||
fprintf(stderr, "[ERROR] failed to open file '%s': %s\n", argv[o.ind], strerror(errno));
|
||||
@@ -350,6 +457,7 @@ int main(int argc, char *argv[])
|
||||
if (opt.best_n == 0 && (opt.flag&MM_F_CIGAR) && mm_verbose >= 2)
|
||||
fprintf(stderr, "[WARNING]\033[1;31m `-N 0' reduces alignment accuracy. Please use --secondary=no to suppress secondary alignments.\033[0m\n");
|
||||
while ((mi = mm_idx_reader_read(idx_rdr, n_threads)) != 0) {
|
||||
int ret;
|
||||
if ((opt.flag & MM_F_CIGAR) && (mi->flag & MM_I_NO_SEQ)) {
|
||||
fprintf(stderr, "[ERROR] the prebuilt index doesn't contain sequences.\n");
|
||||
mm_idx_destroy(mi);
|
||||
@@ -357,9 +465,11 @@ int main(int argc, char *argv[])
|
||||
return 1;
|
||||
}
|
||||
if ((opt.flag & MM_F_OUT_SAM) && idx_rdr->n_parts == 1) {
|
||||
int ret;
|
||||
if (mm_idx_reader_eof(idx_rdr)) {
|
||||
ret = mm_write_sam_hdr(mi, rg, MM_VERSION, argc, argv);
|
||||
if (opt.split_prefix == 0)
|
||||
ret = mm_write_sam_hdr(mi, rg, MM_VERSION, argc, argv);
|
||||
else
|
||||
ret = mm_write_sam_hdr(0, rg, MM_VERSION, argc, argv);
|
||||
} else {
|
||||
ret = mm_write_sam_hdr(0, rg, MM_VERSION, argc, argv);
|
||||
if (opt.split_prefix == 0 && mm_verbose >= 2)
|
||||
@@ -376,15 +486,45 @@ int main(int argc, char *argv[])
|
||||
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), mi->n_seq);
|
||||
if (argc != o.ind + 1) mm_mapopt_update(&opt, mi);
|
||||
if (mm_verbose >= 3) mm_idx_stat(mi);
|
||||
#ifdef LISA_INDEX
|
||||
mm_idx_dump_hash(preset_arg.c_str(), mi);
|
||||
#endif
|
||||
if (junc_bed) mm_idx_bed_read(mi, junc_bed, 1);
|
||||
if (alt_list) mm_idx_alt_read(mi, alt_list);
|
||||
if (argc - (o.ind + 1) == 0) {
|
||||
mm_idx_destroy(mi);
|
||||
continue; // no query files
|
||||
}
|
||||
ret = 0;
|
||||
#ifdef LISA_HASH
|
||||
fprintf(stderr, "Using LISA_HASH..\n");
|
||||
mm_idx_destroy_mm_hash(mi);
|
||||
char* prefix;
|
||||
lh = new lisa_hash<uint64_t, uint64_t>(preset_arg, prefix);
|
||||
fprintf(stderr, "Loading done.\n");
|
||||
// total_time = __rdtsc();
|
||||
// fprintf(stderr, "\nIndexing Real time: %.3f sec;\n", realtime() - mapping_time);
|
||||
#endif
|
||||
mm_realtime0 = realtime();
|
||||
if (!(opt.flag & MM_F_FRAG_MODE)) {
|
||||
for (i = o.ind + 1; i < argc; ++i)
|
||||
mm_map_file(mi, argv[i], &opt, n_threads);
|
||||
for (i = o.ind + 1; i < argc; ++i) {
|
||||
ret = mm_map_file(mi, argv[i], &opt, n_threads);
|
||||
if (ret < 0) break;
|
||||
}
|
||||
} else {
|
||||
mm_map_file_frag(mi, argc - (o.ind + 1), (const char**)&argv[o.ind + 1], &opt, n_threads);
|
||||
ret = mm_map_file_frag(mi, argc - (o.ind + 1), (const char**)&argv[o.ind + 1], &opt, n_threads);
|
||||
}
|
||||
//mm_idx_destroy(mi);
|
||||
if (ret < 0) {
|
||||
fprintf(stderr, "ERROR: failed to map the query file\n");
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
mm_idx_destroy(mi);
|
||||
}
|
||||
#ifdef LISA_HASH
|
||||
mm_idx_destroy_seq(mi);
|
||||
#else
|
||||
mm_idx_destroy(mi);
|
||||
#endif
|
||||
n_parts = idx_rdr->n_parts;
|
||||
mm_idx_reader_close(idx_rdr);
|
||||
|
||||
@@ -403,5 +543,10 @@ int main(int argc, char *argv[])
|
||||
fprintf(stderr, " %s", argv[i]);
|
||||
fprintf(stderr, "\n[M::%s] Real time: %.3f sec; CPU: %.3f sec; Peak RSS: %.3f GB\n", __func__, realtime() - mm_realtime0, cputime(), peakrss() / 1024.0 / 1024.0 / 1024.0);
|
||||
}
|
||||
|
||||
fprintf(stderr, "minimizer-lookup: %lld dp: %lld rmq: %lld rmq_t1: %lld rmq_t2: %lld rmq_t3: %lld rmq_t4: %lld alignment: %lld %lld\n", minimizer_lookup_time, dp_time, rmq_time, rmq_t1, rmq_t2, rmq_t3, rmq_t4, alignment_time, avg);
|
||||
#ifdef LISA_HASH
|
||||
delete lh;
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -9,6 +9,12 @@
|
||||
#include "mmpriv.h"
|
||||
#include "bseq.h"
|
||||
#include "khash.h"
|
||||
#include <x86intrin.h>
|
||||
|
||||
#ifdef MANUAL_PROFILING
|
||||
extern uint64_t minimizer_lookup_time;
|
||||
extern uint64_t rmq_time;
|
||||
#endif
|
||||
|
||||
struct mm_tbuf_s {
|
||||
void *km;
|
||||
@@ -80,49 +86,7 @@ static void collect_minimizers(void *km, const mm_mapopt_t *opt, const mm_idx_t
|
||||
#define heap_lt(a, b) ((a).x > (b).x)
|
||||
KSORT_INIT(heap, mm128_t, heap_lt)
|
||||
|
||||
typedef struct {
|
||||
uint32_t n;
|
||||
uint32_t q_pos, q_span;
|
||||
uint32_t seg_id:31, is_tandem:1;
|
||||
const uint64_t *cr;
|
||||
} mm_match_t;
|
||||
|
||||
static mm_match_t *collect_matches(void *km, int *_n_m, int max_occ, const mm_idx_t *mi, const mm128_v *mv, int64_t *n_a, int *rep_len, int *n_mini_pos, uint64_t **mini_pos)
|
||||
{
|
||||
int rep_st = 0, rep_en = 0, n_m;
|
||||
size_t i;
|
||||
mm_match_t *m;
|
||||
*n_mini_pos = 0;
|
||||
*mini_pos = (uint64_t*)kmalloc(km, mv->n * sizeof(uint64_t));
|
||||
m = (mm_match_t*)kmalloc(km, mv->n * sizeof(mm_match_t));
|
||||
for (i = 0, n_m = 0, *rep_len = 0, *n_a = 0; i < mv->n; ++i) {
|
||||
const uint64_t *cr;
|
||||
mm128_t *p = &mv->a[i];
|
||||
uint32_t q_pos = (uint32_t)p->y, q_span = p->x & 0xff;
|
||||
int t;
|
||||
cr = mm_idx_get(mi, p->x>>8, &t);
|
||||
if (t >= max_occ) {
|
||||
int en = (q_pos >> 1) + 1, st = en - q_span;
|
||||
if (st > rep_en) {
|
||||
*rep_len += rep_en - rep_st;
|
||||
rep_st = st, rep_en = en;
|
||||
} else rep_en = en;
|
||||
} else {
|
||||
mm_match_t *q = &m[n_m++];
|
||||
q->q_pos = q_pos, q->q_span = q_span, q->cr = cr, q->n = t, q->seg_id = p->y >> 32;
|
||||
q->is_tandem = 0;
|
||||
if (i > 0 && p->x>>8 == mv->a[i - 1].x>>8) q->is_tandem = 1;
|
||||
if (i < mv->n - 1 && p->x>>8 == mv->a[i + 1].x>>8) q->is_tandem = 1;
|
||||
*n_a += q->n;
|
||||
(*mini_pos)[(*n_mini_pos)++] = (uint64_t)q_span<<32 | q_pos>>1;
|
||||
}
|
||||
}
|
||||
*rep_len += rep_en - rep_st;
|
||||
*_n_m = n_m;
|
||||
return m;
|
||||
}
|
||||
|
||||
static inline int skip_seed(int flag, uint64_t r, const mm_match_t *q, const char *qname, int qlen, const mm_idx_t *mi, int *is_self)
|
||||
static inline int skip_seed(int flag, uint64_t r, const mm_seed_t *q, const char *qname, int qlen, const mm_idx_t *mi, int *is_self)
|
||||
{
|
||||
*is_self = 0;
|
||||
if (qname && (flag & (MM_F_NO_DIAG|MM_F_NO_DUAL))) {
|
||||
@@ -151,10 +115,10 @@ static mm128_t *collect_seed_hits_heap(void *km, const mm_mapopt_t *opt, int max
|
||||
{
|
||||
int i, n_m, heap_size = 0;
|
||||
int64_t j, n_for = 0, n_rev = 0;
|
||||
mm_match_t *m;
|
||||
mm_seed_t *m;
|
||||
mm128_t *a, *heap;
|
||||
|
||||
m = collect_matches(km, &n_m, max_occ, mi, mv, n_a, rep_len, n_mini_pos, mini_pos);
|
||||
m = mm_collect_matches(km, &n_m, qlen, max_occ, opt->max_max_occ, opt->occ_dist, mi, mv, n_a, rep_len, n_mini_pos, mini_pos);
|
||||
|
||||
heap = (mm128_t*)kmalloc(km, n_m * sizeof(mm128_t));
|
||||
a = (mm128_t*)kmalloc(km, *n_a * sizeof(mm128_t));
|
||||
@@ -168,7 +132,7 @@ static mm128_t *collect_seed_hits_heap(void *km, const mm_mapopt_t *opt, int max
|
||||
}
|
||||
ks_heapmake_heap(heap_size, heap);
|
||||
while (heap_size > 0) {
|
||||
mm_match_t *q = &m[heap->y>>32];
|
||||
mm_seed_t *q = &m[heap->y>>32];
|
||||
mm128_t *p;
|
||||
uint64_t r = heap->x;
|
||||
int32_t is_self, rpos = (uint32_t)r >> 1;
|
||||
@@ -215,13 +179,16 @@ static mm128_t *collect_seed_hits_heap(void *km, const mm_mapopt_t *opt, int max
|
||||
static mm128_t *collect_seed_hits(void *km, const mm_mapopt_t *opt, int max_occ, const mm_idx_t *mi, const char *qname, const mm128_v *mv, int qlen, int64_t *n_a, int *rep_len,
|
||||
int *n_mini_pos, uint64_t **mini_pos)
|
||||
{
|
||||
#ifdef MANUAL_PROFILING
|
||||
uint64_t lookup_start = __rdtsc();
|
||||
#endif
|
||||
int i, n_m;
|
||||
mm_match_t *m;
|
||||
mm_seed_t *m;
|
||||
mm128_t *a;
|
||||
m = collect_matches(km, &n_m, max_occ, mi, mv, n_a, rep_len, n_mini_pos, mini_pos);
|
||||
m = mm_collect_matches(km, &n_m, qlen, max_occ, opt->max_max_occ, opt->occ_dist, mi, mv, n_a, rep_len, n_mini_pos, mini_pos);
|
||||
a = (mm128_t*)kmalloc(km, *n_a * sizeof(mm128_t));
|
||||
for (i = 0, *n_a = 0; i < n_m; ++i) {
|
||||
mm_match_t *q = &m[i];
|
||||
mm_seed_t *q = &m[i];
|
||||
const uint64_t *r = q->cr;
|
||||
uint32_t k;
|
||||
for (k = 0; k < q->n; ++k) {
|
||||
@@ -232,9 +199,13 @@ static mm128_t *collect_seed_hits(void *km, const mm_mapopt_t *opt, int max_occ,
|
||||
if ((r[k]&1) == (q->q_pos&1)) { // forward strand
|
||||
p->x = (r[k]&0xffffffff00000000ULL) | rpos;
|
||||
p->y = (uint64_t)q->q_span << 32 | q->q_pos >> 1;
|
||||
} else { // reverse strand
|
||||
} else if (!(opt->flag & MM_F_QSTRAND)) { // reverse strand and not in the query-strand mode
|
||||
p->x = 1ULL<<63 | (r[k]&0xffffffff00000000ULL) | rpos;
|
||||
p->y = (uint64_t)q->q_span << 32 | (qlen - ((q->q_pos>>1) + 1 - q->q_span) - 1);
|
||||
} else { // reverse strand; query-strand
|
||||
int32_t len = mi->seq[r[k]>>32].len;
|
||||
p->x = 1ULL<<63 | (r[k]&0xffffffff00000000ULL) | (len - (rpos + 1 - q->q_span) - 1); // coordinate only accurate for non-HPC seeds
|
||||
p->y = (uint64_t)q->q_span << 32 | q->q_pos >> 1;
|
||||
}
|
||||
p->y |= (uint64_t)q->seg_id << MM_SEED_SEG_SHIFT;
|
||||
if (q->is_tandem) p->y |= MM_SEED_TANDEM;
|
||||
@@ -243,17 +214,18 @@ static mm128_t *collect_seed_hits(void *km, const mm_mapopt_t *opt, int max_occ,
|
||||
}
|
||||
kfree(km, m);
|
||||
radix_sort_128x(a, a + (*n_a));
|
||||
#ifdef MANUAL_PROFILING
|
||||
minimizer_lookup_time += __rdtsc() - lookup_start;
|
||||
#endif
|
||||
return a;
|
||||
}
|
||||
|
||||
static void chain_post(const mm_mapopt_t *opt, int max_chain_gap_ref, const mm_idx_t *mi, void *km, int qlen, int n_segs, const int *qlens, int *n_regs, mm_reg1_t *regs, mm128_t *a)
|
||||
{
|
||||
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
||||
mm_set_parent(km, opt->mask_level, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL);
|
||||
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
||||
if (n_segs <= 1) mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, n_regs, regs);
|
||||
else mm_select_sub_multi(km, opt->pri_ratio, 0.2f, 0.7f, max_chain_gap_ref, mi->k*2, opt->best_n, n_segs, qlens, n_regs, regs);
|
||||
if (!(opt->flag & (MM_F_SPLICE|MM_F_SR|MM_F_NO_LJOIN))) // long join not working well without primary chains
|
||||
mm_join_long(km, opt, qlen, n_regs, regs, a);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -262,7 +234,7 @@ static mm_reg1_t *align_regs(const mm_mapopt_t *opt, const mm_idx_t *mi, void *k
|
||||
if (!(opt->flag & MM_F_CIGAR)) return regs;
|
||||
regs = mm_align_skeleton(km, opt, mi, qlen, seq, n_regs, regs, a); // this calls mm_filter_regs()
|
||||
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
||||
mm_set_parent(km, opt->mask_level, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL);
|
||||
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
||||
mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, n_regs, regs);
|
||||
mm_set_sam_pri(*n_regs, regs);
|
||||
}
|
||||
@@ -313,9 +285,41 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
if (max_chain_gap_ref < opt->max_gap) max_chain_gap_ref = opt->max_gap;
|
||||
} else max_chain_gap_ref = opt->max_gap;
|
||||
|
||||
a = mm_chain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||
if (opt->flag & MM_F_RMQ) {
|
||||
a = mg_lchain_rmq(opt->max_gap, opt->rmq_inner_dist, opt->bw, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
||||
opt->chain_gap_scale * 0.01 * mi->k, 0.0f, n_a, a, &n_regs0, &u, b->km);
|
||||
// a = mg_lchain_dp(opt->max_gap, opt->rmq_inner_dist, opt->bw, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
||||
// opt->chain_gap_scale * 0.01 * mi->k, 0.0f, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||
|
||||
if (opt->max_occ > opt->mid_occ && rep_len > 0) {
|
||||
|
||||
} else {
|
||||
//fprintf(stderr, "dp call - n_a = %lld\n", n_a);
|
||||
a = mg_lchain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score,
|
||||
opt->chain_gap_scale * 0.01 * mi->k, 0.0f, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||
}
|
||||
|
||||
if (opt->bw_long > opt->bw && (opt->flag & (MM_F_SPLICE|MM_F_SR|MM_F_NO_LJOIN)) == 0 && n_segs == 1 && n_regs0 > 1) { // re-chain/long-join for long sequences
|
||||
int32_t st = (int32_t)a[0].y, en = (int32_t)a[(int32_t)u[0] - 1].y;
|
||||
if (qlen_sum - (en - st) > opt->rmq_rescue_size || en - st > qlen_sum * opt->rmq_rescue_ratio) {
|
||||
#ifdef MANUAL_PROFILING
|
||||
// uint64_t tim = __rdtsc();
|
||||
#endif
|
||||
// fprintf(stderr, "pre: rmq rechain call - n_a = %lld n_regs = %lld\n",n_a, n_regs0);
|
||||
int32_t i;
|
||||
int64_t prev_n_a = n_a;
|
||||
for (i = 0, n_a = 0; i < n_regs0; ++i) n_a += (int32_t)u[i];
|
||||
kfree(b->km, u);
|
||||
radix_sort_128x(a, a + n_a);
|
||||
// fprintf(stderr, "post: rmq rechain call - prev_n_a = %lld n_a = %lld n_regs = %lld\n",prev_n_a, n_a, n_regs0);
|
||||
// a = mg_lchain_dp(opt->max_gap, opt->rmq_inner_dist, opt->bw_long, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
||||
// opt->chain_gap_scale * 0.01 * mi->k, 0.0f, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||
a = mg_lchain_rmq(opt->max_gap, opt->rmq_inner_dist, opt->bw_long, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
||||
opt->chain_gap_scale * 0.01 * mi->k, 0.0f, n_a, a, &n_regs0, &u, b->km);
|
||||
#ifdef MANUAL_PROFILING
|
||||
// rmq_time += __rdtsc() - tim;
|
||||
#endif
|
||||
}
|
||||
} else if (opt->max_occ > opt->mid_occ && rep_len > 0 && !(opt->flag & MM_F_RMQ)) { // re-chain, mostly for short reads
|
||||
int rechain = 0;
|
||||
if (n_regs0 > 0) { // test if the best chain has all the segments
|
||||
int n_chained_segs = 1, max = 0, max_i = -1, max_off = -1, off = 0;
|
||||
@@ -335,13 +339,18 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
kfree(b->km, mini_pos);
|
||||
if (opt->flag & MM_F_HEAP_SORT) a = collect_seed_hits_heap(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||
else a = collect_seed_hits(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||
a = mm_chain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||
a = mg_lchain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score,
|
||||
opt->chain_gap_scale * 0.01 * mi->k, 0.0f, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||
}
|
||||
}
|
||||
b->frag_gap = max_chain_gap_ref;
|
||||
b->rep_len = rep_len;
|
||||
|
||||
regs0 = mm_gen_regs(b->km, hash, qlen_sum, n_regs0, u, a);
|
||||
regs0 = mm_gen_regs(b->km, hash, qlen_sum, n_regs0, u, a, !!(opt->flag&MM_F_QSTRAND));
|
||||
if (mi->n_alt) {
|
||||
mm_mark_alt(mi, n_regs0, regs0);
|
||||
mm_hit_sort(b->km, &n_regs0, regs0, opt->alt_drop); // this step can be merged into mm_gen_regs(); will do if this shows up in profile
|
||||
}
|
||||
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_SEED)
|
||||
for (j = 0; j < n_regs0; ++j)
|
||||
@@ -350,10 +359,12 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
i == regs0[j].as? 0 : ((int32_t)a[i].y - (int32_t)a[i-1].y) - ((int32_t)a[i].x - (int32_t)a[i-1].x));
|
||||
|
||||
chain_post(opt, max_chain_gap_ref, mi, b->km, qlen_sum, n_segs, qlens, &n_regs0, regs0, a);
|
||||
if (!is_sr) mm_est_err(mi, qlen_sum, n_regs0, regs0, a, n_mini_pos, mini_pos);
|
||||
if (!is_sr && !(opt->flag&MM_F_QSTRAND))
|
||||
mm_est_err(mi, qlen_sum, n_regs0, regs0, a, n_mini_pos, mini_pos);
|
||||
|
||||
if (n_segs == 1) { // uni-segment
|
||||
regs0 = align_regs(opt, mi, b->km, qlens[0], seqs[0], &n_regs0, regs0, a);
|
||||
regs0 = (mm_reg1_t*)realloc(regs0, sizeof(*regs0) * n_regs0);
|
||||
mm_set_mapq(b->km, n_regs0, regs0, opt->min_chain_score, opt->a, rep_len, is_sr);
|
||||
n_regs[0] = n_regs0, regs[0] = regs0;
|
||||
} else { // multi-segment
|
||||
@@ -361,7 +372,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
seg = mm_seg_gen(b->km, hash, n_segs, qlens, n_regs0, regs0, n_regs, regs, a); // split fragment chain to separate segment chains
|
||||
free(regs0);
|
||||
for (i = 0; i < n_segs; ++i) {
|
||||
mm_set_parent(b->km, opt->mask_level, n_regs[i], regs[i], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL); // update mm_reg1_t::parent
|
||||
mm_set_parent(b->km, opt->mask_level, opt->mask_len, n_regs[i], regs[i], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop); // update mm_reg1_t::parent
|
||||
regs[i] = align_regs(opt, mi, b->km, qlens[i], seqs[i], &n_regs[i], regs[i], seg[i].a);
|
||||
mm_set_mapq(b->km, n_regs[i], regs[i], opt->min_chain_score, opt->a, rep_len, is_sr);
|
||||
}
|
||||
@@ -380,7 +391,9 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||
fprintf(stderr, "QM\t%s\t%d\tcap=%ld,nCore=%ld,largest=%ld\n", qname, qlen_sum, kmst.capacity, kmst.n_cores, kmst.largest);
|
||||
assert(kmst.n_blocks == kmst.n_cores); // otherwise, there is a memory leak
|
||||
if (kmst.largest > 1U<<28) {
|
||||
if (kmst.largest > 1U<<28 || (opt->cap_kalloc > 0 && kmst.capacity > opt->cap_kalloc)) {
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||
fprintf(stderr, "[W::%s] reset thread-local memory after read %s\n", __func__, qname);
|
||||
km_destroy(b->km);
|
||||
b->km = km_init();
|
||||
}
|
||||
@@ -399,7 +412,8 @@ mm_reg1_t *mm_map(const mm_idx_t *mi, int qlen, const char *seq, int *n_regs, mm
|
||||
**************************/
|
||||
|
||||
typedef struct {
|
||||
int mini_batch_size, n_processed, n_threads, n_fp;
|
||||
int n_processed, n_threads, n_fp;
|
||||
int64_t mini_batch_size;
|
||||
const mm_mapopt_t *opt;
|
||||
mm_bseq_file_t **fp;
|
||||
const mm_idx_t *mi;
|
||||
@@ -424,10 +438,13 @@ static void worker_for(void *_data, long i, int tid) // kt_for() callback
|
||||
step_t *s = (step_t*)_data;
|
||||
int qlens[MM_MAX_SEG], j, off = s->seg_off[i], pe_ori = s->p->opt->pe_ori;
|
||||
const char *qseqs[MM_MAX_SEG];
|
||||
double t = 0.0;
|
||||
mm_tbuf_t *b = s->buf[tid];
|
||||
assert(s->n_seg[i] <= MM_MAX_SEG);
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME) {
|
||||
fprintf(stderr, "QR\t%s\t%d\t%d\n", s->seq[off].name, tid, s->seq[off].l_seq);
|
||||
t = realtime();
|
||||
}
|
||||
for (j = 0; j < s->n_seg[i]; ++j) {
|
||||
if (s->n_seg[i] == 2 && ((j == 0 && (pe_ori>>1&1)) || (j == 1 && (pe_ori&1))))
|
||||
mm_revcomp_bseq(&s->seq[off + j]);
|
||||
@@ -459,6 +476,8 @@ static void worker_for(void *_data, long i, int tid) // kt_for() callback
|
||||
r->rev = !r->rev;
|
||||
}
|
||||
}
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||
fprintf(stderr, "QT\t%s\t%d\t%.6f\n", s->seq[off].name, tid, realtime() - t);
|
||||
}
|
||||
|
||||
static void merge_hits(step_t *s)
|
||||
@@ -503,8 +522,16 @@ static void merge_hits(step_t *s)
|
||||
}
|
||||
}
|
||||
}
|
||||
mm_hit_sort(km, &s->n_reg[k], s->reg[k]);
|
||||
mm_set_parent(km, opt->mask_level, s->n_reg[k], s->reg[k], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL);
|
||||
if (!(opt->flag&MM_F_SR) && s->seq[k].l_seq >= opt->rank_min_len)
|
||||
mm_update_dp_max(s->seq[k].l_seq, s->n_reg[k], s->reg[k], opt->rank_frac, opt->a, opt->b);
|
||||
for (j = 0; j < s->n_reg[k]; ++j) {
|
||||
mm_reg1_t *r = &s->reg[k][j];
|
||||
if (r->p) r->p->dp_max2 = 0; // reset ->dp_max2 as mm_set_parent() doesn't clear it; necessary with mm_update_dp_max()
|
||||
r->subsc = 0; // this may not be necessary
|
||||
r->n_sub = 0; // n_sub will be an underestimate as we don't see all the chains now, but it can't be accurate anyway
|
||||
}
|
||||
mm_hit_sort(km, &s->n_reg[k], s->reg[k], opt->alt_drop);
|
||||
mm_set_parent(km, opt->mask_level, opt->mask_len, s->n_reg[k], s->reg[k], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
||||
if (!(opt->flag & MM_F_ALL_CHAINS)) {
|
||||
mm_select_sub(km, opt->pri_ratio, s->p->mi->k*2, opt->best_n, &s->n_reg[k], s->reg[k]);
|
||||
mm_set_sam_pri(s->n_reg[k], s->reg[k]);
|
||||
@@ -564,6 +591,7 @@ static void *worker_pipeline(void *shared, int step, void *in)
|
||||
if ((p->opt->flag & MM_F_OUT_CS) && !(mm_dbg_flag & MM_DBG_NO_KALLOC)) km = km_init();
|
||||
for (k = 0; k < s->n_frag; ++k) {
|
||||
int seg_st = s->seg_off[k], seg_en = s->seg_off[k] + s->n_seg[k];
|
||||
#ifndef DISABLE_OUTPUT
|
||||
for (i = seg_st; i < seg_en; ++i) {
|
||||
mm_bseq1_t *t = &s->seq[i];
|
||||
if (p->opt->split_prefix && p->n_parts == 0) { // then write to temporary files
|
||||
@@ -598,6 +626,7 @@ static void *worker_pipeline(void *shared, int step, void *in)
|
||||
mm_err_puts(p->str.s);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
for (i = seg_st; i < seg_en; ++i) {
|
||||
for (j = 0; j < s->n_reg[i]; ++j) free(s->reg[i][j].p);
|
||||
free(s->reg[i]);
|
||||
|
||||
@@ -36,6 +36,9 @@
|
||||
#define MM_F_NO_END_FLT 0x10000000
|
||||
#define MM_F_HARD_MLEVEL 0x20000000
|
||||
#define MM_F_SAM_HIT_ONLY 0x40000000
|
||||
#define MM_F_RMQ (0x80000000LL)
|
||||
#define MM_F_QSTRAND (0x100000000LL)
|
||||
#define MM_F_NO_INV (0x200000000LL)
|
||||
|
||||
#define MM_I_HPC 0x1
|
||||
#define MM_I_NO_SEQ 0x2
|
||||
@@ -45,6 +48,18 @@
|
||||
|
||||
#define MM_MAX_SEG 255
|
||||
|
||||
#define MM_CIGAR_MATCH 0
|
||||
#define MM_CIGAR_INS 1
|
||||
#define MM_CIGAR_DEL 2
|
||||
#define MM_CIGAR_N_SKIP 3
|
||||
#define MM_CIGAR_SOFTCLIP 4
|
||||
#define MM_CIGAR_HARDCLIP 5
|
||||
#define MM_CIGAR_PADDING 6
|
||||
#define MM_CIGAR_EQ_MATCH 7
|
||||
#define MM_CIGAR_X_MISMATCH 8
|
||||
|
||||
#define MM_CIGAR_STR "MIDNSHP=XB"
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
@@ -58,12 +73,14 @@ typedef struct {
|
||||
char *name; // name of the db sequence
|
||||
uint64_t offset; // offset in mm_idx_t::S
|
||||
uint32_t len; // length
|
||||
uint32_t is_alt;
|
||||
} mm_idx_seq_t;
|
||||
|
||||
typedef struct {
|
||||
int32_t b, w, k, flag;
|
||||
uint32_t n_seq; // number of reference sequences
|
||||
int32_t index;
|
||||
int32_t n_alt;
|
||||
mm_idx_seq_t *seq; // sequence name, length and offset
|
||||
uint32_t *S; // 4-bit packed sequence
|
||||
struct mm_idx_bucket_s *B; // index (hidden)
|
||||
@@ -91,7 +108,7 @@ typedef struct {
|
||||
int32_t mlen, blen; // seeded exact match length; seeded alignment block length
|
||||
int32_t n_sub; // number of suboptimal mappings
|
||||
int32_t score0; // initial chaining score (before chain merging/spliting)
|
||||
uint32_t mapq:8, split:2, rev:1, inv:1, sam_pri:1, proper_frag:1, pe_thru:1, seg_split:1, seg_id:8, split_inv:1, dummy:7;
|
||||
uint32_t mapq:8, split:2, rev:1, inv:1, sam_pri:1, proper_frag:1, pe_thru:1, seg_split:1, seg_id:8, split_inv:1, is_alt:1, dummy:6;
|
||||
uint32_t hash;
|
||||
float div;
|
||||
mm_extra_t *p;
|
||||
@@ -100,7 +117,7 @@ typedef struct {
|
||||
// indexing and mapping options
|
||||
typedef struct {
|
||||
short k, w, flag, bucket_bits;
|
||||
int mini_batch_size;
|
||||
int64_t mini_batch_size;
|
||||
uint64_t batch_size;
|
||||
} mm_idxopt_t;
|
||||
|
||||
@@ -111,20 +128,23 @@ typedef struct {
|
||||
|
||||
int max_qlen; // max query length
|
||||
|
||||
int bw; // bandwidth
|
||||
int bw, bw_long; // bandwidth
|
||||
int max_gap, max_gap_ref; // break a chain if there are no minimizers in a max_gap window
|
||||
int max_frag_len;
|
||||
int max_chain_skip, max_chain_iter;
|
||||
int min_cnt; // min number of minimizers on each chain
|
||||
int min_chain_score; // min chaining score
|
||||
float chain_gap_scale;
|
||||
int rmq_size_cap, rmq_inner_dist;
|
||||
int rmq_rescue_size;
|
||||
float rmq_rescue_ratio;
|
||||
|
||||
float mask_level;
|
||||
int mask_len;
|
||||
float pri_ratio;
|
||||
int best_n; // top best_n chains are subjected to DP alignment
|
||||
|
||||
int max_join_long, max_join_short;
|
||||
int min_join_flank_sc;
|
||||
float min_join_flank_ratio;
|
||||
float alt_drop;
|
||||
|
||||
int a, b, q, e, q2, e2; // matching score, mismatch, gap-open and gap-ext penalties
|
||||
int sc_ambi; // score when one or both bases are "N"
|
||||
@@ -137,14 +157,18 @@ typedef struct {
|
||||
int anchor_ext_len, anchor_ext_shift;
|
||||
float max_clip_ratio; // drop an alignment if BOTH ends are clipped above this ratio
|
||||
|
||||
int rank_min_len;
|
||||
float rank_frac;
|
||||
|
||||
int pe_ori, pe_bonus;
|
||||
|
||||
float mid_occ_frac; // only used by mm_mapopt_update(); see below
|
||||
int32_t min_mid_occ;
|
||||
int32_t min_mid_occ, max_mid_occ;
|
||||
int32_t mid_occ; // ignore seeds with occurrences above this threshold
|
||||
int32_t max_occ;
|
||||
int mini_batch_size; // size of a batch of query bases to process in parallel
|
||||
int32_t max_occ, max_max_occ, occ_dist;
|
||||
int64_t mini_batch_size; // size of a batch of query bases to process in parallel
|
||||
int64_t max_sw_mat;
|
||||
int64_t cap_kalloc;
|
||||
|
||||
const char *split_prefix;
|
||||
} mm_mapopt_t;
|
||||
@@ -261,6 +285,13 @@ mm_idx_t *mm_idx_load(FILE *fp);
|
||||
*/
|
||||
void mm_idx_dump(FILE *fp, const mm_idx_t *mi);
|
||||
|
||||
/**
|
||||
* Store hash table from minimap2 index into a file
|
||||
* @param f_name File name for output file
|
||||
* @param mi minimap2 index
|
||||
*/
|
||||
void mm_idx_dump_hash(const char* f_name, const mm_idx_t *mi);
|
||||
|
||||
/**
|
||||
* Create an index from strings in memory
|
||||
*
|
||||
@@ -289,6 +320,19 @@ void mm_idx_stat(const mm_idx_t *idx);
|
||||
* @param r minimap2 index
|
||||
*/
|
||||
void mm_idx_destroy(mm_idx_t *mi);
|
||||
/**
|
||||
* Destroy/deallocate an hash table index
|
||||
*
|
||||
* @param r minimap2 index
|
||||
*/
|
||||
void mm_idx_destroy_mm_hash(mm_idx_t *mi);
|
||||
|
||||
/**
|
||||
* Destroy/deallocate target sequences
|
||||
*
|
||||
* @param r minimap2 index
|
||||
*/
|
||||
void mm_idx_destroy_seq(mm_idx_t *mi);
|
||||
|
||||
/**
|
||||
* Initialize a thread-local buffer for mapping
|
||||
@@ -368,6 +412,7 @@ int mm_idx_index_name(mm_idx_t *mi);
|
||||
int mm_idx_name2id(const mm_idx_t *mi, const char *name);
|
||||
int mm_idx_getseq(const mm_idx_t *mi, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq);
|
||||
|
||||
int mm_idx_alt_read(mm_idx_t *mi, const char *fn);
|
||||
int mm_idx_bed_read(mm_idx_t *mi, const char *fn, int read_junc);
|
||||
int mm_idx_bed_junc(const mm_idx_t *mi, int32_t ctg, int32_t st, int32_t en, uint8_t *s);
|
||||
|
||||
|
||||
+86
-54
@@ -1,4 +1,4 @@
|
||||
.TH minimap2 1 "4 May 2019" "minimap2-2.17 (r941)" "Bioinformatics tools"
|
||||
.TH minimap2 1 "7 August 2021" "minimap2-2.22 (r1101)" "Bioinformatics tools"
|
||||
.SH NAME
|
||||
.PP
|
||||
minimap2 - mapping and alignment between collections of DNA sequences
|
||||
@@ -121,6 +121,14 @@ provided as the target sequences, options
|
||||
.BR -w ,
|
||||
.B -I
|
||||
will be effectively overridden by the options stored in the index file.
|
||||
.TP
|
||||
.BI --alt \ FILE
|
||||
List of ALT contigs [null]
|
||||
.TP
|
||||
.BI --alt-drop \ FLOAT
|
||||
Drop ALT hits by
|
||||
.I FLOAT
|
||||
fraction when ranking and computing mapping quality [0.15]
|
||||
.SS Mapping options
|
||||
.TP 10
|
||||
.BI -f \ FLOAT | INT1 [, INT2 ]
|
||||
@@ -137,22 +145,32 @@ or
|
||||
.B -xsr
|
||||
mode, which sets the threshold for a second round of seeding.
|
||||
.TP
|
||||
.BI --min-occ-floor \ INT
|
||||
Force minimap2 to always use k-mers occurring
|
||||
.BI -U \ INT1 [, INT2 ]
|
||||
Lower and upper bounds of k-mer occurrences [10,1000000]. The final k-mer occurrence threshold is
|
||||
.RI max{ INT1 ,\ min{ INT2 ,
|
||||
.BR -f }}.
|
||||
This option prevents excessively small or large
|
||||
.B -f
|
||||
estimated from the input reference. It deprecates
|
||||
.B --min-occ-floor
|
||||
in earlier versions of minimap2.
|
||||
.TP
|
||||
.BI -e \ INT
|
||||
Sample a high-frequency minimizer every
|
||||
.I INT
|
||||
times or less [0]. In effect, the max occurrence threshold is set to
|
||||
the
|
||||
.RI max{ INT ,
|
||||
.BR -f }.
|
||||
basepairs [500].
|
||||
.TP
|
||||
.BI -g \ INT
|
||||
.BI -g \ NUM
|
||||
Stop chain enlongation if there are no minimizers within
|
||||
.IR INT -bp
|
||||
[10000].
|
||||
.IR NUM -bp
|
||||
[10k].
|
||||
.TP
|
||||
.BI -r \ INT
|
||||
Bandwidth used in chaining and DP-based alignment [500]. This option
|
||||
approximately controls the maximum gap size.
|
||||
.BI -r \ NUM1 [, NUM2 ]
|
||||
Bandwidth for chaining and base alignment [500,20k].
|
||||
.I NUM1
|
||||
is used for initial chaining and alignment extension;
|
||||
.I NUM2
|
||||
for RMQ-based re-chaining and closing gaps in alignments.
|
||||
.TP
|
||||
.BI -n \ INT
|
||||
Discard chains consisting of
|
||||
@@ -226,10 +244,21 @@ Mark as secondary a chain that overlaps with a better chain by
|
||||
.I FLOAT
|
||||
or more of the shorter chain [0.5]
|
||||
.TP
|
||||
.BR --rmq = no | yes
|
||||
Use the minigraph chaining algorithm [no]. The minigraph algorithm is better
|
||||
for aligning contigs through long INDELs.
|
||||
.TP
|
||||
.B --hard-mask-level
|
||||
Honor option
|
||||
.B -M
|
||||
and disable a heurstic to save unmapped subsequences.
|
||||
and disable a heurstic to save unmapped subsequences and disables
|
||||
.BR --mask-len .
|
||||
.TP
|
||||
.BI --mask-len \ NUM
|
||||
Keep an alignment if dropping it leaves an unaligned region on query longer than
|
||||
.IR INT
|
||||
[inf]. Effective without
|
||||
.BR --hard-mask-level .
|
||||
.TP
|
||||
.BI --max-chain-skip \ INT
|
||||
A heuristics that stops chaining early [25]. Minimap2 uses dynamic programming
|
||||
@@ -245,15 +274,14 @@ Check up to
|
||||
partial chains during chaining [5000]. This is a heuristic to avoid quadratic
|
||||
time complexity in the worst case.
|
||||
.TP
|
||||
.BI --chain-gap-scale \ FLOAT
|
||||
Scale of gap cost during chaining [1.0]
|
||||
.TP
|
||||
.B --no-long-join
|
||||
Disable the long gap patching heuristic. When this option is applied, the
|
||||
maximum alignment gap is mostly controlled by
|
||||
.BR -r .
|
||||
.TP
|
||||
.BI --lj-min-ratio \ FLOAT
|
||||
Fraction of query sequence length required to bridge a long gap [0.5]. A
|
||||
smaller value helps to recover longer gaps, at the cost of more false gaps.
|
||||
.TP
|
||||
.B --splice
|
||||
Enable the splice alignment mode.
|
||||
.TP
|
||||
@@ -373,7 +401,7 @@ BED12 file can be converted from GTF/GFF3 with `paftools.js gff2bed anno.gtf'
|
||||
.BR --junc-bonus \ INT
|
||||
Score bonus for a splice donor or acceptor found in annotation (effective with
|
||||
.BR --junc-bed )
|
||||
[0].
|
||||
[9].
|
||||
.TP
|
||||
.BI --end-seed-pen \ INT
|
||||
Drop a terminal anchor if
|
||||
@@ -394,6 +422,11 @@ alignment.
|
||||
.BI --cap-sw-mem \ NUM
|
||||
Skip alignment if the DP matrix size is above
|
||||
.IR NUM .
|
||||
Set 0 to disable [100m].
|
||||
.TP
|
||||
.BI --cap-kalloc \ NUM
|
||||
Free thread-local kalloc memory reservoir if after the alignment the size of the reservoir above
|
||||
.IR NUM .
|
||||
Set 0 to disable [0].
|
||||
.SS Input/output options
|
||||
.TP 10
|
||||
@@ -505,60 +538,47 @@ Available
|
||||
.I STR
|
||||
are:
|
||||
.RS
|
||||
.TP 8
|
||||
.B map-pb
|
||||
PacBio/Oxford Nanopore read to reference mapping
|
||||
.RB ( -Hk19 )
|
||||
.TP
|
||||
.TP 10
|
||||
.B map-ont
|
||||
Slightly more sensitive for Oxford Nanopore to reference mapping
|
||||
.RB ( -k15 ).
|
||||
For PacBio reads, HPC minimizers consistently leads to faster performance and
|
||||
more sensitive results in comparison to normal minimizers. For Oxford Nanopore
|
||||
data, normal minimizers are better, though not much. The effectiveness of HPC
|
||||
is determined by the sequencing error mode.
|
||||
Align noisy long reads of ~10% error rate to a reference genome. This is the
|
||||
default mode.
|
||||
.TP
|
||||
.B map-hifi
|
||||
Align PacBio high-fidelity (HiFi) reads to a reference genome
|
||||
.RB ( -k19
|
||||
.B -w19 -U50,500 -g10k -A1 -B4 -O6,26 -E2,1
|
||||
.BR -s200 ).
|
||||
.TP
|
||||
.B map-pb
|
||||
Align older PacBio continuous long (CLR) reads to a reference genome
|
||||
.RB ( -Hk19 ).
|
||||
.TP
|
||||
.B asm5
|
||||
Long assembly to reference mapping
|
||||
.RB ( -k19
|
||||
.B -w19 -A1 -B19 -O39,81 -E3,1 -s200 -z200 -N50
|
||||
.BR --min-occ-floor=100 ).
|
||||
.B -w19 -U50,500 --rmq -r100k -g10k -A1 -B19 -O39,81 -E3,1 -s200 -z200
|
||||
.BR -N50 ).
|
||||
Typically, the alignment will not extend to regions with 5% or higher sequence
|
||||
divergence. Only use this preset if the average divergence is far below 5%.
|
||||
.TP
|
||||
.B asm10
|
||||
Long assembly to reference mapping
|
||||
.RB ( -k19
|
||||
.B -w19 -A1 -B9 -O16,41 -E2,1 -s200 -z200 -N50
|
||||
.BR --min-occ-floor=100 ).
|
||||
.B -w19 -U50,500 --rmq -r100k -g10k -A1 -B9 -O16,41 -E2,1 -s200 -z200
|
||||
.BR -N50 ).
|
||||
Up to 10% sequence divergence.
|
||||
.TP
|
||||
.B asm20
|
||||
Long assembly to reference mapping
|
||||
.RB ( -k19
|
||||
.B -w10 -A1 -B4 -O6,26 -E2,1 -s200 -z200 -N50
|
||||
.BR --min-occ-floor=100 ).
|
||||
.B -w10 -U50,500 --rmq -r100k -g10k -A1 -B4 -O6,26 -E2,1 -s200 -z200
|
||||
.BR -N50 ).
|
||||
Up to 20% sequence divergence.
|
||||
.TP
|
||||
.B ava-pb
|
||||
PacBio all-vs-all overlap mapping
|
||||
.RB ( -Hk19
|
||||
.B -Xw5 -m100 -g10000 --max-chain-skip
|
||||
.BR 25 ).
|
||||
.TP
|
||||
.B ava-ont
|
||||
Oxford Nanopore all-vs-all overlap mapping
|
||||
.RB ( -k15
|
||||
.B -Xw5 -m100 -g10000 -r2000 --max-chain-skip
|
||||
.BR 25 ).
|
||||
Similarly, the major difference from
|
||||
.B ava-pb
|
||||
is that this preset is not using HPC minimizers.
|
||||
.TP
|
||||
.B splice
|
||||
Long-read spliced alignment
|
||||
.RB ( -k15
|
||||
.B -w5 --splice -g2000 -G200k -A1 -B2 -O2,32 -E1,0 -C9 -z200 -ub --junc-bonus=9
|
||||
.B -w5 --splice -g2k -G200k -A1 -B2 -O2,32 -E1,0 -b0 -C9 -z200 -ub --junc-bonus=9 --cap-sw-mem=0
|
||||
.BR --splice-flank=yes ).
|
||||
In the splice mode, 1) long deletions are taken as introns and represented as
|
||||
the
|
||||
@@ -577,9 +597,21 @@ Long-read splice alignment for PacBio CCS reads
|
||||
.B sr
|
||||
Short single-end reads without splicing
|
||||
.RB ( -k21
|
||||
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -r50 -p.5 -N20 -f1000,5000 -n2 -m20
|
||||
.B -s40 -g200 -2K50m --heap-sort=yes
|
||||
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -b0 -r100 -p.5 -N20 -f1000,5000 -n2 -m20
|
||||
.B -s40 -g100 -2K50m --heap-sort=yes
|
||||
.BR --secondary=no ).
|
||||
.TP
|
||||
.B ava-pb
|
||||
PacBio CLR all-vs-all overlap mapping
|
||||
.RB ( -Hk19
|
||||
.B -Xw5 -e0
|
||||
.BR -m100 ).
|
||||
.TP
|
||||
.B ava-ont
|
||||
Oxford Nanopore all-vs-all overlap mapping
|
||||
.RB ( -k15
|
||||
.B -Xw5 -e0 -m100
|
||||
.BR -r2k ).
|
||||
.RE
|
||||
.SS Miscellaneous options
|
||||
.TP 10
|
||||
|
||||
@@ -159,3 +159,4 @@ KRADIX_SORT_INIT(128x, mm128_t, sort_key_128x, 8)
|
||||
KRADIX_SORT_INIT(64, uint64_t, sort_key_64, 8)
|
||||
|
||||
KSORT_INIT_GENERIC(uint32_t)
|
||||
KSORT_INIT_GENERIC(uint64_t)
|
||||
|
||||
Executable
+335
@@ -0,0 +1,335 @@
|
||||
#!/usr/bin/env k8
|
||||
|
||||
var getopt = function(args, ostr) {
|
||||
var oli; // option letter list index
|
||||
if (typeof(getopt.place) == 'undefined')
|
||||
getopt.ind = 0, getopt.arg = null, getopt.place = -1;
|
||||
if (getopt.place == -1) { // update scanning pointer
|
||||
if (getopt.ind >= args.length || args[getopt.ind].charAt(getopt.place = 0) != '-') {
|
||||
getopt.place = -1;
|
||||
return null;
|
||||
}
|
||||
if (getopt.place + 1 < args[getopt.ind].length && args[getopt.ind].charAt(++getopt.place) == '-') { // found "--"
|
||||
++getopt.ind;
|
||||
getopt.place = -1;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
var optopt = args[getopt.ind].charAt(getopt.place++); // character checked for validity
|
||||
if (optopt == ':' || (oli = ostr.indexOf(optopt)) < 0) {
|
||||
if (optopt == '-') return null; // if the user didn't specify '-' as an option, assume it means null.
|
||||
if (getopt.place < 0) ++getopt.ind;
|
||||
return '?';
|
||||
}
|
||||
if (oli+1 >= ostr.length || ostr.charAt(++oli) != ':') { // don't need argument
|
||||
getopt.arg = null;
|
||||
if (getopt.place < 0 || getopt.place >= args[getopt.ind].length) ++getopt.ind, getopt.place = -1;
|
||||
} else { // need an argument
|
||||
if (getopt.place >= 0 && getopt.place < args[getopt.ind].length)
|
||||
getopt.arg = args[getopt.ind].substr(getopt.place);
|
||||
else if (args.length <= ++getopt.ind) { // no arg
|
||||
getopt.place = -1;
|
||||
if (ostr.length > 0 && ostr.charAt(0) == ':') return ':';
|
||||
return '?';
|
||||
} else getopt.arg = args[getopt.ind]; // white space
|
||||
getopt.place = -1;
|
||||
++getopt.ind;
|
||||
}
|
||||
return optopt;
|
||||
}
|
||||
|
||||
function read_fastx(file, buf)
|
||||
{
|
||||
if (file.readline(buf) < 0) return null;
|
||||
var m, line = buf.toString();
|
||||
if ((m = /^([>@])(\S+)/.exec(line)) == null)
|
||||
throw Error("wrong fastx format");
|
||||
var is_fq = (m[1] == '@');
|
||||
var name = m[2];
|
||||
if (file.readline(buf) < 0)
|
||||
throw Error("missing sequence line");
|
||||
var seq = buf.toString();
|
||||
if (is_fq) { // skip quality
|
||||
file.readline(buf);
|
||||
file.readline(buf);
|
||||
}
|
||||
return [name, seq];
|
||||
}
|
||||
|
||||
function filter_paf(a, opt)
|
||||
{
|
||||
if (a.length == 0) return;
|
||||
var k = 0;
|
||||
for (var i = 0; i < a.length; ++i) {
|
||||
var ai = a[i];
|
||||
if (ai[10] < opt.min_blen) continue;
|
||||
if (ai[9] < ai[10] * opt.min_iden) continue;
|
||||
var clip = [0, 0];
|
||||
if (ai[4] == '+') {
|
||||
clip[0] = ai[2] < ai[7]? ai[2] : ai[7];
|
||||
clip[1] = ai[1] - ai[3] < ai[6] - ai[8]? ai[1] - ai[3] : ai[6] - ai[8];
|
||||
} else {
|
||||
clip[0] = ai[2] < ai[6] - ai[8]? ai[2] : ai[6] - ai[8];
|
||||
clip[1] = ai[1] - ai[3] < ai[7]? ai[1] - ai[3] : ai[7];
|
||||
}
|
||||
if (clip[0] > opt.max_clip_len || clip[1] > opt.max_clip_len) continue;
|
||||
a[k++] = ai;
|
||||
}
|
||||
a.length = k;
|
||||
}
|
||||
|
||||
function parse_events(t, ev, id, buf)
|
||||
{
|
||||
var re = /(:(\d+))|(([\+\-\*])([a-z]+))/g;
|
||||
var m, cs = null;
|
||||
for (var j = 12; j < t.length; ++j) {
|
||||
if ((m = /^cs:Z:(\S+)/.exec(t[j])) != null) {
|
||||
cs = m[1].toLowerCase();
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (cs == null) {
|
||||
warn("Warning: no cs tag for read '" + t[0] + "'");
|
||||
return;
|
||||
}
|
||||
var st = t[2], en = t[3];
|
||||
var x = st;
|
||||
while ((m = re.exec(cs)) != null) {
|
||||
var l;
|
||||
if (m[2] != null) { // an identitcal match ":\d+"
|
||||
l = parseInt(m[2]);
|
||||
// [start, end, type, index, changed_base]
|
||||
ev.push([x, x + l, 0, id]);
|
||||
} else {
|
||||
if (m[4] == '*') {
|
||||
l = 1;
|
||||
ev.push([x, x + 1, 1, id, m[5][0]]);
|
||||
} else if (m[4] == '+') {
|
||||
l = m[5].length;
|
||||
ev.push([x, x + l, 2, id]);
|
||||
} else if (m[4] == '-') {
|
||||
l = 0;
|
||||
ev.push([x, x, -1, id, m[5]]);
|
||||
}
|
||||
}
|
||||
x += l;
|
||||
}
|
||||
if (x != en)
|
||||
throw Error("inconsistent cs for read '" + t[0] + "'");
|
||||
}
|
||||
|
||||
function find_het_sub(ev, a, opt)
|
||||
{
|
||||
var n = a.length, last0_i = -1, h = [], d = [];
|
||||
for (var i = 0; i < n; ++i) h[i] = [], d[i] = [];
|
||||
for (var i = 0; i < ev.length; ++i) {
|
||||
if (ev[i][2] == 0) {
|
||||
if (last0_i < 0 || ev[i][0] != ev[last0_i][0]) last0_i = i;
|
||||
else if (ev[i][1] > ev[last0_i][1])
|
||||
last0_i = i;
|
||||
} else if (ev[i][2] == 1 && last0_i >= 0 && ev[i][0] < ev[last0_i][1]) {
|
||||
if (ev[last0_i][1] - ev[last0_i][0] >= opt.min_mlen) {
|
||||
if (opt.dbg_ev) print("EV", ev[last0_i].join("\t"), "|", ev[i].join("\t"));
|
||||
var e0 = ev[last0_i], hl = h[e0[3]];
|
||||
if (hl.length == 0 || hl[hl.length-1][0] != e0[0])
|
||||
hl.push([e0[0], e0[1]]);
|
||||
d[ev[i][3]].push([ev[i][0], e0[1] - e0[0]]);
|
||||
}
|
||||
}
|
||||
}
|
||||
var b = [];
|
||||
for (var i = 0; i < n; ++i) {
|
||||
var sh = 0, dh = 0;
|
||||
for (var j = 0; j < h[i].length; ++j)
|
||||
sh += h[i][j][1] - h[i][j][0];
|
||||
for (var j = 0; j < d[i].length; ++j)
|
||||
dh += d[i][j][1];
|
||||
// [start, end, index, #consistent, lenConsistent, #conflictive, lenConflictive, identity, mlen]
|
||||
b[i] = [a[i][2], a[i][3], i, h[i].length, sh, d[i].length, dh, a[i][9] / a[i][10], a[i][9]];
|
||||
}
|
||||
return b;
|
||||
}
|
||||
|
||||
function flt_utg_for_ec(b, opt)
|
||||
{
|
||||
var k = 0;
|
||||
for (var i = 0; i < b.length; ++i) {
|
||||
var bi = b[i];
|
||||
if (bi[4] == 0 && bi[6] == 0) b[k++] = bi; // entirely ambiguous
|
||||
else if (bi[6] < (bi[4] + bi[6]) * opt.max_ratio0) b[k++] = bi;
|
||||
}
|
||||
b.length = k;
|
||||
if (b.length == 0) return;
|
||||
// find the longest contiguous segment
|
||||
b.sort(function(x,y) { return x[0]-y[0] });
|
||||
var st = b[0][0], en = b[0][1], max_st = 0, max_en = 0, max_max_en = en;
|
||||
for (var i = 1; i < b.length; ++i) {
|
||||
if (b[i][0] > en) {
|
||||
if (en - st > max_en - max_st)
|
||||
max_st = st, max_en = en;
|
||||
st = b[i][0], en = b[i][1];
|
||||
} else {
|
||||
en = en > b[i][1]? en : b[i][1];
|
||||
}
|
||||
max_max_en = max_max_en > b[i][1]? max_max_en : b[i][1];
|
||||
}
|
||||
if (en - st > max_en - max_st)
|
||||
max_st = st, max_en = en;
|
||||
if (max_max_en != en || st != b[0][0]) {
|
||||
var k = 0;
|
||||
for (var i = 0; i < b.length; ++i)
|
||||
if (b[i][0] < max_en && b[i][1] > max_st)
|
||||
b[k++] = b[i];
|
||||
b.length = k;
|
||||
}
|
||||
}
|
||||
|
||||
function flt_utg_for_bin(b, opt) // filter out alignments clearly on the wrong phase
|
||||
{
|
||||
var k = 0;
|
||||
for (var i = 0; i < b.length; ++i) {
|
||||
var bi = b[i];
|
||||
if (bi[4] + bi[6] == 0 || bi[4] >= (bi[4] + bi[6]) * opt.max_ratio0) b[k++] = bi;
|
||||
}
|
||||
b.length = k;
|
||||
}
|
||||
|
||||
function ec_core(b, n_a, ev, buf, ecb) // error correction
|
||||
{
|
||||
var intv = [];
|
||||
for (var i = 0; i < n_a; ++i)
|
||||
intv[i] = null;
|
||||
intv[b[0][2]] = [b[0][0], b[0][1]];
|
||||
var en = b[0][1];
|
||||
for (var i = 1; i < b.length; ++i) {
|
||||
if (b[i][1] <= en) continue;
|
||||
intv[b[i][2]] = [en, b[i][1]];
|
||||
en = b[i][1];
|
||||
}
|
||||
var k = 0;
|
||||
ecb.capacity = buf.capacity;
|
||||
ecb.length = 0;
|
||||
for (var i = 0; i < ev.length; ++i) {
|
||||
var e = ev[i], I = intv[e[3]];
|
||||
if (I == null) continue;
|
||||
if (e[0] >= I[0] && e[0] < I[1]) { // this is to reduce duplicated events around junctions
|
||||
//print("X", e.join("\t"));
|
||||
if (e[2] == 0) {
|
||||
ecb.length += e[1] - e[0];
|
||||
for (var j = e[0]; j < e[1]; ++j)
|
||||
ecb[k++] = buf[j];
|
||||
} else if (e[2] == 1) {
|
||||
++ecb.length;
|
||||
ecb[k++] = e[4].charCodeAt(0);
|
||||
} else if (e[2] < 0) {
|
||||
ecb.length += e[4].length;
|
||||
for (var j = 0; j < e[4].length; ++j)
|
||||
ecb[k++] = e[4].charCodeAt(j);
|
||||
} // else, skip e[2] == 2
|
||||
}
|
||||
}
|
||||
if (ecb.length != k) throw Error("BUG!");
|
||||
}
|
||||
|
||||
function process_paf(a, opt, fp_seq, buf, ecb)
|
||||
{
|
||||
if (a.length == 0) return;
|
||||
var len = a[0][1], name = a[0][0], seq = null;
|
||||
if (len < opt.min_rlen) return;
|
||||
if (fp_seq) {
|
||||
var ret;
|
||||
while ((ret = read_fastx(fp_seq, buf)) != null)
|
||||
if (ret[0] == a[0][0])
|
||||
break;
|
||||
if (ret == null)
|
||||
throw Error("failed to find sequence for read '" + a[0][0] + "'");
|
||||
name = ret[0], seq = ret[1];
|
||||
if (seq.length != len)
|
||||
throw Error("inconsistent length for read '" + name + "'");
|
||||
}
|
||||
filter_paf(a, opt);
|
||||
if (a.length == 0) return;
|
||||
var ev = [];
|
||||
for (var i = 0; i < a.length; ++i)
|
||||
parse_events(a[i], ev, i, buf);
|
||||
ev.sort(function(x,y) { return x[0]!=y[0]? x[0]-y[0] : x[2]-y[2] });
|
||||
if (seq == null) print("SQ", name, a[0][1], a.length);
|
||||
var b = find_het_sub(ev, a, opt);
|
||||
if (opt.ec) flt_utg_for_ec(b, opt);
|
||||
else flt_utg_for_bin(b, opt);
|
||||
if (seq == null) {
|
||||
for (var i = 0; i < b.length; ++i) {
|
||||
var m, ai = a[b[i][2]], score = 0;
|
||||
for (var j = 10; j < ai.length; ++j)
|
||||
if ((m = /^AS:i:(\d+)/.exec(ai[j])) != null)
|
||||
score = m[1];
|
||||
print("TS", b[i][2], b[i][0], b[i][1], ai.slice(5, 9).join("\t"), b[i].slice(3, 7).join("\t"), score);
|
||||
}
|
||||
print("//");
|
||||
} else { // error correction
|
||||
if (b.length == 0) return;
|
||||
buf.set(seq, 0);
|
||||
ec_core(b, a.length, ev, buf, ecb);
|
||||
print(">" + name);
|
||||
print(ecb);
|
||||
}
|
||||
}
|
||||
|
||||
function main(args)
|
||||
{
|
||||
var c, opt = { min_rlen:5000, min_blen:5000, min_iden:0.8, min_mlen:5, max_clip_len:500, max_ratio0:0.25, dbg_ev:false };
|
||||
while ((c = getopt(args, "l:b:d:m:c:r:E")) != null) {
|
||||
if (c == 'l') opt.min_rlen = parseInt(getopt.arg);
|
||||
else if (c == 'b') opt.min_blen = parseInt(getopt.arg);
|
||||
else if (c == 'd') opt.min_iden = parseFloat(getopt.arg);
|
||||
else if (c == 'm') opt.min_slen = parseInt(getopt.arg);
|
||||
else if (c == 'c') opt.max_clip_len = parseInt(getopt.arg);
|
||||
else if (c == 'r') opt.max_ratio0 = parseFloat(getopt.arg);
|
||||
else if (c == 'E') opt.dbg_ev = true;
|
||||
}
|
||||
if (args.length - getopt.ind < 1) {
|
||||
print("Usage: mmphase.js [options] <map-with-cs.paf> [reads.fa]");
|
||||
print("Options:");
|
||||
print(" -l INT min read length [" + opt.min_rlen + "]");
|
||||
print(" -b INT min alignment length [" + opt.min_blen + "]");
|
||||
print(" -d FLOAT min identity [" + opt.min_iden + "]");
|
||||
print(" -s INT min match length [" + opt.min_mlen + "]");
|
||||
print(" -c INT max clip length [" + opt.max_clip_len + "]");
|
||||
print(" -r FLOAT initial ratio for haplotype filtering [" + opt.max_ratio0 + "]");
|
||||
return 0;
|
||||
}
|
||||
|
||||
opt.ec = args.length - getopt.ind < 2? false : true;
|
||||
if (!opt.ec) {
|
||||
print("CC");
|
||||
print("CC", "SQ qName qLen nHits");
|
||||
print("CC", "TS index qStart qEnd tName tLen tStart tEnd nConsistent lCons nConflictive lConf score");
|
||||
print("CC");
|
||||
}
|
||||
|
||||
var buf = new Bytes(), ecb = new Bytes();
|
||||
var fp_paf = new File(args[getopt.ind]);
|
||||
var fp_seq = args.length - getopt.ind >= 2? new File(args[getopt.ind+1]) : null;
|
||||
var a = [];
|
||||
while (fp_paf.readline(buf) >= 0) {
|
||||
var t = buf.toString().split("\t");
|
||||
if (a.length > 0 && a[0][0] != t[0]) {
|
||||
process_paf(a, opt, fp_seq, buf, ecb);
|
||||
a.length = 0;
|
||||
}
|
||||
for (var i = 1; i <= 3; ++i) t[i] = parseInt(t[i]);
|
||||
if (t[1] < opt.min_rlen) continue;
|
||||
for (var i = 6; i <= 10; ++i) t[i] = parseInt(t[i]);
|
||||
if (t[10] < opt.min_blen) continue;
|
||||
a.push(t);
|
||||
}
|
||||
if (a.length >= 0)
|
||||
process_paf(a, opt, fp_seq, buf, ecb);
|
||||
if (fp_seq) fp_seq.close();
|
||||
fp_paf.close();
|
||||
ecb.destroy();
|
||||
buf.destroy();
|
||||
}
|
||||
|
||||
var ret = main(arguments)
|
||||
exit(ret)
|
||||
+628
-38
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env k8
|
||||
|
||||
var paftools_version = '2.17-r949-dirty';
|
||||
var paftools_version = '2.22-r1101';
|
||||
|
||||
/*****************************
|
||||
***** Library functions *****
|
||||
@@ -640,6 +640,23 @@ function paf_asmstat(args)
|
||||
}
|
||||
}
|
||||
|
||||
function AUN(lens, tot) {
|
||||
lens.sort(function(a,b) { return b - a; });
|
||||
if (tot == null) {
|
||||
tot = 0;
|
||||
for (var k = 0; k < lens.length; ++k)
|
||||
tot += lens[k];
|
||||
}
|
||||
var x = 0, y = 0;
|
||||
for (var k = 0; k < lens.length; ++k) {
|
||||
var l = x + lens[k] <= tot? lens[k] : tot - x;
|
||||
x += lens[k];
|
||||
y += l * (l / tot);
|
||||
if (x >= tot) break;
|
||||
}
|
||||
return y.toFixed(0);
|
||||
}
|
||||
|
||||
function count_bp(bp, min_blen, min_gap) {
|
||||
var n_bp = 0;
|
||||
for (var k = 0; k < bp.length; ++k)
|
||||
@@ -660,7 +677,7 @@ function paf_asmstat(args)
|
||||
return (NM - n_gaps + n_gapo) / (n_M + n_gapo);
|
||||
}
|
||||
|
||||
var labels = ['Length', 'l_cov', 'Rcov', 'Rdup', 'Qcov', 'NG75', 'NG50', 'NGA50', '#breaks', 'bp(' + min_seg_len + ',0)', 'bp(' + min_seg_len + ',10k)'];
|
||||
var labels = ['Length', 'l_cov', 'Rcov', 'Rdup', 'Qcov', 'NG75', 'NG50', 'NGA50', 'AUNGA', '#breaks', 'bp(' + min_seg_len + ',0)', 'bp(' + min_seg_len + ',10k)'];
|
||||
var rst = [];
|
||||
for (var i = 0; i < labels.length; ++i)
|
||||
rst[i] = [];
|
||||
@@ -688,11 +705,9 @@ function paf_asmstat(args)
|
||||
qinfo[t[0]].bp = [];
|
||||
if (t.length < 9 || t[5] == "*") continue;
|
||||
if (!/\ttp:A:[PI]/.test(line)) continue;
|
||||
if ((m = /\tcg:Z:(\S+)/.exec(line)) == null) continue;
|
||||
var cigar = m[1];
|
||||
if ((m = /\tNM:i:(\d+)/.exec(line)) == null) continue;
|
||||
var NM = parseInt(m[1]);
|
||||
var diff = compute_diff(cigar, NM);
|
||||
var cigar = (m = /\tcg:Z:(\S+)/.exec(line)) != null? m[1] : null;
|
||||
var NM = (m = /\tNM:i:(\d+)/.exec(line)) != null? parseInt(m[1]) : null;
|
||||
var diff = cigar != null && NM != null? compute_diff(cigar, NM) : 0;
|
||||
t[2] = parseInt(t[2]);
|
||||
t[3] = parseInt(t[3]);
|
||||
t[7] = parseInt(t[7]);
|
||||
@@ -765,10 +780,13 @@ function paf_asmstat(args)
|
||||
// compute NGA50
|
||||
rst[7][i] = N50(qblock_len, ref_len, 0.5);
|
||||
|
||||
// compute AUNGA
|
||||
rst[8][i] = AUN(qblock_len, ref_len);
|
||||
|
||||
// compute break points
|
||||
rst[8][i] = n_breaks;
|
||||
rst[9][i] = count_bp(bp, 500, 0);
|
||||
rst[10][i] = count_bp(bp, 500, 10000);
|
||||
rst[9][i] = n_breaks;
|
||||
rst[10][i] = count_bp(bp, 500, 0);
|
||||
rst[11][i] = count_bp(bp, 500, 10000);
|
||||
|
||||
// nb-plot; NOT USED
|
||||
/*
|
||||
@@ -887,23 +905,24 @@ function paf_asmgene(args)
|
||||
gene_nr[gene_list[last][0]] = 1;
|
||||
|
||||
// count and print
|
||||
var col1 = ["full_sgl", "full_dup", "frag", "part50+", "part10+", "part10-"];
|
||||
var col1 = ["full_sgl", "full_dup", "frag", "part50+", "part10+", "part10-", "dup_cnt", "dup_sum"];
|
||||
var rst = [];
|
||||
for (var k = 0; k < col1.length; ++k) {
|
||||
rst[k] = [];
|
||||
for (var i = 0; i < n_fn; ++i)
|
||||
rst[k][i] = 0;
|
||||
}
|
||||
for (var g in gene) {
|
||||
for (var g in gene) { // count single-copy genes
|
||||
if (gene[g][0] == null || gene[g][0][0] != 1) continue;
|
||||
if (gene_nr[g] == null) continue;
|
||||
if (auto_only && /^(chr)?[XY]$/.test(refpos[g][2])) continue;
|
||||
for (var i = 0; i < n_fn; ++i) {
|
||||
if (gene[g][i] == null) {
|
||||
rst[4][i]++;
|
||||
rst[5][i]++;
|
||||
if (print_err) print('M', header[i], refpos[g].join("\t"));
|
||||
} else if (gene[g][i][0] == 1) rst[0][i]++;
|
||||
else if (gene[g][i][0] > 1) {
|
||||
} else if (gene[g][i][0] == 1) {
|
||||
rst[0][i]++;
|
||||
} else if (gene[g][i][0] > 1) {
|
||||
rst[1][i]++;
|
||||
if (print_err) print('D', header[i], refpos[g].join("\t"));
|
||||
} else if (gene[g][i][1] >= opt.min_cov) {
|
||||
@@ -921,6 +940,19 @@ function paf_asmgene(args)
|
||||
}
|
||||
}
|
||||
}
|
||||
for (var g in gene) { // count multi-copy genes
|
||||
if (gene[g][0] == null || gene[g][0][0] <= 1) continue;
|
||||
if (gene_nr[g] == null) continue;
|
||||
if (auto_only && /^(chr)?[XY]$/.test(refpos[g][2])) continue;
|
||||
for (var i = 0; i < n_fn; ++i) {
|
||||
if (gene[g][i] != null) rst[7][i] += gene[g][i][0];
|
||||
if (gene[g][i] != null && gene[g][i][0] > 1) {
|
||||
rst[6][i]++;
|
||||
} else if (print_err) {
|
||||
print('d', header[i], gene[g][0][0], refpos[g].join("\t"));
|
||||
}
|
||||
}
|
||||
}
|
||||
print('H', 'Metric', header.join("\t"));
|
||||
for (var k = 0; k < rst.length; ++k) {
|
||||
print('X', col1[k], rst[k].join("\t"));
|
||||
@@ -930,12 +962,13 @@ function paf_asmgene(args)
|
||||
|
||||
function paf_stat(args)
|
||||
{
|
||||
var c, gap_out_len = null;
|
||||
while ((c = getopt(args, "l:")) != null)
|
||||
var c, gap_out_len = null, count_err = false;
|
||||
while ((c = getopt(args, "cl:")) != null)
|
||||
if (c == 'l') gap_out_len = parseInt(getopt.arg);
|
||||
else if (c == 'c') count_err = true;
|
||||
|
||||
if (getopt.ind == args.length) {
|
||||
print("Usage: paftools.js stat [-l gapOutLen] <in.sam>|<in.paf>");
|
||||
print("Usage: paftools.js stat [-c] [-l gapOutLen] <in.sam>|<in.paf>");
|
||||
exit(1);
|
||||
}
|
||||
|
||||
@@ -944,7 +977,7 @@ function paf_stat(args)
|
||||
var re = /(\d+)([MIDSHNX=])/g;
|
||||
|
||||
var lineno = 0, n_pri = 0, n_2nd = 0, n_seq = 0, n_cigar_64k = 0, l_tot = 0, l_cov = 0;
|
||||
var n_gap = [[0, 0, 0, 0, 0, 0], [0, 0, 0, 0, 0, 0]];
|
||||
var n_gap = [[0, 0, 0, 0, 0, 0], [0, 0, 0, 0, 0, 0]], n_sub = 0;
|
||||
|
||||
function cov_len(regs)
|
||||
{
|
||||
@@ -966,7 +999,7 @@ function paf_stat(args)
|
||||
if (line.charAt(0) != '@') {
|
||||
var t = line.split("\t", 12);
|
||||
var m, rs, cigar = null, is_pri = false, is_sam = false, is_rev = false, tname = null;
|
||||
var atlen = null, aqlen, qs, qe, mapq, ori_qlen;
|
||||
var atlen = null, aqlen, qs, qe, mapq, ori_qlen, NM = null, nn = 0;
|
||||
if (t.length < 2) continue;
|
||||
if (t[4] == '+' || t[4] == '-' || t[4] == '*') { // PAF
|
||||
if (t[4] == '*') continue; // unmapped
|
||||
@@ -974,6 +1007,10 @@ function paf_stat(args)
|
||||
++n_2nd;
|
||||
continue;
|
||||
}
|
||||
if ((m = /\tNM:i:(\d+)/.exec(line)) != null)
|
||||
NM = parseInt(m[1]);
|
||||
if ((m = /\tnn:i:(\d+)/.exec(line)) != null)
|
||||
nn = parseInt(m[1]);
|
||||
if ((m = /\tcg:Z:(\S+)/.exec(line)) != null)
|
||||
cigar = m[1];
|
||||
if (cigar == null) {
|
||||
@@ -995,6 +1032,10 @@ function paf_stat(args)
|
||||
++n_2nd;
|
||||
continue;
|
||||
}
|
||||
if ((m = /\tNM:i:(\d+)/.exec(line)) != null)
|
||||
NM = parseInt(m[1]);
|
||||
if ((m = /\tnn:i:(\d+)/.exec(line)) != null)
|
||||
nn = parseInt(m[1]);
|
||||
cigar = t[5];
|
||||
tname = t[2];
|
||||
rs = parseInt(t[3]) - 1;
|
||||
@@ -1013,11 +1054,13 @@ function paf_stat(args)
|
||||
++n_seq, last = t[0];
|
||||
}
|
||||
var M = 0, tl = 0, ql = 0, clip = [0, 0], n_cigar = 0, sclip = 0;
|
||||
var n_gapo = 0, n_gap_all = 0, l_match = 0;
|
||||
while ((m = re.exec(cigar)) != null) {
|
||||
var l = parseInt(m[1]);
|
||||
++n_cigar;
|
||||
if (m[2] == 'M' || m[2] == '=' || m[2] == 'X') {
|
||||
tl += l, ql += l, M += l;
|
||||
l_match += l;
|
||||
} else if (m[2] == 'I' || m[2] == 'D') {
|
||||
var type;
|
||||
if (l < 50) type = 0;
|
||||
@@ -1030,6 +1073,7 @@ function paf_stat(args)
|
||||
else tl += l, ++n_gap[1][type];
|
||||
if (gap_out_len != null && l >= gap_out_len)
|
||||
print(t[0], ql, is_rev? '-' : '+', tname, rs + tl, m[2], l);
|
||||
++n_gapo, n_gap_all += l;
|
||||
} else if (m[2] == 'N') {
|
||||
tl += l;
|
||||
} else if (m[2] == 'S') {
|
||||
@@ -1038,6 +1082,12 @@ function paf_stat(args)
|
||||
clip[M == 0? 0 : 1] = l;
|
||||
}
|
||||
}
|
||||
if (NM != null) {
|
||||
var tmp = NM - n_gap_all - nn;
|
||||
if (tmp < 0 && nn == 0) warn("WARNING: NM is smaller than the number of gaps at line " + lineno + ": NM=" + NM + ", nn=" + nn + ", G=" + n_gap_all);
|
||||
if (tmp < 0) tmp = 0;
|
||||
n_sub += tmp;
|
||||
}
|
||||
if (n_cigar > 65535) ++n_cigar_64k;
|
||||
if (ql + sclip != aqlen)
|
||||
warn("WARNING: aligned query length is inconsistent with CIGAR at line " + lineno + " (" + (ql+sclip) + " != " + aqlen + ")");
|
||||
@@ -1047,6 +1097,12 @@ function paf_stat(args)
|
||||
qs = clip[is_rev? 1 : 0], qe = qs + ql;
|
||||
ori_qlen = clip[0] + ql + clip[1];
|
||||
}
|
||||
if (count_err && NM != null) {
|
||||
var n_mm = NM - n_gap_all;
|
||||
if (n_mm < 0) warn("WARNING: NM is smaller than the number of gaps at line " + lineno);
|
||||
if (n_mm < 0) n_mm = 0;
|
||||
print(t[0], ori_qlen, t[11], ori_qlen - (qe - qs), NM, l_match + n_gap_all, n_mm + n_gapo, l_match + n_gapo);
|
||||
}
|
||||
regs.push([qs, qe]);
|
||||
last_qlen = ori_qlen;
|
||||
}
|
||||
@@ -1059,13 +1115,14 @@ function paf_stat(args)
|
||||
file.close();
|
||||
buf.destroy();
|
||||
|
||||
if (gap_out_len == null) {
|
||||
if (gap_out_len == null && !count_err) {
|
||||
print("Number of mapped sequences: " + n_seq);
|
||||
print("Number of primary alignments: " + n_pri);
|
||||
print("Number of secondary alignments: " + n_2nd);
|
||||
print("Number of primary alignments with >65535 CIGAR operations: " + n_cigar_64k);
|
||||
print("Number of bases in mapped sequences: " + l_tot);
|
||||
print("Number of mapped bases: " + l_cov);
|
||||
print("Number of substitutions: " + n_sub);
|
||||
print("Number of insertions in [0,50): " + n_gap[0][0]);
|
||||
print("Number of insertions in [50,100): " + n_gap[0][1]);
|
||||
print("Number of insertions in [100,300): " + n_gap[0][2]);
|
||||
@@ -1373,7 +1430,7 @@ function paf_view(args)
|
||||
|
||||
var s_ref = new Bytes(), s_qry = new Bytes(), s_mid = new Bytes(); // these are used to show padded alignment
|
||||
var re_cs = /([:=\-\+\*])(\d+|[A-Za-z]+)/g;
|
||||
var re_cg = /(\d+)([MIDNSH])/g;
|
||||
var re_cg = /(\d+)([MIDNSHP=X])/g;
|
||||
|
||||
var buf = new Bytes();
|
||||
var file = args[getopt.ind] == "-"? new File() : new File(args[getopt.ind]);
|
||||
@@ -1429,8 +1486,14 @@ function paf_view(args)
|
||||
warn("WARNING: converting to BLAST-like alignment requires the 'cs' tag, which is absent on line " + lineno);
|
||||
continue;
|
||||
}
|
||||
var n_mm = 0, n_oi = 0, n_od = 0, n_ei = 0, n_ed = 0;
|
||||
while ((m = re_cs.exec(cs)) != null) {
|
||||
if (m[1] == '*') ++n_mm;
|
||||
else if (m[1] == '+') ++n_oi, n_ei += m[2].length;
|
||||
else if (m[1] == '-') ++n_od, n_ed += m[2].length;
|
||||
}
|
||||
line = line.replace(/\tc[sg]:Z:\S+/g, ""); // get rid of cs or cg tags
|
||||
print('>' + line);
|
||||
print('>' + line + "\tmm:i:"+n_mm + "\toi:i:"+n_oi + "\tei:i:"+n_ei + "\tod:i:"+n_od + "\ted:i:"+n_ed);
|
||||
var rs = parseInt(t[7]), qs = t[4] == '+'? parseInt(t[2]) : parseInt(t[3]);
|
||||
var n_blocks = 0;
|
||||
while ((m = re_cs.exec(cs)) != null) {
|
||||
@@ -1853,7 +1916,7 @@ function paf_splice2bed(args)
|
||||
a.length = 0;
|
||||
}
|
||||
|
||||
var re = /(\d+)([MIDNSH])/g;
|
||||
var re = /(\d+)([MIDNSHP=X])/g;
|
||||
var c, fmt = "bed", fn_name_conv = null, keep_multi = false;
|
||||
while ((c = getopt(args, "f:n:m")) != null) {
|
||||
if (c == 'f') fmt = getopt.arg;
|
||||
@@ -2323,28 +2386,43 @@ function paf_junceval(args)
|
||||
|
||||
file = getopt.ind+1 >= args.length || args[getopt.ind+1] == '-'? new File() : new File(args[getopt.ind+1]);
|
||||
var last_qname = null;
|
||||
var re_cigar = /(\d+)([MIDNSHX=])/g;
|
||||
var re_cigar = /(\d+)([MIDNSHP=X])/g;
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, t = buf.toString().split("\t");
|
||||
var ctg_name = null, cigar = null, pos = null, qname = t[0];
|
||||
|
||||
if (t[0].charAt(0) == '@') continue;
|
||||
if (chr_only && !/^(chr)?([0-9]+|X|Y)$/.test(t[2])) continue;
|
||||
var flag = parseInt(t[1]);
|
||||
if (flag&0x100) continue;
|
||||
if (first_only && last_qname == t[0]) continue;
|
||||
if (t[2] == '*') {
|
||||
if (t[4] == '+' || t[4] == '-' || t[4] == '*') { // PAF
|
||||
ctg_name = t[5], pos = parseInt(t[7]);
|
||||
var type = 'P';
|
||||
for (i = 12; i < t.length; ++i) {
|
||||
if ((m = /^(tp:A|cg:Z):(\S+)/.exec(t[i])) != null) {
|
||||
if (m[1] == 'tp:A') type = m[2];
|
||||
else cigar = m[2];
|
||||
}
|
||||
}
|
||||
if (type == 'S') continue; // secondary
|
||||
} else { // SAM
|
||||
ctg_name = t[2], pos = parseInt(t[3]) - 1, cigar = t[5];
|
||||
var flag = parseInt(t[1]);
|
||||
if (flag&0x100) continue; // secondary
|
||||
}
|
||||
|
||||
if (chr_only && !/^(chr)?([0-9]+|X|Y)$/.test(ctg_name)) continue;
|
||||
if (first_only && last_qname == qname) continue;
|
||||
if (ctg_name == '*') { // unmapped
|
||||
++n_unmapped;
|
||||
continue;
|
||||
} else {
|
||||
++n_pri;
|
||||
if (last_qname != t[0]) {
|
||||
if (last_qname != qname) {
|
||||
++n_mapped;
|
||||
last_qname = t[0];
|
||||
last_qname = qname;
|
||||
}
|
||||
}
|
||||
|
||||
var pos = parseInt(t[3]) - 1, intron = [];
|
||||
while ((m = re_cigar.exec(t[5])) != null) {
|
||||
var intron = [];
|
||||
while ((m = re_cigar.exec(cigar)) != null) {
|
||||
var len = parseInt(m[1]), op = m[2];
|
||||
if (op == 'N') {
|
||||
intron.push([pos, pos + len]);
|
||||
@@ -2357,7 +2435,7 @@ function paf_junceval(args)
|
||||
}
|
||||
n_splice += intron.length;
|
||||
|
||||
var chr = anno[t[2]];
|
||||
var chr = anno[ctg_name];
|
||||
if (chr != null) {
|
||||
for (var i = 0; i < intron.length; ++i) {
|
||||
var o = Interval.find_ovlp(chr, intron[i][0], intron[i][1]);
|
||||
@@ -2381,12 +2459,12 @@ function paf_junceval(args)
|
||||
x += '(' + o[j][0] + "," + o[j][1] + ')';
|
||||
}
|
||||
x += ']';
|
||||
print(type, t[0], i+1, t[2], intron[i][0], intron[i][1], x);
|
||||
print(type, qname, i+1, ctg_name, intron[i][0], intron[i][1], x);
|
||||
}
|
||||
} else {
|
||||
++n_splice_novel;
|
||||
if (print_ovlp)
|
||||
print('N', t[0], i+1, t[2], intron[i][0], intron[i][1]);
|
||||
print('N', qname, i+1, ctg_name, intron[i][0], intron[i][1]);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
@@ -2480,6 +2558,509 @@ function paf_ov_eval(args)
|
||||
print((100 * (1 - n_missing / n_ovlp)).toFixed(2) + "% sensitivity");
|
||||
}
|
||||
|
||||
function paf_vcfstat(args)
|
||||
{
|
||||
var c, ts = { "AG":1, "GA":1, "CT":1, "TC":1 };
|
||||
while ((c = getopt(args, "")) != null) {
|
||||
}
|
||||
var buf = new Bytes();
|
||||
var file = args.length == getopt.ind? new File() : new File(args[getopt.ind]);
|
||||
var x = { sub:0, ts:0, tv:0, ins:0, del:0, ins1:0, del1:0, ins2:0, del2:0, ins50:0, del50:0, ins1k:0, del1k:0, ins7k:0, del7k:0, insinf:0, delinf:0 };
|
||||
while (file.readline(buf) >= 0) {
|
||||
var t = buf.toString().split("\t");
|
||||
if (t[0][0] == '#') continue;
|
||||
var alt = t[4].split(",");
|
||||
var ref = t[3];
|
||||
for (var i = 0; i < alt.length; ++i) {
|
||||
var a = alt[i];
|
||||
if (a[0] == '<' || a[1] == '>') continue;
|
||||
var l = ref.length < a.length? ref.length : a.length;
|
||||
for (var j = 0; j < l; ++j) {
|
||||
if (ref[j] != a[j]) {
|
||||
++x.sub;
|
||||
if (ts[ref[j] + a[j]]) ++x.ts;
|
||||
else ++x.tv;
|
||||
}
|
||||
}
|
||||
var d = a.length - ref.length;
|
||||
if (d > 0) {
|
||||
++x.ins;
|
||||
if (d == 1) ++x.ins1;
|
||||
else if (d == 2) ++x.ins2;
|
||||
else if (d < 50) ++x.ins50;
|
||||
else if (d < 1000) ++x.ins1k;
|
||||
else if (d < 7000) ++x.ins7k;
|
||||
else ++x.insinf;
|
||||
} else if (d < 0) {
|
||||
d = -d;
|
||||
++x.del;
|
||||
if (d == 1) ++x.del1;
|
||||
else if (d == 2) ++x.del2;
|
||||
else if (d < 50) ++x.del50;
|
||||
else if (d < 1000) ++x.del1k;
|
||||
else if (d < 7000) ++x.del7k;
|
||||
else ++x.delinf;
|
||||
}
|
||||
}
|
||||
}
|
||||
file.close();
|
||||
buf.destroy();
|
||||
print("# substitutions: " + x.sub);
|
||||
print("ts/tv: " + (x.ts / x.tv).toFixed(3));
|
||||
print("# insertions: " + x.ins);
|
||||
print("# 1bp insertions: " + x.ins1);
|
||||
print("# 2bp insertions: " + x.ins2);
|
||||
print("# [3,50) insertions: " + x.ins50);
|
||||
print("# [50,1000) insertions: " + x.ins1k);
|
||||
print("# [1000,7000) insertions: " + x.ins7k);
|
||||
print("# >=7000 insertions: " + x.insinf);
|
||||
print("# deletions: " + x.del);
|
||||
print("# 1bp deletions: " + x.del1);
|
||||
print("# 2bp deletions: " + x.del2);
|
||||
print("# [3,50) deletions: " + x.del50);
|
||||
print("# [50,1000) deletions: " + x.del1k);
|
||||
print("# [1000,7000) deletions: " + x.del7k);
|
||||
print("# >=7000 deletions: " + x.delinf);
|
||||
}
|
||||
|
||||
function paf_parseNum(s) {
|
||||
var m, x = null;
|
||||
if ((m = /^(\d*\.?\d*)([mMgGkK]?)/.exec(s)) != null) {
|
||||
x = parseFloat(m[1]);
|
||||
if (m[2] == 'k' || m[2] == 'K') x *= 1000;
|
||||
else if (m[2] == 'm' || m[2] == 'M') x *= 1000000;
|
||||
else if (m[2] == 'g' || m[2] == 'G') x *= 1000000000;
|
||||
}
|
||||
return Math.floor(x + .499);
|
||||
}
|
||||
|
||||
function paf_misjoin(args)
|
||||
{
|
||||
var c, min_seg_len = 1000000, max_gap = 1000000, fn_cen = null, show_long = false, show_err = false, cen_ratio = 0.5;
|
||||
var n_diff = [0, 0], n_gap = [0, 0], n_inv = [0, 0], n_inv_end = [0, 0];
|
||||
while ((c = getopt(args, "l:g:c:per:")) != null) {
|
||||
if (c == 'l') min_seg_len = paf_parseNum(getopt.arg);
|
||||
else if (c == 'g') max_gap = paf_parseNum(getopt.arg);
|
||||
else if (c == 'c') fn_cen = getopt.arg;
|
||||
else if (c == 'r') cen_ratio = parseFloat(getopt.arg);
|
||||
else if (c == 'p') show_long = true;
|
||||
else if (c == 'e') show_err = true;
|
||||
}
|
||||
if (args.length == getopt.ind) {
|
||||
print("Usage: paftools.js misjoin [options] <in.paf>");
|
||||
print("Options:");
|
||||
print(" -c FILE BED for centromeres []");
|
||||
print(" -r FLOAT count a centromeric event if overlap ratio > FLOAT [" + cen_ratio + "]");
|
||||
print(" -l NUM min alignment block length [1m]");
|
||||
print(" -g NUM max gap size [1m]");
|
||||
print(" -e output misjoins not involving centromeres");
|
||||
print(" -p output long alignment blocks for debugging");
|
||||
return;
|
||||
}
|
||||
var cen = {};
|
||||
var file, buf = new Bytes();
|
||||
if (fn_cen != null) {
|
||||
file = new File(fn_cen);
|
||||
while (file.readline(buf) >= 0) {
|
||||
var t = buf.toString().split("\t");
|
||||
if (cen[t[0]] == null) cen[t[0]] = [];
|
||||
cen[t[0]].push([parseInt(t[1]), parseInt(t[2])]);
|
||||
}
|
||||
file.close();
|
||||
}
|
||||
|
||||
function test_cen(cen, chr, st, en) {
|
||||
var b = cen[chr], len = 0;
|
||||
if (b == null) return false;
|
||||
for (var j = 0; j < b.length; ++j)
|
||||
if (b[j][0] < en && b[j][1] > st) {
|
||||
var s = b[j][0] > st? b[j][0] : st;
|
||||
var e = b[j][1] < en? b[j][1] : en;
|
||||
len += e - s;
|
||||
}
|
||||
return len < (en - st) * cen_ratio? false : true;
|
||||
}
|
||||
|
||||
function process(a) {
|
||||
var k = 0;
|
||||
for (var i = 0; i < a.length; ++i) {
|
||||
for (var j = 1; j <= 3; ++j) a[i][j] = parseInt(a[i][j]);
|
||||
for (var j = 6; j <= 11; ++j) a[i][j] = parseInt(a[i][j]);
|
||||
if (a[i][10] >= min_seg_len) a[k++] = a[i];
|
||||
}
|
||||
a.length = k;
|
||||
if (a.length == 1) return;
|
||||
a = a.sort(function(x,y){return x[2]-y[2]});
|
||||
if (show_long) for (var i = 0; i < a.length; ++i) print(a[i].join("\t"));
|
||||
for (var i = 1; i < a.length; ++i) {
|
||||
var ov = [false, false];
|
||||
ov[0] = test_cen(cen, a[i-1][5], a[i-1][7], a[i-1][8]);
|
||||
ov[1] = test_cen(cen, a[i][5], a[i][7], a[i][8]);
|
||||
if (a[i-1][5] != a[i][5]) { // different chr
|
||||
if (ov[0] || ov[1]) ++n_diff[1];
|
||||
else if (show_err) {
|
||||
print("J", a[i-1].slice(0, 12).join("\t"));
|
||||
print("J", a[i].slice(0, 12).join("\t"));
|
||||
}
|
||||
++n_diff[0];
|
||||
} else if (a[i-1][4] == a[i][4]) { // a gap
|
||||
var dq = a[i][2] - a[i-1][3];
|
||||
var dr = a[i][4] == '+'? a[i][7] - a[i-1][8] : a[i-1][7] - a[i][8];
|
||||
var gap = dr > dq? dr - dq : dq - dr;
|
||||
if (gap > max_gap) {
|
||||
if (ov[0] || ov[1]) ++n_gap[1];
|
||||
else if (show_err) {
|
||||
print("G", a[i-1].slice(0, 12).join("\t"));
|
||||
print("G", a[i].slice(0, 12).join("\t"));
|
||||
}
|
||||
++n_gap[0];
|
||||
}
|
||||
} else if (i + 1 < a.length && a[i+1][4] == a[i-1][4]) { // bracketed inversion
|
||||
if (ov[0] || ov[1]) ++n_inv[1];
|
||||
else if (show_err) {
|
||||
print("M", a[i-1].slice(0, 12).join("\t"));
|
||||
print("M", a[i].slice(0, 12).join("\t"));
|
||||
print("M", a[i+1].slice(0, 12).join("\t"));
|
||||
}
|
||||
++n_inv[0];
|
||||
++i;
|
||||
} else { // hanging inversion
|
||||
if (ov[0] || ov[1]) ++n_inv_end[1];
|
||||
++n_inv_end[0];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
file = args[getopt.ind] == "-"? new File() : new File(args[getopt.ind]);
|
||||
var a = [];
|
||||
while (file.readline(buf) >= 0) {
|
||||
var t = buf.toString().split("\t");
|
||||
if (a.length > 0 && a[0][0] != t[0]) {
|
||||
process(a);
|
||||
a.length = 0;
|
||||
}
|
||||
a.push(t);
|
||||
}
|
||||
if (a.length > 0) process(a);
|
||||
file.close();
|
||||
buf.destroy();
|
||||
print("# inter-chromosomal misjoins: " + n_diff.join(","));
|
||||
print("# intra-chromosomal gaps: " + n_gap.join(","));
|
||||
print("# candidate inversions in the middle: " + n_inv.join(","));
|
||||
print("# candidate inversions at contig ends: " + n_inv_end.join(","));
|
||||
}
|
||||
|
||||
function _paf_get_alen(t)
|
||||
{
|
||||
var svlen = null, alen = null;
|
||||
if ((m = /(^|;)SVLEN=(-?\d+)/.exec(t[7])) != null)
|
||||
svlen = parseInt(m[2]);
|
||||
var s = t[4].split(",");
|
||||
var min_abs_diff = 1<<30, max_abs_diff = 0;
|
||||
if (svlen != null && svlen != 0)
|
||||
alen = svlen, min_abs_diff = max_abs_diff = svlen > 0? svlen : -svlen;
|
||||
var rlen = t[3].length;
|
||||
for (var i = 0; i < s.length; ++i) {
|
||||
if (/^<\S+>$/.test(s[i])) continue;
|
||||
var diff = s[i].length - rlen;
|
||||
var abs_diff = diff > 0? diff : -diff;
|
||||
min_abs_diff = min_abs_diff < abs_diff? min_abs_diff : abs_diff;
|
||||
if (max_abs_diff < abs_diff)
|
||||
max_abs_diff = abs_diff, alen = diff;
|
||||
}
|
||||
return [alen, min_abs_diff, max_abs_diff];
|
||||
}
|
||||
|
||||
function paf_sveval(args)
|
||||
{
|
||||
var c, min_flt = 30, min_size = 50, max_size = 100000, win_size = 500, print_err = false, print_match = false, bed_fn = null;
|
||||
var len_diff_ratio = 0.5;
|
||||
while ((c = getopt(args, "f:i:x:w:er:pd:")) != null) {
|
||||
if (c == 'f') min_flt = paf_parseNum(getopt.arg);
|
||||
else if (c == 'i') min_size = paf_parseNum(getopt.arg);
|
||||
else if (c == 'x') max_size = paf_parseNum(getopt.arg);
|
||||
else if (c == 'w') win_size = paf_parseNum(getopt.arg);
|
||||
else if (c == 'd') len_diff_ratio = parseFloat(getopt.arg);
|
||||
else if (c == 'r') bed_fn = getopt.arg;
|
||||
else if (c == 'e') print_err = true;
|
||||
else if (c == 'p') print_match = true;
|
||||
}
|
||||
if (args.length - getopt.ind < 2) {
|
||||
print("Usage: paftools.js sveval [options] <base.vcf> <call.vcf>");
|
||||
print("Options:");
|
||||
print(" -r FILE confident region in BED []");
|
||||
print(" -f INT min length to discard [" + min_flt + "]");
|
||||
print(" -i INT min SV length [" + min_size + "]");
|
||||
print(" -x INT max SV length [" + max_size + "]");
|
||||
print(" -w INT fuzzy windown size [" + win_size + "]");
|
||||
print(" -d FLOAT max allele diff if there is a single allele in the window [" + len_diff_ratio + "]");
|
||||
print(" -e print errors");
|
||||
return;
|
||||
}
|
||||
|
||||
function read_bed(fn) {
|
||||
var buf = new Bytes();
|
||||
var file = new File(fn);
|
||||
var bed = {};
|
||||
while (file.readline(buf) >= 0) {
|
||||
var t = buf.toString().split("\t");
|
||||
if (bed[t[0]] == null) bed[t[0]] = [];
|
||||
bed[t[0]].push([parseInt(t[1]), parseInt(t[2])]);
|
||||
}
|
||||
file.close();
|
||||
buf.destroy();
|
||||
for (var x in bed) {
|
||||
Interval.sort(bed[x]);
|
||||
Interval.merge(bed[x]);
|
||||
Interval.index_end(bed[x]);
|
||||
}
|
||||
return bed;
|
||||
}
|
||||
|
||||
var bed = bed_fn != null? read_bed(bed_fn) : null;
|
||||
|
||||
function read_vcf(fn, bed) {
|
||||
var buf = new Bytes();
|
||||
var file = new File(fn);
|
||||
var v = {};
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, t = buf.toString().split("\t");
|
||||
if (t[0][0] == '#') continue;
|
||||
if (bed != null && bed[t[0]] == null) continue;
|
||||
if (t[4] == '<INV>' || t[4] == '<INVDUP>') continue; // no inversion
|
||||
if (/[\[\]]/.test(t[4])) continue; // no break points
|
||||
var st = parseInt(t[1]) - 1, en = st + t[3].length;
|
||||
// parse svlen
|
||||
var b = _paf_get_alen(t), svlen = b[0];
|
||||
var abslen = svlen == null? 0 : svlen > 0? svlen : -svlen;
|
||||
if (abslen < min_flt || abslen > max_size) continue;
|
||||
// update end
|
||||
if ((m = /(^|;)END=(\d+)/.exec(t[7])) != null)
|
||||
en = parseInt(m[2]);
|
||||
else if (svlen != null && svlen < 0)
|
||||
en = st + (-svlen);
|
||||
if (en < st) en = st;
|
||||
if (st == en) --st, ++en;
|
||||
if (bed != null && Interval.find_ovlp(bed[t[0]], st, en).length == 0) continue;
|
||||
// insert
|
||||
if (v[t[0]] == null) v[t[0]] = [];
|
||||
v[t[0]].push([st, en, svlen, abslen]);
|
||||
}
|
||||
file.close();
|
||||
buf.destroy();
|
||||
for (var x in v) {
|
||||
Interval.sort(v[x]);
|
||||
Interval.index_end(v[x]);
|
||||
}
|
||||
return v;
|
||||
}
|
||||
|
||||
function compare_vcf(v0, v1, label) {
|
||||
var m = 0, n = 0;
|
||||
for (var x in v1) {
|
||||
var a1 = v1[x], a0 = v0[x];
|
||||
for (var i = 0; i < a1.length; ++i) {
|
||||
if (a1[i][3] < min_size) continue;
|
||||
++n;
|
||||
if (a0 == null) continue;
|
||||
var ws = win_size + (a1[i][3]>>1);
|
||||
var st = a1[i][0] > ws? a1[i][0] - ws : 0;
|
||||
b = Interval.find_ovlp(a0, st, a1[i][1] + ws);
|
||||
var n_ins = 0, n_del = 0, sv_del = null, sv_ins = null;
|
||||
for (var j = 0; j < b.length; ++j) {
|
||||
if (b[j][2] < 0) ++n_del, sv_del = -b[j][2];
|
||||
else if (b[j][2] > 0) ++n_ins, sv_ins = b[j][2];
|
||||
if (print_match)
|
||||
print("MA", x, a1[i].slice(0, 3).join("\t"), b[j].slice(0, 3).join("\t"));
|
||||
}
|
||||
var match = false;
|
||||
if (a1[i][2] > 0) { // insertion
|
||||
if (n_ins == 1) {
|
||||
var diff = sv_ins - a1[i][3];
|
||||
if (diff < 0) diff = -diff;
|
||||
if (diff < min_size || diff / a1[i][3] < len_diff_ratio)
|
||||
match = true;
|
||||
} else if (n_ins > 1) match = true; // multiple insertions; ambiguous
|
||||
} else if (a1[i][2] < 0) {
|
||||
if (n_del == 1) { // deletion
|
||||
var diff = sv_del - a1[i][3];
|
||||
if (diff < 0) diff = -diff;
|
||||
if (diff < min_size || diff / a1[i][3] < len_diff_ratio)
|
||||
match = true;
|
||||
} else if (n_del > 1) match = true; // multiple deletions; ambiguous
|
||||
}
|
||||
if (match) ++m;
|
||||
else if (print_err) {
|
||||
if ((a1[i][2] > 0 && n_ins > 0) || (a1[i][2] < 0 && n_del > 0))
|
||||
print("MM", x, a1[i].slice(0, 3).join("\t"));
|
||||
print(label, x, a1[i].slice(0, 3).join("\t"));
|
||||
}
|
||||
}
|
||||
}
|
||||
return [n, m];
|
||||
}
|
||||
|
||||
var v_base = read_vcf(args[getopt.ind+0], bed);
|
||||
var v_call = read_vcf(args[getopt.ind+1], bed);
|
||||
var fn = compare_vcf(v_call, v_base, 'FN');
|
||||
var fp = compare_vcf(v_base, v_call, 'FP');
|
||||
print('SN', fn[0], fn[1], (fn[1] / fn[0]).toFixed(6));
|
||||
print('PC', fp[0], fp[1], (fp[1] / fp[0]).toFixed(6));
|
||||
print('F1', ((fn[1] / fn[0] + fp[1] / fp[0]) / 2).toFixed(6));
|
||||
}
|
||||
|
||||
function paf_vcfsel(args)
|
||||
{
|
||||
var c, min_l = 0, max_l = 1<<30;
|
||||
while ((c = getopt(args, "l:L:")) != null) {
|
||||
if (c == 'l') min_l = parseInt(getopt.arg);
|
||||
else if (c == 'L') max_l = parseInt(getopt.arg);
|
||||
}
|
||||
|
||||
var buf = new Bytes();
|
||||
if (getopt.ind == args.length) {
|
||||
print("Usage: paftools.js vcfsel [options] <in.vcf>");
|
||||
return 1;
|
||||
}
|
||||
var file = args[getopt.ind] == "-"? new File() : new File(args[getopt.ind]);
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, line = buf.toString();
|
||||
if (line[0] == '#') {
|
||||
print(line);
|
||||
continue;
|
||||
}
|
||||
var t = line.split("\t");
|
||||
var st = parseInt(t[1]), en = st + t[3].length - 1;
|
||||
if ((m = /(^|;)END=(\d+)/.exec(t[7])) != null)
|
||||
en = parseInt(m[2]);
|
||||
if (en < st) {
|
||||
warn("END is smaller than POS: " + en + " < " + st);
|
||||
en = st;
|
||||
}
|
||||
var b = _paf_get_alen(t);
|
||||
var alen = b[0], min_abs_diff = b[1], max_abs_diff = b[2];
|
||||
if (max_abs_diff < min_l || min_abs_diff > max_l)
|
||||
continue;
|
||||
print(line);
|
||||
}
|
||||
file.close();
|
||||
buf.destroy();
|
||||
}
|
||||
|
||||
function paf_pafcmp(args)
|
||||
{
|
||||
var c, opt = { min_len:5000, min_mapq:10, min_ovlp:0.5 };
|
||||
while ((c = getopt(args, "q:")) != null) {
|
||||
if (c == 'q') opt.min_mapq = parseInt(opt.arg);
|
||||
}
|
||||
|
||||
var buf = new Bytes();
|
||||
if (args.length - getopt.ind < 2) {
|
||||
print("Usage: paftools.js pafcmp [options] <base.paf> <test.paf>");
|
||||
print("Options:");
|
||||
print(" -q INT min mapping quality [" + opt.min_mapq + "]");
|
||||
return 1;
|
||||
}
|
||||
|
||||
var eval = { n_base:0, n_test:0, n_out_high:0, n_out_low:0, n_hit:0, n_wrong:0, n_miss:0 };
|
||||
|
||||
function process_base(base, a) {
|
||||
if (a.length != 1) return;
|
||||
for (var i = 1; i < 4; ++i)
|
||||
a[0][i] = parseInt(a[0][i]);
|
||||
for (var i = 6; i < 12; ++i)
|
||||
a[0][i] = parseInt(a[0][i]);
|
||||
if (a[0][1] < opt.min_len) return;
|
||||
if (a[0][11] >= opt.min_mapq) ++eval.n_base;
|
||||
base[a[0][0]] = [a[0][5], a[0][7], a[0][8], a[0][11], 0, 0];
|
||||
}
|
||||
|
||||
var file = new File(args[getopt.ind]);
|
||||
warn("Reading " + args[getopt.ind] + "...");
|
||||
var a = [], base = {};
|
||||
while (file.readline(buf) >= 0) {
|
||||
var line = buf.toString();
|
||||
var t = line.split("\t");
|
||||
if (/\ttp:A:S/.test(line)) continue;
|
||||
if (a.length > 0 && a[0][0] != t[0]) {
|
||||
process_base(base, a);
|
||||
a = [];
|
||||
}
|
||||
a.push(t);
|
||||
}
|
||||
process_base(base, a);
|
||||
file.close();
|
||||
|
||||
function process_test(base, a) {
|
||||
for (var i = 1; i < 4; ++i)
|
||||
a[0][i] = parseInt(a[0][i]);
|
||||
for (var i = 6; i < 12; ++i)
|
||||
a[0][i] = parseInt(a[0][i]);
|
||||
if (a[0][1] < opt.min_len) return;
|
||||
if (a[0][11] >= opt.min_mapq) ++eval.n_test;
|
||||
var c = [a[0][5], a[0][7], a[0][8], a[0][11]];
|
||||
if (base[a[0][0]] == null) {
|
||||
if (c[3] >= opt.min_mapq) ++opt.n_out_high;
|
||||
else ++opt.n_out_low;
|
||||
} else {
|
||||
var b = base[a[0][0]];
|
||||
var inter = 0, union = (b[2] - b[1]) + (c[2] - c[1]);
|
||||
if (b[0] == c[0]) { // same chr
|
||||
if (b[1] < c[1]) {
|
||||
if (b[2] > c[1])
|
||||
inter = b[2] - c[1], union = c[2] - b[1];
|
||||
} else { // c[1] < b[1]
|
||||
if (c[2] > b[1])
|
||||
inter = c[2] - b[1], union = b[2] - c[1];
|
||||
}
|
||||
}
|
||||
if (inter >= union * opt.min_ovlp) {
|
||||
if (b[3] >= opt.min_mapq) ++eval.n_hit;
|
||||
++b[4];
|
||||
} else {
|
||||
if (b[3] >= opt.min_mapq) {
|
||||
print("W", a[0][0], b.slice(0, 4).join("\t"), c.join("\t"));
|
||||
++eval.n_wrong;
|
||||
}
|
||||
++b[5];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
file = new File(args[getopt.ind+1]);
|
||||
warn("Reading " + args[getopt.ind+1] + "...");
|
||||
a = [];
|
||||
while (file.readline(buf) >= 0) {
|
||||
var line = buf.toString();
|
||||
var t = line.split("\t");
|
||||
if (/\ttp:A:S/.test(line)) continue;
|
||||
if (a.length > 0 && a[0][0] != t[0]) {
|
||||
process_test(base, a);
|
||||
a = [];
|
||||
}
|
||||
a.push(t);
|
||||
}
|
||||
process_test(base, a);
|
||||
file.close();
|
||||
|
||||
for (var r in base) {
|
||||
var b = base[r];
|
||||
if (b[3] >= opt.min_mapq && b[4] == 0 && b[5] == 0) {
|
||||
++eval.n_miss;
|
||||
print("M", r, b.slice(0, 4).join("\t"));
|
||||
}
|
||||
}
|
||||
|
||||
print("X", eval.n_base + " base alignments with mapQ>=" + opt.min_mapq);
|
||||
// print("X", eval.n_test + " test alignments with mapQ>=" + opt.min_mapq);
|
||||
print("X", eval.n_hit + " base alignments correctly mapped by test");
|
||||
print("X", eval.n_wrong + " wrong test alignment");
|
||||
print("X", eval.n_miss + " base alignments missing");
|
||||
print("X", eval.n_out_high + " additional test alignments with mapQ>=" + opt.min_mapq);
|
||||
|
||||
buf.destroy();
|
||||
}
|
||||
|
||||
/*************************
|
||||
***** main function *****
|
||||
*************************/
|
||||
@@ -2497,13 +3078,17 @@ function main(args)
|
||||
print("");
|
||||
print(" stat collect basic mapping information in PAF/SAM");
|
||||
print(" asmstat collect basic assembly information");
|
||||
print(" asmgene evaluate gene completeness (EXPERIMENTAL)");
|
||||
print(" asmgene evaluate gene completeness");
|
||||
print(" misjoin evaluate large-scale misjoins");
|
||||
print(" liftover simplistic liftOver");
|
||||
print(" call call variants from asm-to-ref alignment with the cs tag");
|
||||
print(" bedcov compute the number of bases covered");
|
||||
print(" vcfstat VCF statistics");
|
||||
print(" sveval compare two SV callsets in VCF");
|
||||
print(" version print paftools.js version");
|
||||
print("");
|
||||
print(" mapeval evaluate mapping accuracy using mason2/PBSIM-simulated FASTQ");
|
||||
print(" pafcmp compare two PAF files");
|
||||
print(" mason2fq convert mason2-simulated SAM to FASTQ");
|
||||
print(" pbsim2fq convert PBSIM-simulated MAF to FASTQ");
|
||||
print(" junceval evaluate splice junction consistency with known annotations");
|
||||
@@ -2520,15 +3105,20 @@ function main(args)
|
||||
else if (cmd == 'stat') paf_stat(args);
|
||||
else if (cmd == 'asmstat') paf_asmstat(args);
|
||||
else if (cmd == 'asmgene') paf_asmgene(args);
|
||||
else if (cmd == 'misjoin') paf_misjoin(args);
|
||||
else if (cmd == 'liftover' || cmd == 'liftOver') paf_liftover(args);
|
||||
else if (cmd == 'vcfpair') paf_vcfpair(args);
|
||||
else if (cmd == 'call') paf_call(args);
|
||||
else if (cmd == 'mapeval') paf_mapeval(args);
|
||||
else if (cmd == 'pafcmp') paf_pafcmp(args);
|
||||
else if (cmd == 'bedcov') paf_bedcov(args);
|
||||
else if (cmd == 'mason2fq') paf_mason2fq(args);
|
||||
else if (cmd == 'pbsim2fq') paf_pbsim2fq(args);
|
||||
else if (cmd == 'junceval') paf_junceval(args);
|
||||
else if (cmd == 'ov-eval') paf_ov_eval(args);
|
||||
else if (cmd == 'vcfstat') paf_vcfstat(args);
|
||||
else if (cmd == 'sveval') paf_sveval(args);
|
||||
else if (cmd == 'vcfsel') paf_vcfsel(args);
|
||||
else if (cmd == 'version') print(paftools_version);
|
||||
else throw Error("unrecognized command: " + cmd);
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#include <assert.h>
|
||||
#include "minimap.h"
|
||||
#include "bseq.h"
|
||||
#include "kseq.h"
|
||||
|
||||
#define MM_PARENT_UNSET (-1)
|
||||
#define MM_PARENT_TMP_PRI (-2)
|
||||
@@ -35,13 +36,13 @@
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
#ifndef KSTRING_T
|
||||
#define KSTRING_T kstring_t
|
||||
typedef struct __kstring_t {
|
||||
unsigned l, m;
|
||||
char *s;
|
||||
} kstring_t;
|
||||
#endif
|
||||
typedef struct {
|
||||
uint32_t n;
|
||||
uint32_t q_pos;
|
||||
uint32_t q_span:31, flt:1;
|
||||
uint32_t seg_id:31, is_tandem:1;
|
||||
const uint64_t *cr;
|
||||
} mm_seed_t;
|
||||
|
||||
typedef struct {
|
||||
int n_u, n_a;
|
||||
@@ -59,31 +60,42 @@ uint32_t ks_ksmall_uint32_t(size_t n, uint32_t arr[], size_t kk);
|
||||
|
||||
void mm_sketch(void *km, const char *str, int len, int w, int k, uint32_t rid, int is_hpc, mm128_v *p);
|
||||
|
||||
mm_seed_t *mm_collect_matches(void *km, int *_n_m, int qlen, int max_occ, int max_max_occ, int dist, const mm_idx_t *mi, const mm128_v *mv, int64_t *n_a, int *rep_len, int *n_mini_pos, uint64_t **mini_pos);
|
||||
|
||||
double mm_event_identity(const mm_reg1_t *r);
|
||||
int mm_write_sam_hdr(const mm_idx_t *mi, const char *rg, const char *ver, int argc, char *argv[]);
|
||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int opt_flag);
|
||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int opt_flag, int rep_len);
|
||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag);
|
||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len);
|
||||
void mm_write_sam(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int n_regs, const mm_reg1_t *regs);
|
||||
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regs, const mm_reg1_t *const* regs, void *km, int opt_flag);
|
||||
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int opt_flag, int rep_len);
|
||||
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regs, const mm_reg1_t *const* regs, void *km, int64_t opt_flag);
|
||||
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int64_t opt_flag, int rep_len);
|
||||
|
||||
void mm_idxopt_init(mm_idxopt_t *opt);
|
||||
const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n);
|
||||
int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f);
|
||||
mm128_t *mm_chain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
||||
int mm_idx_getseq2(const mm_idx_t *mi, int is_rev, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq);
|
||||
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a);
|
||||
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a, int is_qstrand);
|
||||
|
||||
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a);
|
||||
void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a);
|
||||
mm128_t *mm_chain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float gap_scale,
|
||||
int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
||||
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
||||
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
||||
|
||||
void mm_mark_alt(const mm_idx_t *mi, int n, mm_reg1_t *r);
|
||||
void mm_split_reg(mm_reg1_t *r, mm_reg1_t *r2, int n, int qlen, mm128_t *a, int is_qstrand);
|
||||
void mm_sync_regs(void *km, int n_regs, mm_reg1_t *regs);
|
||||
int mm_squeeze_a(void *km, int n_regs, mm_reg1_t *regs, mm128_t *a);
|
||||
int mm_set_sam_pri(int n, mm_reg1_t *r);
|
||||
void mm_set_parent(void *km, float mask_level, int n, mm_reg1_t *r, int sub_diff, int hard_mask_level);
|
||||
void mm_set_parent(void *km, float mask_level, int mask_len, int n, mm_reg1_t *r, int sub_diff, int hard_mask_level, float alt_diff_frac);
|
||||
void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int *n_, mm_reg1_t *r);
|
||||
void mm_select_sub_multi(void *km, float pri_ratio, float pri1, float pri2, int max_gap_ref, int min_diff, int best_n, int n_segs, const int *qlens, int *n_, mm_reg1_t *r);
|
||||
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs);
|
||||
void mm_join_long(void *km, const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs, mm128_t *a);
|
||||
void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r);
|
||||
void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r, float alt_diff_frac);
|
||||
void mm_set_mapq(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr);
|
||||
void mm_update_dp_max(int qlen, int n_regs, mm_reg1_t *regs, float frac, int a, int b);
|
||||
|
||||
void mm_est_err(const mm_idx_t *mi, int qlen, int n_regs, mm_reg1_t *regs, const mm128_t *a, int32_t n, const uint64_t *mini_pos);
|
||||
|
||||
@@ -100,6 +112,16 @@ void mm_err_puts(const char *str);
|
||||
void mm_err_fwrite(const void *p, size_t size, size_t nitems, FILE *fp);
|
||||
void mm_err_fread(void *p, size_t size, size_t nitems, FILE *fp);
|
||||
|
||||
static inline float mg_log2(float x) // NB: this doesn't work when x<2
|
||||
{
|
||||
union { float f; uint32_t i; } z = { x };
|
||||
float log_2 = ((z.i >> 23) & 255) - 128;
|
||||
z.i &= ~(255 << 23);
|
||||
z.i += 127 << 23;
|
||||
log_2 += (-0.34484843f * z.f + 2.02466578f) * z.f - 0.67487759f;
|
||||
return log_2;
|
||||
}
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#include <stdio.h>
|
||||
#include <limits.h>
|
||||
#include "mmpriv.h"
|
||||
|
||||
extern bool enable_vect_dp_chaining;
|
||||
void mm_idxopt_init(mm_idxopt_t *opt)
|
||||
{
|
||||
memset(opt, 0, sizeof(mm_idxopt_t));
|
||||
@@ -15,24 +16,31 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
||||
memset(opt, 0, sizeof(mm_mapopt_t));
|
||||
opt->seed = 11;
|
||||
opt->mid_occ_frac = 2e-4f;
|
||||
opt->min_mid_occ = 10;
|
||||
opt->max_mid_occ = 1000000;
|
||||
opt->sdust_thres = 0; // no SDUST masking
|
||||
|
||||
opt->min_cnt = 3;
|
||||
opt->min_chain_score = 40;
|
||||
opt->bw = 500;
|
||||
opt->bw = 500, opt->bw_long = 20000;
|
||||
opt->max_gap = 5000;
|
||||
opt->max_gap_ref = -1;
|
||||
opt->max_chain_skip = 25;
|
||||
opt->max_chain_iter = 5000;
|
||||
opt->rmq_inner_dist = 1000;
|
||||
opt->rmq_size_cap = 100000;
|
||||
opt->rmq_rescue_size = 1000;
|
||||
opt->rmq_rescue_ratio = 0.1f;
|
||||
opt->chain_gap_scale = 0.8f;
|
||||
opt->max_max_occ = 4095;
|
||||
opt->occ_dist = 500;
|
||||
|
||||
opt->mask_level = 0.5f;
|
||||
opt->mask_len = INT_MAX;
|
||||
opt->pri_ratio = 0.8f;
|
||||
opt->best_n = 5;
|
||||
|
||||
opt->max_join_long = 20000;
|
||||
opt->max_join_short = 2000;
|
||||
opt->min_join_flank_sc = 1000;
|
||||
opt->min_join_flank_ratio = 0.5f;
|
||||
opt->alt_drop = 0.15f;
|
||||
|
||||
opt->a = 2, opt->b = 4, opt->q = 4, opt->e = 2, opt->q2 = 24, opt->e2 = 1;
|
||||
opt->sc_ambi = 1;
|
||||
@@ -43,6 +51,10 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
||||
opt->anchor_ext_len = 20, opt->anchor_ext_shift = 6;
|
||||
opt->max_clip_ratio = 1.0f;
|
||||
opt->mini_batch_size = 500000000;
|
||||
opt->max_sw_mat = 100000000;
|
||||
|
||||
opt->rank_min_len = 500;
|
||||
opt->rank_frac = 0.9f;
|
||||
|
||||
opt->pe_ori = 0; // FF
|
||||
opt->pe_bonus = 33;
|
||||
@@ -52,10 +64,13 @@ void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
||||
{
|
||||
if ((opt->flag & MM_F_SPLICE_FOR) || (opt->flag & MM_F_SPLICE_REV))
|
||||
opt->flag |= MM_F_SPLICE;
|
||||
if (opt->mid_occ <= 0)
|
||||
if (opt->mid_occ <= 0) {
|
||||
opt->mid_occ = mm_idx_cal_max_occ(mi, opt->mid_occ_frac);
|
||||
if (opt->mid_occ < opt->min_mid_occ)
|
||||
opt->mid_occ = opt->min_mid_occ;
|
||||
if (opt->mid_occ < opt->min_mid_occ)
|
||||
opt->mid_occ = opt->min_mid_occ;
|
||||
if (opt->max_mid_occ > opt->min_mid_occ && opt->mid_occ > opt->max_mid_occ)
|
||||
opt->mid_occ = opt->max_mid_occ;
|
||||
}
|
||||
if (mm_verbose >= 3)
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] mid_occ = %d\n", __func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), opt->mid_occ);
|
||||
}
|
||||
@@ -63,7 +78,7 @@ void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
||||
void mm_mapopt_max_intron_len(mm_mapopt_t *opt, int max_intron_len)
|
||||
{
|
||||
if ((opt->flag & MM_F_SPLICE) && max_intron_len > 0)
|
||||
opt->max_gap_ref = opt->bw = max_intron_len;
|
||||
opt->max_gap_ref = opt->bw = opt->bw_long = max_intron_len;
|
||||
}
|
||||
|
||||
int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
@@ -71,37 +86,50 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
if (preset == 0) {
|
||||
mm_idxopt_init(io);
|
||||
mm_mapopt_init(mo);
|
||||
} else if (strcmp(preset, "map-ont") == 0) { // this is the same as the default
|
||||
} else if (strcmp(preset, "ava-ont") == 0) {
|
||||
io->flag = 0, io->k = 15, io->w = 5;
|
||||
mo->flag |= MM_F_ALL_CHAINS | MM_F_NO_DIAG | MM_F_NO_DUAL | MM_F_NO_LJOIN;
|
||||
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_gap = 10000, mo->max_chain_skip = 25;
|
||||
mo->bw = 2000;
|
||||
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_chain_skip = 25;
|
||||
mo->bw = mo->bw_long = 2000;
|
||||
mo->occ_dist = 0;
|
||||
} else if (strcmp(preset, "map10k") == 0 || strcmp(preset, "map-pb") == 0) {
|
||||
#if defined (PARALLEL_CHAINING) && (defined(__AVX2__)) && (!defined(__AVX512BW__))
|
||||
enable_vect_dp_chaining = false;
|
||||
#endif
|
||||
io->flag |= MM_I_HPC, io->k = 19;
|
||||
} else if (strcmp(preset, "ava-pb") == 0) {
|
||||
io->flag |= MM_I_HPC, io->k = 19, io->w = 5;
|
||||
mo->flag |= MM_F_ALL_CHAINS | MM_F_NO_DIAG | MM_F_NO_DUAL | MM_F_NO_LJOIN;
|
||||
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_gap = 10000, mo->max_chain_skip = 25;
|
||||
} else if (strcmp(preset, "map10k") == 0 || strcmp(preset, "map-pb") == 0) {
|
||||
io->flag |= MM_I_HPC, io->k = 19;
|
||||
} else if (strcmp(preset, "map-ont") == 0) {
|
||||
io->flag = 0, io->k = 15;
|
||||
} else if (strcmp(preset, "asm5") == 0) {
|
||||
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_chain_skip = 25;
|
||||
mo->bw_long = mo->bw;
|
||||
mo->occ_dist = 0;
|
||||
} else if (strcmp(preset, "map-hifi") == 0 || strcmp(preset, "map-ccs") == 0) {
|
||||
#if defined (PARALLEL_CHAINING) && (defined(__AVX2__)) && (!defined(__AVX512BW__))
|
||||
enable_vect_dp_chaining = false;
|
||||
#endif
|
||||
io->flag = 0, io->k = 19, io->w = 19;
|
||||
mo->a = 1, mo->b = 19, mo->q = 39, mo->q2 = 81, mo->e = 3, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
||||
mo->min_mid_occ = 100;
|
||||
mo->max_gap = 10000;
|
||||
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1;
|
||||
mo->occ_dist = 500;
|
||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
||||
mo->min_dp_max = 200;
|
||||
mo->best_n = 50;
|
||||
} else if (strcmp(preset, "asm10") == 0) {
|
||||
} else if (strncmp(preset, "asm", 3) == 0) {
|
||||
io->flag = 0, io->k = 19, io->w = 19;
|
||||
mo->a = 1, mo->b = 9, mo->q = 16, mo->q2 = 41, mo->e = 2, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
||||
mo->min_mid_occ = 100;
|
||||
mo->min_dp_max = 200;
|
||||
mo->best_n = 50;
|
||||
} else if (strcmp(preset, "asm20") == 0) {
|
||||
io->flag = 0, io->k = 19, io->w = 10;
|
||||
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
||||
mo->min_mid_occ = 100;
|
||||
mo->bw = mo->bw_long = 100000;
|
||||
mo->max_gap = 10000;
|
||||
mo->flag |= MM_F_RMQ;
|
||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
||||
mo->min_dp_max = 200;
|
||||
mo->best_n = 50;
|
||||
if (strcmp(preset, "asm5") == 0) {
|
||||
mo->a = 1, mo->b = 19, mo->q = 39, mo->q2 = 81, mo->e = 3, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
||||
} else if (strcmp(preset, "asm10") == 0) {
|
||||
mo->a = 1, mo->b = 9, mo->q = 16, mo->q2 = 41, mo->e = 2, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
||||
} else if (strcmp(preset, "asm20") == 0) {
|
||||
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1, mo->zdrop = mo->zdrop_inv = 200;
|
||||
io->w = 10;
|
||||
} else return -1;
|
||||
} else if (strcmp(preset, "short") == 0 || strcmp(preset, "sr") == 0) {
|
||||
io->flag = 0, io->k = 21, io->w = 11;
|
||||
mo->flag |= MM_F_SR | MM_F_FRAG_MODE | MM_F_NO_PRINT_2ND | MM_F_2_IO_THREADS | MM_F_HEAP_SORT;
|
||||
@@ -111,7 +139,7 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
mo->end_bonus = 10;
|
||||
mo->max_frag_len = 800;
|
||||
mo->max_gap = 100;
|
||||
mo->bw = 100;
|
||||
mo->bw = mo->bw_long = 100;
|
||||
mo->pri_ratio = 0.5f;
|
||||
mo->min_cnt = 2;
|
||||
mo->min_chain_score = 25;
|
||||
@@ -123,7 +151,8 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
} else if (strncmp(preset, "splice", 6) == 0 || strcmp(preset, "cdna") == 0) {
|
||||
io->flag = 0, io->k = 15, io->w = 5;
|
||||
mo->flag |= MM_F_SPLICE | MM_F_SPLICE_FOR | MM_F_SPLICE_REV | MM_F_SPLICE_FLANK;
|
||||
mo->max_gap = 2000, mo->max_gap_ref = mo->bw = 200000;
|
||||
mo->max_sw_mat = 0;
|
||||
mo->max_gap = 2000, mo->max_gap_ref = mo->bw = mo->bw_long = 200000;
|
||||
mo->a = 1, mo->b = 2, mo->q = 2, mo->e = 1, mo->q2 = 32, mo->e2 = 0;
|
||||
mo->noncan = 9;
|
||||
mo->junc_bonus = 9;
|
||||
@@ -136,6 +165,16 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
|
||||
int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
||||
{
|
||||
if (mo->bw > mo->bw_long) {
|
||||
if (mm_verbose >= 1)
|
||||
fprintf(stderr, "[ERROR]\033[1;31m with '-rNUM1,NUM2', NUM1 (%d) can't be larger than NUM2 (%d)\033[0m\n", mo->bw, mo->bw_long);
|
||||
return -8;
|
||||
}
|
||||
if ((mo->flag & MM_F_RMQ) && (mo->flag & (MM_F_SR|MM_F_SPLICE))) {
|
||||
if (mm_verbose >= 1)
|
||||
fprintf(stderr, "[ERROR]\033[1;31m --rmq doesn't work with --sr or --splice\033[0m\n");
|
||||
return -7;
|
||||
}
|
||||
if (mo->split_prefix && (mo->flag & (MM_F_OUT_CS|MM_F_OUT_MD))) {
|
||||
if (mm_verbose >= 1)
|
||||
fprintf(stderr, "[ERROR]\033[1;31m --cs or --MD doesn't work with --split-prefix\033[0m\n");
|
||||
@@ -188,5 +227,10 @@ int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
||||
fprintf(stderr, "[ERROR]\033[1;31m -X/-P and --secondary=no can't be applied at the same time\033[0m\n");
|
||||
return -5;
|
||||
}
|
||||
if ((mo->flag & MM_F_QSTRAND) && ((mo->flag & (MM_F_OUT_SAM|MM_F_SPLICE|MM_F_FRAG_MODE)) || (io->flag & MM_I_HPC))) {
|
||||
if (mm_verbose >= 1)
|
||||
fprintf(stderr, "[ERROR]\033[1;31m --qstrand doesn't work with -a, -H, --frag or --splice\033[0m\n");
|
||||
return -5;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
+1
-1
@@ -144,7 +144,7 @@ properties:
|
||||
* **mlen**: length of the matching bases in the alignment, excluding ambiguous
|
||||
base matches.
|
||||
|
||||
* **NM**: number of mismatches, gaps and ambiguous poistions in the alignment
|
||||
* **NM**: number of mismatches, gaps and ambiguous positions in the alignment
|
||||
|
||||
* **trans_strand**: transcript strand. +1 if on the forward strand; -1 if on the
|
||||
reverse strand; 0 if unknown
|
||||
|
||||
+21
-6
@@ -6,26 +6,34 @@ cdef extern from "minimap.h":
|
||||
#
|
||||
ctypedef struct mm_idxopt_t:
|
||||
short k, w, flag, bucket_bits
|
||||
int mini_batch_size
|
||||
int64_t mini_batch_size
|
||||
uint64_t batch_size
|
||||
|
||||
ctypedef struct mm_mapopt_t:
|
||||
int64_t flag
|
||||
int seed
|
||||
int sdust_thres
|
||||
|
||||
int max_qlen
|
||||
int bw
|
||||
|
||||
int bw, bw_long
|
||||
int max_gap, max_gap_ref
|
||||
int max_frag_len
|
||||
int max_chain_skip, max_chain_iter
|
||||
int min_cnt
|
||||
int min_chain_score
|
||||
float chain_gap_scale
|
||||
int rmq_size_cap, rmq_inner_dist
|
||||
int rmq_rescue_size
|
||||
float rmq_rescue_ratio
|
||||
|
||||
float mask_level
|
||||
int mask_len
|
||||
float pri_ratio
|
||||
int best_n
|
||||
int max_join_long, max_join_short
|
||||
int min_join_flank_sc
|
||||
float min_join_flank_ratio
|
||||
|
||||
float alt_drop
|
||||
|
||||
int a, b, q, e, q2, e2
|
||||
int sc_ambi
|
||||
int noncan
|
||||
@@ -36,13 +44,20 @@ cdef extern from "minimap.h":
|
||||
int min_ksw_len
|
||||
int anchor_ext_len, anchor_ext_shift
|
||||
float max_clip_ratio
|
||||
|
||||
int rank_min_len
|
||||
float rank_frac
|
||||
|
||||
int pe_ori, pe_bonus
|
||||
|
||||
float mid_occ_frac
|
||||
int32_t min_mid_occ
|
||||
int32_t mid_occ
|
||||
int32_t max_occ
|
||||
int mini_batch_size
|
||||
int64_t mini_batch_size
|
||||
int64_t max_sw_mat
|
||||
int64_t cap_kalloc
|
||||
|
||||
const char *split_prefix
|
||||
|
||||
int mm_set_opt(char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
|
||||
+2
-2
@@ -3,7 +3,7 @@ from libc.stdlib cimport free
|
||||
cimport cmappy
|
||||
import sys
|
||||
|
||||
__version__ = '2.17'
|
||||
__version__ = '2.22'
|
||||
|
||||
cmappy.mm_reset_timer()
|
||||
|
||||
@@ -82,7 +82,7 @@ cdef class Alignment:
|
||||
|
||||
@property
|
||||
def cigar_str(self):
|
||||
return "".join(map(lambda x: str(x[0]) + 'MIDNSH'[x[1]], self._cigar))
|
||||
return "".join(map(lambda x: str(x[0]) + 'MIDNSHP=XB'[x[1]], self._cigar))
|
||||
|
||||
def __str__(self):
|
||||
if self._strand > 0: strand = '+'
|
||||
|
||||
@@ -0,0 +1,151 @@
|
||||
#include "mmpriv.h"
|
||||
#include "kalloc.h"
|
||||
#include "ksort.h"
|
||||
#include <stdlib.h>
|
||||
#include<algorithm>
|
||||
#include <x86intrin.h>
|
||||
|
||||
#ifdef LISA_HASH
|
||||
#include "lisa_hash.h"
|
||||
extern lisa_hash<uint64_t, uint64_t> *lh;
|
||||
#endif
|
||||
extern uint64_t minimizer_lookup_time;
|
||||
|
||||
mm_seed_t *mm_seed_collect_all(void *km, const mm_idx_t *mi, const mm128_v *mv, int32_t *n_m_)
|
||||
{
|
||||
//#ifdef MANUAL_PROFILING
|
||||
// uint64_t lookup_start = __rdtsc();
|
||||
//#endif
|
||||
|
||||
#ifdef LISA_HASH
|
||||
//-----------------------------------
|
||||
uint64_t** cr_batch = (uint64_t**) malloc((mv->n)*sizeof(uint64_t*));
|
||||
int* t_batch = (int*)malloc((mv->n)*sizeof(int));
|
||||
uint64_t* minimizers = (uint64_t*) malloc((mv->n)*sizeof(uint64_t));
|
||||
int64_t* lisa_pos = (int64_t*) malloc((max(32, (int)mv->n))* sizeof(int64_t));
|
||||
|
||||
for (size_t i = 0; i < mv->n; i++) {
|
||||
mm128_t *p = &mv->a[i];
|
||||
minimizers[i] = p->x>>8;
|
||||
}
|
||||
|
||||
lh->mm_idx_get_batched(minimizers, mv->n, lisa_pos, cr_batch, t_batch);
|
||||
//-----------------------------------
|
||||
|
||||
#endif
|
||||
|
||||
|
||||
mm_seed_t *m;
|
||||
size_t i;
|
||||
int32_t k;
|
||||
m = (mm_seed_t*)kmalloc(km, mv->n * sizeof(mm_seed_t));
|
||||
for (i = k = 0; i < mv->n; ++i) {
|
||||
const uint64_t *cr;
|
||||
mm_seed_t *q;
|
||||
mm128_t *p = &mv->a[i];
|
||||
uint32_t q_pos = (uint32_t)p->y, q_span = p->x & 0xff;
|
||||
int t;
|
||||
#ifdef LISA_HASH
|
||||
t = t_batch[i];
|
||||
cr = cr_batch[i];
|
||||
#else
|
||||
cr = mm_idx_get(mi, p->x>>8, &t);
|
||||
#endif
|
||||
if (t == 0) continue;
|
||||
q = &m[k++];
|
||||
q->q_pos = q_pos, q->q_span = q_span, q->cr = cr, q->n = t, q->seg_id = p->y >> 32;
|
||||
q->is_tandem = q->flt = 0;
|
||||
if (i > 0 && p->x>>8 == mv->a[i - 1].x>>8) q->is_tandem = 1;
|
||||
if (i < mv->n - 1 && p->x>>8 == mv->a[i + 1].x>>8) q->is_tandem = 1;
|
||||
}
|
||||
#ifdef LISA_HASH
|
||||
free(cr_batch);
|
||||
free(t_batch);
|
||||
free(minimizers);
|
||||
free(lisa_pos);
|
||||
#endif
|
||||
*n_m_ = k;
|
||||
//#ifdef MANUAL_PROFILING
|
||||
// minimizer_lookup_time += __rdtsc() - lookup_start;
|
||||
//#endif
|
||||
return m;
|
||||
}
|
||||
|
||||
#define MAX_MAX_HIGH_OCC 128
|
||||
|
||||
void mm_seed_select(int32_t n, mm_seed_t *a, int len, int max_occ, int max_max_occ, int dist)
|
||||
{ // for high-occ minimizers, choose up to max_high_occ in each high-occ streak
|
||||
extern void ks_heapdown_uint64_t(size_t i, size_t n, uint64_t*);
|
||||
extern void ks_heapmake_uint64_t(size_t n, uint64_t*);
|
||||
int32_t i, last0, m;
|
||||
uint64_t b[MAX_MAX_HIGH_OCC]; // this is to avoid a heap allocation
|
||||
|
||||
if (n == 0 || n == 1) return;
|
||||
for (i = m = 0; i < n; ++i)
|
||||
if (a[i].n > max_occ) ++m;
|
||||
if (m == 0) return; // no high-frequency k-mers; do nothing
|
||||
for (i = 0, last0 = -1; i <= n; ++i) {
|
||||
if (i == n || a[i].n <= max_occ) {
|
||||
if (i - last0 > 1) {
|
||||
int32_t ps = last0 < 0? 0 : (uint32_t)a[last0].q_pos>>1;
|
||||
int32_t pe = i == n? len : (uint32_t)a[i].q_pos>>1;
|
||||
int32_t j, k, st = last0 + 1, en = i;
|
||||
int32_t max_high_occ = (int32_t)((double)(pe - ps) / dist + .499);
|
||||
if (max_high_occ > 0) {
|
||||
if (max_high_occ > MAX_MAX_HIGH_OCC)
|
||||
max_high_occ = MAX_MAX_HIGH_OCC;
|
||||
for (j = st, k = 0; j < en && k < max_high_occ; ++j, ++k)
|
||||
b[k] = (uint64_t)a[j].n<<32 | j;
|
||||
ks_heapmake_uint64_t(k, b); // initialize the binomial heap
|
||||
for (; j < en; ++j) { // if there are more, choose top max_high_occ
|
||||
if (a[j].n < (int32_t)(b[0]>>32)) { // then update the heap
|
||||
b[0] = (uint64_t)a[j].n<<32 | j;
|
||||
ks_heapdown_uint64_t(0, k, b);
|
||||
}
|
||||
}
|
||||
for (j = 0; j < k; ++j) a[(uint32_t)b[j]].flt = 1;
|
||||
}
|
||||
for (j = st; j < en; ++j) a[j].flt ^= 1;
|
||||
for (j = st; j < en; ++j)
|
||||
if (a[j].n > max_max_occ)
|
||||
a[j].flt = 1;
|
||||
}
|
||||
last0 = i;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
mm_seed_t *mm_collect_matches(void *km, int *_n_m, int qlen, int max_occ, int max_max_occ, int dist, const mm_idx_t *mi, const mm128_v *mv, int64_t *n_a, int *rep_len, int *n_mini_pos, uint64_t **mini_pos)
|
||||
{
|
||||
int rep_st = 0, rep_en = 0, n_m, n_m0;
|
||||
size_t i;
|
||||
mm_seed_t *m;
|
||||
*n_mini_pos = 0;
|
||||
*mini_pos = (uint64_t*)kmalloc(km, mv->n * sizeof(uint64_t));
|
||||
m = mm_seed_collect_all(km, mi, mv, &n_m0);
|
||||
if (dist > 0 && max_max_occ > max_occ) {
|
||||
mm_seed_select(n_m0, m, qlen, max_occ, max_max_occ, dist);
|
||||
} else {
|
||||
for (i = 0; i < n_m0; ++i)
|
||||
if (m[i].n > max_occ)
|
||||
m[i].flt = 1;
|
||||
}
|
||||
for (i = 0, n_m = 0, *rep_len = 0, *n_a = 0; i < n_m0; ++i) {
|
||||
mm_seed_t *q = &m[i];
|
||||
//fprintf(stderr, "X\t%d\t%d\t%d\n", q->q_pos>>1, q->n, q->flt);
|
||||
if (q->flt) {
|
||||
int en = (q->q_pos >> 1) + 1, st = en - q->q_span;
|
||||
if (st > rep_en) {
|
||||
*rep_len += rep_en - rep_st;
|
||||
rep_st = st, rep_en = en;
|
||||
} else rep_en = en;
|
||||
} else {
|
||||
*n_a += q->n;
|
||||
(*mini_pos)[(*n_mini_pos)++] = (uint64_t)q->q_span<<32 | q->q_pos>>1;
|
||||
m[n_m++] = *q;
|
||||
}
|
||||
}
|
||||
*rep_len += rep_en - rep_st;
|
||||
*_n_m = n_m;
|
||||
return m;
|
||||
}
|
||||
@@ -4,16 +4,6 @@ except ImportError:
|
||||
from distutils.core import setup
|
||||
from distutils.extension import Extension
|
||||
|
||||
cmdclass = {}
|
||||
|
||||
try:
|
||||
from Cython.Build import build_ext
|
||||
except ImportError: # without Cython
|
||||
module_src = 'python/mappy.c'
|
||||
else: # with Cython
|
||||
module_src = 'python/mappy.pyx'
|
||||
cmdclass['build_ext'] = build_ext
|
||||
|
||||
import sys, platform
|
||||
|
||||
sys.path.append('python')
|
||||
@@ -33,7 +23,7 @@ def readme():
|
||||
|
||||
setup(
|
||||
name = 'mappy',
|
||||
version = '2.17',
|
||||
version = '2.22',
|
||||
url = 'https://github.com/lh3/minimap2',
|
||||
description = 'Minimap2 python binding',
|
||||
long_description = readme(),
|
||||
@@ -42,8 +32,8 @@ setup(
|
||||
license = 'MIT',
|
||||
keywords = 'sequence-alignment',
|
||||
scripts = ['python/minimap2.py'],
|
||||
ext_modules = [Extension('mappy',
|
||||
sources = [module_src, 'align.c', 'bseq.c', 'chain.c', 'format.c', 'hit.c', 'index.c', 'pe.c', 'options.c',
|
||||
ext_modules = [Extension('mappy',
|
||||
sources = ['python/mappy.pyx', 'align.c', 'bseq.c', 'lchain.c', 'seed.c', 'format.c', 'hit.c', 'index.c', 'pe.c', 'options.c',
|
||||
'ksw2_extd2_sse.c', 'ksw2_exts2_sse.c', 'ksw2_extz2_sse.c', 'ksw2_ll_sse.c',
|
||||
'kalloc.c', 'kthread.c', 'map.c', 'misc.c', 'sdust.c', 'sketch.c', 'esterr.c', 'splitidx.c'],
|
||||
depends = ['minimap.h', 'bseq.h', 'kalloc.h', 'kdq.h', 'khash.h', 'kseq.h', 'ksort.h',
|
||||
@@ -62,4 +52,4 @@ setup(
|
||||
'Programming Language :: Python :: 3',
|
||||
'Intended Audience :: Science/Research',
|
||||
'Topic :: Scientific/Engineering :: Bio-Informatics'],
|
||||
cmdclass = cmdclass)
|
||||
setup_requires=["cython"])
|
||||
|
||||
@@ -338,3 +338,114 @@
|
||||
Title = {Introducing difference recurrence relations for faster semi-global alignment of long sequences},
|
||||
Volume = {19},
|
||||
Year = {2018}}
|
||||
|
||||
@article{Li:2018ab,
|
||||
Author = {Li, Heng},
|
||||
Journal = {Bioinformatics},
|
||||
Pages = {3094-3100},
|
||||
Title = {Minimap2: pairwise alignment for nucleotide sequences},
|
||||
Volume = {34},
|
||||
Year = {2018}}
|
||||
|
||||
@article{Jain:2020aa,
|
||||
Author = {Jain, Chirag and others},
|
||||
Journal = {Bioinformatics},
|
||||
Pages = {i111-i118},
|
||||
Title = {Weighted minimizer sampling improves long read mapping},
|
||||
Volume = {36},
|
||||
Year = {2020}}
|
||||
|
||||
@article{Miga:2020aa,
|
||||
Author = {Miga, Karen H and others},
|
||||
Journal = {Nature},
|
||||
Pages = {79-84},
|
||||
Title = {Telomere-to-telomere assembly of a complete human {X} chromosome},
|
||||
Volume = {585},
|
||||
Year = {2020}}
|
||||
|
||||
@article {Jain2020.11.01.363887,
|
||||
author = {Jain, Chirag and others},
|
||||
title = {A long read mapping method for highly repetitive reference sequences},
|
||||
elocation-id = {2020.11.01.363887},
|
||||
year = {2020},
|
||||
doi = {10.1101/2020.11.01.363887},
|
||||
publisher = {Cold Spring Harbor Laboratory},
|
||||
abstract = {About 5-10\% of the human genome remains inaccessible for functional analysis due to the presence of repetitive sequences such as segmental duplications and tandem repeat arrays. To enable high-quality resequencing of personal genomes, it is crucial to support end-to-end genome variant discovery using repeat-aware read mapping methods. In this study, we highlight the fact that existing long read mappers often yield incorrect alignments and variant calls within long, near-identical repeats, as they remain vulnerable to allelic bias. In the presence of a non-reference allele within a repeat, a read sampled from that region could be mapped to an incorrect repeat copy because the standard pairwise sequence alignment scoring system penalizes true variants.To address the above problem, we propose a novel, long read mapping method that addresses allelic bias by making use of minimal confidently alignable substrings (MCASs). MCASs are formulated as minimal length substrings of a read that have unique alignments to a reference locus with sufficient mapping confidence (i.e., a mapping quality score above a user-specified threshold). This approach treats each read mapping as a collection of confident sub-alignments, which is more tolerant of structural variation and more sensitive to paralog-specific variants (PSVs) within repeats. We mathematically define MCASs and discuss an exact algorithm as well as a practical heuristic to compute them. The proposed method, referred to as Winnowmap2, is evaluated using simulated as well as real long read benchmarks using the recently completed gapless assemblies of human chromosomes X and 8 as a reference. We show that Winnowmap2 successfully addresses the issue of allelic bias, enabling more accurate downstream variant calls in repetitive sequences. As an example, using simulated PacBio HiFi reads and structural variants in chromosome 8, Winnowmap2 alignments achieved the lowest false-negative and false-positive rates (1.89\%, 1.89\%) for calling structural variants within near-identical repeats compared to minimap2 (39.62\%, 5.88\%) and NGMLR (56.60\%, 36.11\%) respectively.Winnowmap2 code is accessible at https://github.com/marbl/WinnowmapCompeting Interest StatementThe authors have declared no competing interest.},
|
||||
URL = {https://www.biorxiv.org/content/early/2020/11/02/2020.11.01.363887},
|
||||
eprint = {https://www.biorxiv.org/content/early/2020/11/02/2020.11.01.363887.full.pdf},
|
||||
journal = {bioRxiv}
|
||||
}
|
||||
|
||||
@article{Li:2020aa,
|
||||
Author = {Li, Heng and others},
|
||||
Journal = {Genome Biol},
|
||||
Pages = {265},
|
||||
Title = {The design and construction of reference pangenome graphs with minigraph},
|
||||
Volume = {21},
|
||||
Year = {2020}}
|
||||
|
||||
@article{Ren:2021aa,
|
||||
Author = {Ren, Jingwen and Chaisson, Mark J P},
|
||||
Journal = {PLoS Comput Biol},
|
||||
Pages = {e1009078},
|
||||
Title = {lra: A long read aligner for sequences and contigs},
|
||||
Volume = {17},
|
||||
Year = {2021}}
|
||||
|
||||
@inproceedings{DBLP:conf/wabi/AbouelhodaO03,
|
||||
Author = {Mohamed Ibrahim Abouelhoda and Enno Ohlebusch},
|
||||
Booktitle = {Algorithms in Bioinformatics, Third International Workshop, {WABI} 2003, Budapest, Hungary, September 15-20, 2003, Proceedings},
|
||||
Crossref = {DBLP:conf/wabi/2003},
|
||||
Pages = {1--16},
|
||||
Title = {A Local Chaining Algorithm and Its Applications in Comparative Genomics},
|
||||
Year = {2003}}
|
||||
|
||||
@article{Ono:2021aa,
|
||||
Author = {Ono, Yukiteru and others},
|
||||
Journal = {Bioinformatics},
|
||||
Pages = {589-595},
|
||||
Title = {{PBSIM2}: a simulator for long-read sequencers with a novel generative model of quality scores},
|
||||
Volume = {37},
|
||||
Year = {2021}}
|
||||
|
||||
@article{Sedlazeck:2018ab,
|
||||
Author = {Sedlazeck, Fritz J and others},
|
||||
Journal = {Nat Methods},
|
||||
Pages = {461-468},
|
||||
Title = {Accurate detection of complex structural variations using single-molecule sequencing},
|
||||
Volume = {15},
|
||||
Year = {2018}}
|
||||
|
||||
@article{Jeffares:2017aa,
|
||||
Author = {Jeffares, Daniel C and others},
|
||||
Journal = {Nat Commun},
|
||||
Pages = {14061},
|
||||
Title = {Transient structural variations have strong effects on quantitative traits and reproductive isolation in fission yeast},
|
||||
Volume = {8},
|
||||
Year = {2017}}
|
||||
|
||||
@article{Zook:2020aa,
|
||||
Author = {Zook, Justin M and others},
|
||||
Journal = {Nat Biotechnol},
|
||||
Pages = {1347-1355},
|
||||
Title = {A robust benchmark for detection of germline large deletions and insertions},
|
||||
Volume = {38},
|
||||
Year = {2020}}
|
||||
|
||||
@article{Harpak:2017aa,
|
||||
Author = {Harpak, Arbel and others},
|
||||
Journal = {Proc Natl Acad Sci U S A},
|
||||
Pages = {12779-12784},
|
||||
Title = {Frequent nonallelic gene conversion on the human lineage and its effect on the divergence of gene duplicates},
|
||||
Volume = {114},
|
||||
Year = {2017}}
|
||||
|
||||
@article{Li:2018aa,
|
||||
Author = {Li, Heng and others},
|
||||
Journal = {Nat Methods},
|
||||
Month = {Aug},
|
||||
Number = {8},
|
||||
Pages = {595-597},
|
||||
Title = {A synthetic-diploid benchmark for accurate variant-calling evaluation},
|
||||
Volume = {15},
|
||||
Year = {2018}}
|
||||
|
||||
@@ -0,0 +1,225 @@
|
||||
\documentclass{bioinfo}
|
||||
\copyrightyear{2021}
|
||||
\pubyear{2021}
|
||||
|
||||
\usepackage{graphicx}
|
||||
\usepackage{hyperref}
|
||||
\usepackage{url}
|
||||
\usepackage{amsmath}
|
||||
\usepackage[ruled,vlined]{algorithm2e}
|
||||
\newcommand\mycommfont[1]{\footnotesize\rmfamily{\it #1}}
|
||||
\SetCommentSty{mycommfont}
|
||||
\SetKwComment{Comment}{$\triangleright$\ }{}
|
||||
|
||||
\usepackage{natbib}
|
||||
\bibliographystyle{apalike}
|
||||
|
||||
\DeclareMathOperator*{\argmax}{argmax}
|
||||
|
||||
\begin{document}
|
||||
\firstpage{1}
|
||||
|
||||
\title[Improvements to minimap2]{New strategies to improve minimap2 alignment accuracy}
|
||||
\author[Li]{Heng Li$^{1,2}$}
|
||||
\address{$^1$Dana-Farber Cancer Institute, 450 Brookline Ave, Boston, MA 02215, USA,
|
||||
$^2$Harvard Medical School, 10 Shattuck St, Boston, MA 02215, USA}
|
||||
|
||||
\maketitle
|
||||
|
||||
\begin{abstract}
|
||||
|
||||
\section{Summary:} We present several recent improvements to minimap2, a
|
||||
versatile pairwise aligner for nucleotide sequences. Now minimap2 v2.22 can
|
||||
more accurately map long reads to highly repetitive regions and align through
|
||||
insertions or deletions up to 100kb by default, addressing major weakness in
|
||||
minimap2 v2.18 or earlier.
|
||||
|
||||
\section{Availability and implementation:}
|
||||
\href{https://github.com/lh3/minimap2}{https://github.com/lh3/minimap2}
|
||||
|
||||
\section{Contact:} hli@ds.dfci.harvard.edu
|
||||
\end{abstract}
|
||||
|
||||
\section{Introduction}
|
||||
Minimap2~\citep{Li:2018ab} is widely used for maping long sequence
|
||||
reads and assembly contigs. \citet{Jain:2020aa} found minimap2 v2.18 or earlier occasionally
|
||||
misaligned reads from highly repetitive regions as minimap2 ignored seeds of
|
||||
high occurrence. They also noticed minimap2 may misplace reads with structural
|
||||
variations (SVs) in such regions~\citep{Jain2020.11.01.363887}. These
|
||||
misalignments have become a pressing issue in the advent of
|
||||
temolere-to-telomore human assembly~\citep{Miga:2020aa}. Meanwhile, old minimap2
|
||||
was unable to efficiently align long insertions/deletions (INDELs) and often
|
||||
breaks an alignment around variable-number tandem repeats (VNTRs). This has
|
||||
inspired new chaining algorithms~\citep{Li:2020aa,Ren:2021aa} which are not
|
||||
integrated into minimap2. Here we will describe recent efforts implemented
|
||||
in v2.19 through v2.22 to improve mapping results.
|
||||
|
||||
\begin{methods}
|
||||
\section{Methods}
|
||||
|
||||
\subsection{Rescuing high-occurrence $k$-mers}
|
||||
Minimap2 keeps all $k$-mer minimizers during indexing. Its original
|
||||
implementation only selected low-occurrence minimizers during mapping. The
|
||||
cutoff is a few hundred for mapping long reads against a human genome. If a
|
||||
read habors only a few or even no low-occurrence minimizers, it will fail
|
||||
chaining due to insufficient anchors.
|
||||
|
||||
To resolve this issue, we implemented a new heuristic to add additional
|
||||
minimizers. Suppose we are looking at two adjacent low-occurence $k$-mers
|
||||
located at position $x_1$ and $x_2$, respectively. If $|x_1-x_2|\ge500$,
|
||||
minimap2 v2.22 additionally selects $\lfloor|x_1-x_2|/500\rfloor$ minimizers
|
||||
of the lowest occurrence among minimizers between $x_1$ and $x_2$.
|
||||
We use a binary heap data
|
||||
structure to select minimizers of the lowest occurrence in this interval.
|
||||
This strategy adds necessary anchors at the cost of increasing total alignment
|
||||
time by a few percent on real data.
|
||||
|
||||
\subsection{Aligning through longer INDELs}
|
||||
The original minimap2 may fail to align long INDELs due to its chaining
|
||||
heuristics. Briefly, minimap2 applies dynamic programming (DP) to chain
|
||||
minimizer anchors. This is a quadratic algorithm, which is slow for chaining
|
||||
contigs. For acceptable performance, the original minimap2 uses a 500bp band by
|
||||
default. If there is an INDEL longer than 500bp and the two chains around the INDEL
|
||||
have no overlaps on either the query or the reference sequence, minimap2 may
|
||||
join the two short chains later at a later step. We call it the
|
||||
long-join heuristic. This heuristic may fail around VNTRs because short chains
|
||||
often have overlaps in VNTRs. More subtly, minimap2 may escape the inner DP
|
||||
loop early, again for performance, if the chaining result is not improved for
|
||||
50 iterations. When there is a copy number change in a long segmental
|
||||
duplication, the early escape may break around the event even if users
|
||||
specify a large band.
|
||||
|
||||
In minigraph~\citep{Li:2020aa}, we developed a new chaining algorithm that
|
||||
finds short INDELs with DP-based chaining and goes through long INDELs with a
|
||||
subquadratic algorithm~\citep{DBLP:conf/wabi/AbouelhodaO03}. We ported the same
|
||||
algorithm to minimap2 for contig mapping. For long-read mapping, the minigraph
|
||||
algorithm is slower. Minimap2 v2.22 now still uses the DP-based algorithm to
|
||||
find short chains and then invokes the minigraph algorithm to rechain anchors in
|
||||
these short chains. The rechaining step achieves the same goal as long-join
|
||||
but is more reliable as it can resolve overlaps between short chains. The old
|
||||
long-join heuristic has since been removed.
|
||||
|
||||
\subsection{Properly mapping long reads with SVs}
|
||||
The original minimap2 ranks an alignment by its Smith-Waterman score and
|
||||
outputs the best scoring alignment. However, when there are SVs on the read,
|
||||
the best scoring alignment is sometimes not the correct alignment.
|
||||
\citet{Jain2020.11.01.363887} resolved this dilemma by altering the mapping
|
||||
algorithm.
|
||||
|
||||
In our view, this problem is rooted in impropriate scoring: affine-gap penalty
|
||||
over-penalizes a long INDEL that was often evolutionarily created in one event.
|
||||
We should not penalize a SV linearly in its length. Minimap2 v2.22 rescores
|
||||
an alignment with the following scoring function. Suppose an alignment consists
|
||||
of $M$ matching bases, $N$ substitutions and $G$ gap opens, we empirically
|
||||
score the alignment with
|
||||
$$
|
||||
M-\frac{N+G}{2d}-\sum_{i=1}^G\log_2(1+g_i)
|
||||
$$
|
||||
where $g_i\ge1$ is the length of the $i$-th gap and
|
||||
$$
|
||||
d=\max\left\{\frac{N+G}{M+N+G},0.02\right\}
|
||||
$$
|
||||
Here $d$ approximates per-base sequence divergence with the smallest value set
|
||||
to 2\%. As an analogy to affine-gap scoring, the matching score in our scheme
|
||||
is 1, the mismatch and gap open penalties are both $1/2d$ and the gap extension
|
||||
penalty is a logarithm function of the gap length. Our scoring gives a long SV
|
||||
a much milder penalty. In terms of time complexity, scoring an alignment is
|
||||
linear in the length of the alignment. Time spent on rescoring is negligible in
|
||||
practice.
|
||||
|
||||
%If we assume sequences evolve under a duplication-mutation model, we may have a
|
||||
%better way to choose the best alignment. If a long read can be mapped to $n$
|
||||
%loci, we can take the read as the template and build a
|
||||
%pseudo-multi-sequence-alignment (pMSA) of $n+1$ sequences. In this pMSA, we say
|
||||
%a site on the read is informative if the $n$ reference subsequences differ at
|
||||
%the position.
|
||||
|
||||
\end{methods}
|
||||
|
||||
\section{Results}
|
||||
|
||||
\begin{table}
|
||||
\processtable{Evaluation of minimap2 v2.22}
|
||||
{\footnotesize\label{tab:1}\begin{tabular}{p{4.2cm}rrrr}
|
||||
\toprule
|
||||
$[$Benchmark$]$ Metric & v2.22 & v2.18 & Winno & lra \\
|
||||
\midrule
|
||||
$[$sim-map$]$ \% mapped reads at Q10 & 97.9 & 97.6 & {\bf 99.0} & 97.3 \\
|
||||
$[$sim-map$]$ err. rate at Q10 (phredQ) & {\bf 52} & {\bf 52} & 38 & 24 \\
|
||||
$[$winno-cmp$]$ rate of diff. (phredQ) & {\bf 41} & 37 & N/A & 18 \\
|
||||
$[$sim-sv$]$ \% false negative rate & {\bf 0.5} & 2.0 & {\bf 0.5} & 1.4 \\
|
||||
$[$sim-sv$]$ \% false discovery rate & {\bf 0.0} & 0.1 & {\bf 0.0} & 0.1 \\
|
||||
$[$real-sv-1k$]$ \% false negative rate & {\bf 7.3} & 20.0 & 13.0 & N/A \\
|
||||
$[$real-sv-1k$]$ \% false discovery rate & 2.7 & {\bf 2.4} & 2.7 & N/A \\
|
||||
\botrule
|
||||
\end{tabular}}
|
||||
{In $[$sim-map$]$, 152,713 reads were simulated from the CHM13 telomere-to-telomere assembly v1.1
|
||||
(AC: GCA\_009914755.3) with pbsim2~\citep{Ono:2021aa}: ``pbsim2 -{}-hmm\_model R94.model -{}-length-min
|
||||
5000 -{}-length-mean 20000 -{}-accuracy-mean 0.95''. Alignments of mapping quality
|
||||
10 or higher were evaluated by ``paftools.js mapeval''. The mapping error rate
|
||||
is measured in the phred scale: if the error rate is $e$, $-10\log_{10}e$ is
|
||||
reported in the table. In $[$winno-cmp$]$, 1.39 million CHM13 HiFi reads from
|
||||
SRR11292121 were mapped against CHM13. 99.3\% of them were mapped by Winnowmap2
|
||||
at mapping quality 10 or higher and were taken as ground truth to evaluate
|
||||
minimap2 and lra with ``paftools.js pafcmp''. $[$sim-sv$]$ simulated 1,000
|
||||
50bp to 1000bp INDELs from chr8 in CHM13 using SURVIVOR~\citep{Jeffares:2017aa} and simulated Nanopore
|
||||
reads at 30 folds with the same pbsim2 command line. SVs were called with
|
||||
``sniffles -q 10''~\citep{Sedlazeck:2018ab} and compared to the simulated truth with ``SURVIVOR eval
|
||||
call.vcf truth.bed 50''. In $[$real-sv-1k$]$, small and long variants were
|
||||
called by dipcall-0.3~\citep{Li:2018aa} for HG002 assemblies (AC: GCA\_018852605.1 and
|
||||
GCA\_018852615.1) and compared to the GIAB truth~\citep{Zook:2020aa} using ``truvari -r 2000 -s
|
||||
1000 -S 400 -{}-multimatch -{}-passonly'' which sets the minimum INDEL size to 1kb in evaluation. }
|
||||
\end{table}
|
||||
|
||||
We evaluated minimap2 v2.22 along with v2.18, Winnowmap2 v2.03 and lra v1.3.2
|
||||
(Table~\ref{tab:1}). Both versions of minimap2 achieved high mapping accuracy on
|
||||
simulated Nanopore reads (sim-map). Winnowmap2 aligned more reads at mapping
|
||||
quality 10 or higher (mapQ10). However, it may occasionally assign a high mapping
|
||||
quality to a read with multiple identical best alignments. This reduced its
|
||||
mapping accuracy.
|
||||
|
||||
In lack of groud truth for real data, so we took Winnowmap2 mapping as ground
|
||||
truth to evaluate other mappers (winno-cmp). Out of 1,378,092 reads with mapQ10
|
||||
alignments by Winnowmap2, minimap2 v2.22 could map all of them. 118 reads, less
|
||||
than 0.01\% of all reads, were mapped differently by v2.22. 51 of them have
|
||||
multiple identical best alignments. We believe these are more likely to be
|
||||
Winnowmap2 errors. Most of the remaining 67 (=118-51) reads have multiple
|
||||
highly similar but not identical alignments. We are not sure what are real
|
||||
mapping errors.
|
||||
|
||||
The two benchmarks above only evaluate read mappings without variations.
|
||||
To measure the mapping accuracy in the presence of SVs (sim-sv), we reproduced
|
||||
the results by~\citep{Jain2020.11.01.363887}. Minimap2 v2.22 is as good as
|
||||
Winnowmap2 now. Note that we were setting the Sniffles mapping quality
|
||||
threshold to 10 in consistent with the benchmarks above. If we used the
|
||||
default threshold 20, v2.22 would miss additional 0.5\% SVs, suggesting
|
||||
minimap2 v2.22 could map variant reads correctly but with conservative mapping
|
||||
quality. This observation is more about the interaction between mappers and
|
||||
callers. Furthermore, the simulation here only considers a simple scenario in
|
||||
evolution. Non-allelic gene conversions, which happen often in segmental
|
||||
duplications~\citep{Harpak:2017aa}, would obscure the optimal mapping
|
||||
strategies. How much such simple SV simulation informs real-world SV calling
|
||||
remains a question.
|
||||
|
||||
To see if minimap2 v2.22 could improve long INDEL alignment, we ran dipcall on
|
||||
contig-to-reference alignments and focused on INDELs longer than 1kb
|
||||
(real-sv-1k). v2.22 is more sensitive at comparable specificity, confirming its
|
||||
advantage in more contiguous alignment. lra is supposed to handle long INDELs
|
||||
better, too. However, we could not get lra to work well with dipcall, so did
|
||||
not report the numbers.
|
||||
|
||||
Minimap2 spends most computing time on base alignment. As recent improvements
|
||||
in v2.22 incur little additional computing and do not change the base alignment
|
||||
algorithm, the new version has similar performance to older verions. It is
|
||||
consistently faster than Winnowmap2 by several times. Sometimes simple
|
||||
heuristics can be as effective as more sophisticated yet slower solutions.
|
||||
|
||||
\section*{Acknowledgements}
|
||||
We thank Arang Rhie and Chirag Jain for providing motivating examples where
|
||||
older minimap2 underperforms.
|
||||
|
||||
\paragraph{Funding\textcolon} This work is funded by NHGRI grant R01HG010040.
|
||||
|
||||
\bibliography{minimap2}
|
||||
|
||||
\end{document}
|
||||
Reference in New Issue
Block a user