mirror of
https://github.com/lh3/minimap2.git
synced 2026-09-25 07:18:12 +08:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3c28777e7e | ||
|
|
d8371a403b | ||
|
|
e5066c7976 | ||
|
|
f8381755f1 | ||
|
|
80d92c686f | ||
|
|
37a650f58c | ||
|
|
4ca5a951ca | ||
|
|
dd5b2c04b8 | ||
|
|
ca6f4dc5e2 | ||
|
|
de3c6ec646 | ||
|
|
e2542e6425 | ||
|
|
9bb4d2bed4 | ||
|
|
6d49eb690f | ||
|
|
bd0cba5012 | ||
|
|
370f3f8236 | ||
|
|
79c9cc186b | ||
|
|
ea4c8935bd | ||
|
|
3187782b1a | ||
|
|
005c9a1f6b | ||
|
|
1fd85be6e2 | ||
|
|
e616b0dacf | ||
|
|
b58b97423a | ||
|
|
df9e650346 | ||
|
|
7a540c37ca | ||
|
|
c19e3ccb86 | ||
|
|
94d171b01e | ||
|
|
ff312a2957 | ||
|
|
01ccedd5a0 | ||
|
|
819b3bf017 | ||
|
|
e88110463a | ||
|
|
fb81e150f2 | ||
|
|
d930ea94ad | ||
|
|
a832a42f6f | ||
|
|
a5411fc3c0 | ||
|
|
bd03d975fc | ||
|
|
9c3c4b1ce8 | ||
|
|
3542a3d153 | ||
|
|
75619c7b51 | ||
|
|
fbb9c0fcba | ||
|
|
af094640e5 | ||
|
|
d43f356ef9 | ||
|
|
38acd6617f | ||
|
|
3d351267a0 | ||
|
|
54a4718c9b | ||
|
|
dbc12b2838 | ||
|
|
2ed264db4e | ||
|
|
a8094ad859 | ||
|
|
1877818239 | ||
|
|
9ede5c4255 | ||
|
|
405511fe8d | ||
|
|
dd90d9dde6 | ||
|
|
a8c567b5e9 | ||
|
|
d9d3c0cc3f | ||
|
|
cbe8d61ca4 | ||
|
|
9d06cef13e | ||
|
|
924fc4d671 | ||
|
|
fc2d1e95b3 | ||
|
|
bbf0bb871b | ||
|
|
83e9b2e28c | ||
|
|
e816fd071c | ||
|
|
a955b1f31d | ||
|
|
54fa925e2e | ||
|
|
d4a396c5c3 | ||
|
|
bdf46f5786 | ||
|
|
f536b69b81 | ||
|
|
4b8b4418df | ||
|
|
ce30004e02 | ||
|
|
54f8e5f7d6 | ||
|
|
2857de7dbd | ||
|
|
e18935fbad | ||
|
|
74ebfb2532 | ||
|
|
a0cbe2e4d2 | ||
|
|
618d33515e | ||
|
|
a10d4f4496 | ||
|
|
1eea2fee11 | ||
|
|
1d346d56bc | ||
|
|
46750de966 | ||
|
|
7d8bbb74a8 | ||
|
|
c6db201b38 | ||
|
|
358a39850f | ||
|
|
fcb5d5e6eb | ||
|
|
95807a2224 | ||
|
|
a4c93e9377 | ||
|
|
7d69334e69 | ||
|
|
3e1ab2951d | ||
|
|
68179ed195 | ||
|
|
d1f4c8d232 | ||
|
|
8efe83b744 | ||
|
|
042c8d4d71 | ||
|
|
e4e1f7843b | ||
|
|
69e3629916 | ||
|
|
0cc3cdca27 | ||
|
|
8170693de3 | ||
|
|
e3d8c708ac | ||
|
|
119bdc6029 | ||
|
|
89d4d219cd | ||
|
|
f51ff1abac | ||
|
|
27b254ed6f | ||
|
|
c881b14ba5 | ||
|
|
f18dadb1c4 | ||
|
|
a83b8fe7cc | ||
|
|
c22bfe7722 | ||
|
|
12d441ea22 | ||
|
|
c7433c2811 | ||
|
|
5279377544 | ||
|
|
acab05781e | ||
|
|
98c23bc6d2 | ||
|
|
9b0ff2418c | ||
|
|
b6762503a9 | ||
|
|
9667468e89 | ||
|
|
ba60aac6f6 | ||
|
|
fcd4df2a73 | ||
|
|
0efc886012 | ||
|
|
940388f8e4 | ||
|
|
23d2674c39 | ||
|
|
a12673611f | ||
|
|
8140259974 | ||
|
|
f3e59fc2a0 | ||
|
|
fc2e1607d7 | ||
|
|
bc588c0eeb | ||
|
|
ab717023b6 | ||
|
|
9506e7ac3f | ||
|
|
ce03fbc275 | ||
|
|
98a3aa1b39 | ||
|
|
ae05f8485f | ||
|
|
ace990c381 | ||
|
|
e28a55be86 | ||
|
|
f8d46a7a30 | ||
|
|
4483f89ee5 | ||
|
|
f1b3c7ad06 | ||
|
|
180faa3594 | ||
|
|
704fbc6f5c | ||
|
|
fc24c8a348 | ||
|
|
e68d868806 | ||
|
|
c3d461e22a | ||
|
|
c41518ae85 | ||
|
|
819d843e3c | ||
|
|
5e7242303c | ||
|
|
a026c69b89 | ||
|
|
ea2042a577 | ||
|
|
1834b1fd42 | ||
|
|
7ced0f16a0 | ||
|
|
35732f3025 | ||
|
|
a6fab118c5 | ||
|
|
6ce0dd8b70 | ||
|
|
1d3c3eef03 | ||
|
|
01b98e8e52 | ||
|
|
16b8d50199 | ||
|
|
226fd6114c | ||
|
|
822ccd1733 | ||
|
|
f67849c9af | ||
|
|
b0b199f503 | ||
|
|
c2f07ff2ac | ||
|
|
cefd0d9f6c | ||
|
|
2a319c89aa | ||
|
|
85a5260408 | ||
|
|
6c2cbf7903 | ||
|
|
5aa4355ca8 | ||
|
|
6ed7263670 | ||
|
|
315795eefd | ||
|
|
fc6869a9e8 | ||
|
|
6252e5e367 | ||
|
|
843729df1e | ||
|
|
195c98fa46 | ||
|
|
a2e6659d9b | ||
|
|
e450f161bb | ||
|
|
15cade0f06 | ||
|
|
767556b6f0 | ||
|
|
50a26a60a6 | ||
|
|
31de4fd1bc | ||
|
|
e018caea32 | ||
|
|
7c02742fa8 | ||
|
|
a41f5d1eeb | ||
|
|
ed3d0eb328 | ||
|
|
e6d166a314 | ||
|
|
06fedaadd0 | ||
|
|
fe35e679e9 | ||
|
|
e25aa5ee74 | ||
|
|
3bde3450a0 | ||
|
|
36942ff711 | ||
|
|
d3a89d34d4 | ||
|
|
c8f0a35c40 | ||
|
|
fcaadc22b7 | ||
|
|
db37fc43a7 | ||
|
|
a8f1fa8ea3 | ||
|
|
b276772890 | ||
|
|
d0cff3eb36 | ||
|
|
ac334639ce | ||
|
|
546623dcb4 | ||
|
|
39bdd45875 | ||
|
|
aefa2c0d86 | ||
|
|
7ee62dae1d | ||
|
|
05a8a45d44 | ||
|
|
cc14d1afdf | ||
|
|
bb3048b2a0 | ||
|
|
5113ca2628 |
@@ -7,7 +7,8 @@ on:
|
|||||||
pull_request:
|
pull_request:
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
build:
|
build-linux-x8664:
|
||||||
|
name: Linux x86_64
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
@@ -15,7 +16,53 @@ jobs:
|
|||||||
|
|
||||||
steps:
|
steps:
|
||||||
- name: Checkout minimap2
|
- name: Checkout minimap2
|
||||||
uses: actions/checkout@v2
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
- name: Compile with ${{ matrix.compiler }}
|
- name: Compile with ${{ matrix.compiler }}
|
||||||
run: make CC=${{ matrix.compiler }}
|
run: |
|
||||||
|
make CC=${{ matrix.compiler }}
|
||||||
|
file minimap2 | grep x86-64
|
||||||
|
|
||||||
|
build-linux-aarch64:
|
||||||
|
name: Linux aarch64
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
strategy:
|
||||||
|
matrix:
|
||||||
|
compiler: [gcc]
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- name: Checkout
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Compile with ${{ matrix.compiler }}
|
||||||
|
uses: uraimo/run-on-arch-action@v3
|
||||||
|
with:
|
||||||
|
arch: aarch64
|
||||||
|
distro: ubuntu22.04
|
||||||
|
githubToken: ${{ github.token }}
|
||||||
|
dockerRunArgs: |
|
||||||
|
--volume "${PWD}:/minimap2"
|
||||||
|
install: |
|
||||||
|
apt-get update -q -y
|
||||||
|
apt-get install -q -y make ${{ matrix.compiler }} zlib1g-dev file
|
||||||
|
run: |
|
||||||
|
cd /minimap2
|
||||||
|
make CC=${{ matrix.compiler }} arm_neon=1 aarch64=1 -j
|
||||||
|
file minimap2 | grep aarch64
|
||||||
|
|
||||||
|
build-mac-arm64:
|
||||||
|
name: Mac ARM64
|
||||||
|
runs-on: macos-14
|
||||||
|
strategy:
|
||||||
|
matrix:
|
||||||
|
compiler: [clang]
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- name: Checkout minimap2
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Compile with ${{ matrix.compiler }}
|
||||||
|
run: |
|
||||||
|
make CC=${{ matrix.compiler }} arm_neon=1 aarch64=1 -j
|
||||||
|
file minimap2 | grep arm64
|
||||||
|
|
||||||
|
|||||||
-24
@@ -1,24 +0,0 @@
|
|||||||
matrix:
|
|
||||||
include:
|
|
||||||
- language: c
|
|
||||||
compiler: gcc
|
|
||||||
script: make
|
|
||||||
- language: c
|
|
||||||
compiler: clang
|
|
||||||
script: make
|
|
||||||
- arch: arm64
|
|
||||||
language: c
|
|
||||||
compiler: gcc
|
|
||||||
script: make arm_neon=1 aarch64=1
|
|
||||||
- language: python
|
|
||||||
python: "2.7"
|
|
||||||
before_install: pip install cython
|
|
||||||
script: python setup.py build_ext
|
|
||||||
- language: python
|
|
||||||
python: "3.5"
|
|
||||||
before_install: pip install cython
|
|
||||||
script: python setup.py build_ext
|
|
||||||
- language: python
|
|
||||||
python: "3.9"
|
|
||||||
before_install: pip install cython
|
|
||||||
script: python setup.py build_ext
|
|
||||||
@@ -2,7 +2,7 @@
|
|||||||
|
|
||||||
Without `-a`, `-c` or `--cs`, minimap2 only finds *approximate* mapping
|
Without `-a`, `-c` or `--cs`, minimap2 only finds *approximate* mapping
|
||||||
locations without detailed base alignment. In particular, the start and end
|
locations without detailed base alignment. In particular, the start and end
|
||||||
positions of the alignment are impricise. With one of those options, minimap2
|
positions of the alignment are imprecise. With one of those options, minimap2
|
||||||
will perform base alignment, which is generally more accurate but is much
|
will perform base alignment, which is generally more accurate but is much
|
||||||
slower.
|
slower.
|
||||||
|
|
||||||
|
|||||||
@@ -2,12 +2,16 @@ CFLAGS= -g -Wall -O2 -Wc++-compat #-Wextra
|
|||||||
CPPFLAGS= -DHAVE_KALLOC
|
CPPFLAGS= -DHAVE_KALLOC
|
||||||
INCLUDES=
|
INCLUDES=
|
||||||
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o \
|
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o \
|
||||||
lchain.o align.o hit.o seed.o map.o format.o pe.o esterr.o splitidx.o \
|
lchain.o align.o hit.o seed.o jump.o map.o format.o pe.o esterr.o splitidx.o \
|
||||||
ksw2_ll_sse.o
|
ksw2_ll_sse.o
|
||||||
PROG= minimap2
|
PROG= minimap2
|
||||||
PROG_EXTRA= sdust minimap2-lite
|
PROG_EXTRA= sdust minimap2-lite
|
||||||
LIBS= -lm -lz -lpthread
|
LIBS= -lm -lz -lpthread
|
||||||
|
|
||||||
|
ifneq ($(aarch64),)
|
||||||
|
arm_neon=1
|
||||||
|
endif
|
||||||
|
|
||||||
ifeq ($(arm_neon),) # if arm_neon is not defined
|
ifeq ($(arm_neon),) # if arm_neon is not defined
|
||||||
ifeq ($(sse2only),) # if sse2only is not defined
|
ifeq ($(sse2only),) # if sse2only is not defined
|
||||||
OBJS+=ksw2_extz2_sse41.o ksw2_extd2_sse41.o ksw2_exts2_sse41.o ksw2_extz2_sse2.o ksw2_extd2_sse2.o ksw2_exts2_sse2.o ksw2_dispatch.o
|
OBJS+=ksw2_extz2_sse41.o ksw2_extd2_sse41.o ksw2_exts2_sse41.o ksw2_extz2_sse2.o ksw2_extd2_sse2.o ksw2_exts2_sse2.o ksw2_dispatch.o
|
||||||
@@ -26,12 +30,12 @@ endif
|
|||||||
|
|
||||||
ifneq ($(asan),)
|
ifneq ($(asan),)
|
||||||
CFLAGS+=-fsanitize=address
|
CFLAGS+=-fsanitize=address
|
||||||
LIBS+=-fsanitize=address
|
LIBS+=-fsanitize=address -ldl
|
||||||
endif
|
endif
|
||||||
|
|
||||||
ifneq ($(tsan),)
|
ifneq ($(tsan),)
|
||||||
CFLAGS+=-fsanitize=thread
|
CFLAGS+=-fsanitize=thread
|
||||||
LIBS+=-fsanitize=thread
|
LIBS+=-fsanitize=thread -ldl
|
||||||
endif
|
endif
|
||||||
|
|
||||||
.PHONY:all extra clean depend
|
.PHONY:all extra clean depend
|
||||||
@@ -98,7 +102,7 @@ ksw2_exts2_neon.o:ksw2_exts2_sse.c ksw2.h kalloc.h
|
|||||||
# other non-file targets
|
# other non-file targets
|
||||||
|
|
||||||
clean:
|
clean:
|
||||||
rm -fr gmon.out *.o a.out $(PROG) $(PROG_EXTRA) *~ *.a *.dSYM build dist mappy*.so mappy.c python/mappy.c mappy.egg*
|
rm -fr gmon.out *.o a.out $(PROG) $(PROG_EXTRA) *~ *.a *.dSYM build dist mappy*.so mappy.c python/mappy.c mappy.egg* .eggs
|
||||||
|
|
||||||
depend:
|
depend:
|
||||||
(LC_ALL=C; export LC_ALL; makedepend -Y -- $(CFLAGS) $(CPPFLAGS) -- *.c)
|
(LC_ALL=C; export LC_ALL; makedepend -Y -- $(CFLAGS) $(CPPFLAGS) -- *.c)
|
||||||
@@ -111,8 +115,9 @@ esterr.o: mmpriv.h minimap.h bseq.h kseq.h
|
|||||||
example.o: minimap.h kseq.h
|
example.o: minimap.h kseq.h
|
||||||
format.o: kalloc.h mmpriv.h minimap.h bseq.h kseq.h
|
format.o: kalloc.h mmpriv.h minimap.h bseq.h kseq.h
|
||||||
hit.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h khash.h
|
hit.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h khash.h
|
||||||
index.o: kthread.h bseq.h minimap.h mmpriv.h kseq.h kvec.h kalloc.h khash.h
|
index.o: kthread.h bseq.h minimap.h mmpriv.h kseq.h ksw2.h kalloc.h kvec.h
|
||||||
index.o: ksort.h
|
index.o: khash.h ksort.h
|
||||||
|
jump.o: mmpriv.h minimap.h bseq.h kseq.h
|
||||||
kalloc.o: kalloc.h
|
kalloc.o: kalloc.h
|
||||||
ksw2_extd2_sse.o: ksw2.h kalloc.h
|
ksw2_extd2_sse.o: ksw2.h kalloc.h
|
||||||
ksw2_exts2_sse.o: ksw2.h kalloc.h
|
ksw2_exts2_sse.o: ksw2.h kalloc.h
|
||||||
|
|||||||
@@ -1,3 +1,236 @@
|
|||||||
|
Release 2.31-r1302 (19 May 2026)
|
||||||
|
--------------------------------
|
||||||
|
|
||||||
|
Notable changes to minimap2:
|
||||||
|
|
||||||
|
* Bugfix: supplementary and secondary alignments were occasionally flagged
|
||||||
|
incorrectly.
|
||||||
|
|
||||||
|
* Bugfix: Smith-Waterman alignment for inversion alignment led to an
|
||||||
|
out-of-bound access in rare cases.
|
||||||
|
|
||||||
|
Changes to paftools.js:
|
||||||
|
|
||||||
|
* New feature: new `sim2bed` subcommand to get a BED file from simulated
|
||||||
|
reads.
|
||||||
|
|
||||||
|
* New feature: new `badread2fa` subcommand to format reads simulated by
|
||||||
|
the Badread simulator.
|
||||||
|
|
||||||
|
Change to the python binding:
|
||||||
|
|
||||||
|
* New feature: mappy optionally writes the `ds` tag.
|
||||||
|
|
||||||
|
* Bugfix: a use-after-free error (#1345)
|
||||||
|
|
||||||
|
The two bugs in minimap2 had existed for years. They were caught by Jeremy Wang
|
||||||
|
at UNC when he ported minimap2 to Rust. Due to the two bug fixes, this version
|
||||||
|
occasionally produces alignment different from the last version.
|
||||||
|
|
||||||
|
(2.31: 19 May 2026, r1302)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
Release 2.30-r1287 (15 June 2025)
|
||||||
|
---------------------------------
|
||||||
|
|
||||||
|
Notable changes:
|
||||||
|
|
||||||
|
* Improvement: consolidated `--spsc`.
|
||||||
|
|
||||||
|
* Deprecation: subcommands `splice2bed`, `gff2bed`, `gff2junc`, `junceval` and
|
||||||
|
`exoneval` in `paftools.js` are deprecated by minigff. They will remain
|
||||||
|
indefinitely for backward compatibility.
|
||||||
|
|
||||||
|
(2.30: 15 June 2025, r1287)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
Release 2.29-r1283 (18 April 2025)
|
||||||
|
----------------------------------
|
||||||
|
|
||||||
|
Notable changes to minimap2:
|
||||||
|
|
||||||
|
* New feature: added the `splice:sr` preset for short RNA-seq read alignment.
|
||||||
|
Users may use `-j` to specify known gene annotation to improve spliced
|
||||||
|
alignment close to the ends of short reads. Also added `--write-junc` and
|
||||||
|
`--pass1` for 2-pass short-read RNA-seq alignment.
|
||||||
|
|
||||||
|
* Experimental feature: read splice scores from a file specified by `--spsc`
|
||||||
|
and consider the scores during base alignment. The feature makes it possible
|
||||||
|
to apply advanced splice models and to improve spliced alignment.
|
||||||
|
|
||||||
|
* Change: adjusted the mapping quality calculation for spliced alignment.
|
||||||
|
|
||||||
|
* Bugfixes: a) missing overlap alignment when base alignment is requested
|
||||||
|
(#969); b) incorrect summary information for long genomes (#1192); c)
|
||||||
|
missing parameter check for `--score-N` (#1226).
|
||||||
|
|
||||||
|
* Improvement: a) warn about absent junction files (#1229); b) report an error
|
||||||
|
if a wrong preset prefixed with "splice" is specified (#589).
|
||||||
|
|
||||||
|
Notable changes to mappy:
|
||||||
|
|
||||||
|
* Improvement: allow passing read name (#1260)
|
||||||
|
|
||||||
|
* Improvement: exposed score for ambiguous bases (#1240)
|
||||||
|
|
||||||
|
Minimap2 now supports short/long genomic/RNA-seq read alignment along with
|
||||||
|
contig alignment and all-vs-all read overlapping. It produces identical genomic
|
||||||
|
long-read or contig alignment to v2.27. Short genomic read alignment and the
|
||||||
|
mapping quality of long RNA-seq read alignment may slightly differ in very rare
|
||||||
|
cases.
|
||||||
|
|
||||||
|
(2.29: 18 April 2025, r1283)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
Release 2.28-r1209 (27 March 2024)
|
||||||
|
----------------------------------
|
||||||
|
|
||||||
|
Notable changes to minimap2:
|
||||||
|
|
||||||
|
* Bugfix: `--MD` was not working properly due to the addition of `--ds` in the
|
||||||
|
last release (#1181 and #1182).
|
||||||
|
|
||||||
|
* New feature: added an experimental preset `lq:hqae` for aligning accurate
|
||||||
|
long reads back to their assembly. It has been observed that `map-hifi` and
|
||||||
|
`lr:hq` may produce many wrong alignments around centromeres when accurate
|
||||||
|
long reads (PacBio HiFi or Nanopore duplex/Q20+) are mapped to a diploid
|
||||||
|
assembly constructed from them. This new preset produces much more accurate
|
||||||
|
alignment. It is still experimental and may be subjective to changes in
|
||||||
|
future.
|
||||||
|
|
||||||
|
* Change: reduced the default `--cap-kalloc` to 500m to lower the peak
|
||||||
|
memory consumption (#855).
|
||||||
|
|
||||||
|
Notable changes to mappy:
|
||||||
|
|
||||||
|
* Bugfix: mappy option struct was out of sync with minimap2 (#1177).
|
||||||
|
|
||||||
|
Minimap2 should output identical alignments to v2.27.
|
||||||
|
|
||||||
|
(2.28: 27 March 2024, r1209)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
Release 2.27-r1193 (12 March 2024)
|
||||||
|
----------------------------------
|
||||||
|
|
||||||
|
Notable changes to minimap2:
|
||||||
|
|
||||||
|
* New feature: added the `lr:hq` preset for accurate long reads at ~1% error
|
||||||
|
rate. This was suggested by Oxford Nanopore developers (#1127). It is not
|
||||||
|
clear if this preset also works well for PacBio HiFi reads.
|
||||||
|
|
||||||
|
* New feature: added the `map-iclr` preset for Illumina Complete Long Reads
|
||||||
|
(#1069), provided by Illumina developers.
|
||||||
|
|
||||||
|
* New feature: added option `-b` to specify mismatch penalty for base
|
||||||
|
transitions (i.e. A-to-G or C-to-T changes).
|
||||||
|
|
||||||
|
* New feature: added option `--ds` to generate a new `ds:Z` tag that
|
||||||
|
indicates uncertainty in INDEL positions. It is an extension to `cs`. The
|
||||||
|
`mgutils-es6.js` script in minigraph parses `ds`.
|
||||||
|
|
||||||
|
* Bugfix: avoided a NULL pointer dereference (#1154). This would not have an
|
||||||
|
effect on most systems but would still be good to fix.
|
||||||
|
|
||||||
|
* Bugfix: reverted the value of `ms:i` to pre-2.22 versions (#1146). This was
|
||||||
|
an oversight. See fcd4df2 for details.
|
||||||
|
|
||||||
|
Notable changes to paftools.js and mappy:
|
||||||
|
|
||||||
|
* New feature: expose `bw_long` to mappy's Aligner class (#1124).
|
||||||
|
|
||||||
|
* Bugfix: fixed several compatibility issues with k8 v1.0 (#1161 and #1166).
|
||||||
|
Subcommands "call", "pbsim2fq" and "mason2fq" were not working with v1.0.
|
||||||
|
|
||||||
|
Minimap2 should output identical alignments to v2.26, except the ms tag.
|
||||||
|
|
||||||
|
(2.27: 12 March 2024, r1193)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
Release 2.26-r1175 (29 April 2023)
|
||||||
|
----------------------------------
|
||||||
|
|
||||||
|
Fixed the broken Python package. This is the only change.
|
||||||
|
|
||||||
|
(2.26: 25 April 2023, r1173)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
Release 2.25-r1173 (25 April 2023)
|
||||||
|
----------------------------------
|
||||||
|
|
||||||
|
Notable changes:
|
||||||
|
|
||||||
|
* Improvement: use the miniprot splice model for RNA-seq alignment by default.
|
||||||
|
This model considers non-GT-AG splice sites and leads to slightly higher
|
||||||
|
(<0.1%) accuracy and sensitivity on real human data.
|
||||||
|
|
||||||
|
* Change: increased the default `-I` to `8G` such that minimap2 would create a
|
||||||
|
uni-part index for a pair of mammalian genomes. This change may increase the
|
||||||
|
memory for all-vs-all read overlap alignment given large datasets.
|
||||||
|
|
||||||
|
* New feature: output the sequences in secondary alignments with option
|
||||||
|
`--secondary-seq` (#687).
|
||||||
|
|
||||||
|
* Bugfix: --rmq was not parsed correctly (#1010)
|
||||||
|
|
||||||
|
* Bugfix: possibly incorrect coordinate when applying end bonus to the target
|
||||||
|
sequence (#1025). This is a ksw2 bug. It does not affect minimap2 as
|
||||||
|
minimap2 is not using the affected feature.
|
||||||
|
|
||||||
|
* Improvement: incorporated several changes for better compatibility with
|
||||||
|
Windows (#1051) and for minimap2 integration at Oxford Nanopore Technologies
|
||||||
|
(#1048 and #1033).
|
||||||
|
|
||||||
|
* Improvement: output the HD-line in SAM output (#1019).
|
||||||
|
|
||||||
|
* Improvement: check minimap2 index file in mappy to prevent segmentation
|
||||||
|
fault for certain indices (#1008).
|
||||||
|
|
||||||
|
For genomic sequences, minimap2 should give identical output to v2.24.
|
||||||
|
Long-read RNA-seq alignment may occasionally differ from previous versions.
|
||||||
|
|
||||||
|
(2.25: 25 April 2023, r1173)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
Release 2.24-r1122 (26 December 2021)
|
||||||
|
-------------------------------------
|
||||||
|
|
||||||
|
This release improves alignment around long poorly aligned regions. Older
|
||||||
|
minimap2 may chain through such regions in rare cases which may result in
|
||||||
|
missing alignments later. The issue has become worse since the the change of
|
||||||
|
the chaining algorithm in v2.19. v2.23 implements an incomplete remedy. This
|
||||||
|
release provides a better solution with a X-drop-like heuristic and by enabling
|
||||||
|
two-bandwidth chaining in the assembly mode.
|
||||||
|
|
||||||
|
(2.24: 26 December 2021, r1122)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
Release 2.23-r1111 (18 November 2021)
|
||||||
|
-------------------------------------
|
||||||
|
|
||||||
|
Notable changes:
|
||||||
|
|
||||||
|
* Bugfix: fixed missing alignments around long inversions (#806 and #816).
|
||||||
|
This bug affected v2.19 through v2.22.
|
||||||
|
|
||||||
|
* Improvement: avoid extremely long mapping time for pathologic reads with
|
||||||
|
highly repeated k-mers not in the reference (#771). Use --q-occ-frac=0
|
||||||
|
to disable the new heuristic.
|
||||||
|
|
||||||
|
* Change: use --cap-kalloc=1g by default.
|
||||||
|
|
||||||
|
(2.23: 18 November 2021, r1111)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Release 2.22-r1101 (7 August 2021)
|
Release 2.22-r1101 (7 August 2021)
|
||||||
----------------------------------
|
----------------------------------
|
||||||
|
|
||||||
|
|||||||
@@ -3,6 +3,7 @@
|
|||||||
[](https://pypi.python.org/pypi/mappy)
|
[](https://pypi.python.org/pypi/mappy)
|
||||||
[](https://github.com/lh3/minimap2/actions)
|
[](https://github.com/lh3/minimap2/actions)
|
||||||
## <a name="started"></a>Getting Started
|
## <a name="started"></a>Getting Started
|
||||||
|
**ALERT:** `minimap2.com` is a [phishing site](https://github.com/lh3/minimap2/issues/1316). Please don't use anything from that website.
|
||||||
```sh
|
```sh
|
||||||
git clone https://github.com/lh3/minimap2
|
git clone https://github.com/lh3/minimap2
|
||||||
cd minimap2 && make
|
cd minimap2 && make
|
||||||
@@ -14,13 +15,15 @@ cd minimap2 && make
|
|||||||
# use presets (no test data)
|
# use presets (no test data)
|
||||||
./minimap2 -ax map-pb ref.fa pacbio.fq.gz > aln.sam # PacBio CLR genomic reads
|
./minimap2 -ax map-pb ref.fa pacbio.fq.gz > aln.sam # PacBio CLR genomic reads
|
||||||
./minimap2 -ax map-ont ref.fa ont.fq.gz > aln.sam # Oxford Nanopore genomic reads
|
./minimap2 -ax map-ont ref.fa ont.fq.gz > aln.sam # Oxford Nanopore genomic reads
|
||||||
./minimap2 -ax map-hifi ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.19 or later)
|
./minimap2 -ax map-hifi ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.19+)
|
||||||
./minimap2 -ax asm20 ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.18 or earlier)
|
./minimap2 -ax lr:hq ref.fa ont-Q20.fq.gz > aln.sam # Nanopore Q20 genomic reads (v2.27+)
|
||||||
./minimap2 -ax sr ref.fa read1.fa read2.fa > aln.sam # short genomic paired-end reads
|
./minimap2 -ax sr ref.fa read1.fa read2.fa > aln.sam # short genomic paired-end reads
|
||||||
./minimap2 -ax splice ref.fa rna-reads.fa > aln.sam # spliced long reads (strand unknown)
|
./minimap2 -ax splice ref.fa rna-reads.fa > aln.sam # spliced long reads (strand unknown)
|
||||||
./minimap2 -ax splice -uf -k14 ref.fa reads.fa > aln.sam # noisy Nanopore Direct RNA-seq
|
./minimap2 -ax splice -uf -k14 ref.fa reads.fa > aln.sam # noisy Nanopore direct RNA-seq
|
||||||
./minimap2 -ax splice:hq -uf ref.fa query.fa > aln.sam # Final PacBio Iso-seq or traditional cDNA
|
./minimap2 -ax splice:hq -uf ref.fa query.fa > aln.sam # PacBio Kinnex/Iso-seq (RNA-seq)
|
||||||
./minimap2 -ax splice --junc-bed anno.bed12 ref.fa query.fa > aln.sam # prioritize on annotated junctions
|
./minimap2 -ax splice --junc-bed=anno.bed12 ref.fa query.fa > aln.sam # use annotated junctions
|
||||||
|
./minimap2 -ax splice:sr ref.fa r1.fq r2.fq > aln.sam # short-read RNA-seq (v2.29+)
|
||||||
|
./minimap2 -ax splice:sr -j anno.bed12 ref.fa r1.fq r2.fq > aln.sam
|
||||||
./minimap2 -cx asm5 asm1.fa asm2.fa > aln.paf # intra-species asm-to-asm alignment
|
./minimap2 -cx asm5 asm1.fa asm2.fa > aln.paf # intra-species asm-to-asm alignment
|
||||||
./minimap2 -x ava-pb reads.fa reads.fa > overlaps.paf # PacBio read overlap
|
./minimap2 -x ava-pb reads.fa reads.fa > overlaps.paf # PacBio read overlap
|
||||||
./minimap2 -x ava-ont reads.fa reads.fa > overlaps.paf # Nanopore read overlap
|
./minimap2 -x ava-ont reads.fa reads.fa > overlaps.paf # Nanopore read overlap
|
||||||
@@ -38,7 +41,8 @@ man ./minimap2.1
|
|||||||
- [Map long noisy genomic reads](#map-long-genomic)
|
- [Map long noisy genomic reads](#map-long-genomic)
|
||||||
- [Map long mRNA/cDNA reads](#map-long-splice)
|
- [Map long mRNA/cDNA reads](#map-long-splice)
|
||||||
- [Find overlaps between long reads](#long-overlap)
|
- [Find overlaps between long reads](#long-overlap)
|
||||||
- [Map short accurate genomic reads](#short-genomic)
|
- [Map short genomic reads](#short-genomic)
|
||||||
|
- [Map short RNA-seq reads](#short-rna-seq)
|
||||||
- [Full genome/assembly alignment](#full-genome)
|
- [Full genome/assembly alignment](#full-genome)
|
||||||
- [Advanced features](#advanced)
|
- [Advanced features](#advanced)
|
||||||
- [Working with >65535 CIGAR operations](#long-cigar)
|
- [Working with >65535 CIGAR operations](#long-cigar)
|
||||||
@@ -74,8 +78,8 @@ Detailed evaluations are available from the [minimap2 paper][doi] or the
|
|||||||
Minimap2 is optimized for x86-64 CPUs. You can acquire precompiled binaries from
|
Minimap2 is optimized for x86-64 CPUs. You can acquire precompiled binaries from
|
||||||
the [release page][release] with:
|
the [release page][release] with:
|
||||||
```sh
|
```sh
|
||||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.22/minimap2-2.22_x64-linux.tar.bz2 | tar -jxvf -
|
curl -L https://github.com/lh3/minimap2/releases/download/v2.31/minimap2-2.31_x64-linux.tar.bz2 | tar -jxvf -
|
||||||
./minimap2-2.22_x64-linux/minimap2
|
./minimap2-2.31_x64-linux/minimap2
|
||||||
```
|
```
|
||||||
If you want to compile from the source, you need to have a C compiler, GNU make
|
If you want to compile from the source, you need to have a C compiler, GNU make
|
||||||
and zlib development files installed. Then type `make` in the source code
|
and zlib development files installed. Then type `make` in the source code
|
||||||
@@ -139,12 +143,15 @@ parameters at the same time. The default setting is the same as `map-ont`.
|
|||||||
```sh
|
```sh
|
||||||
minimap2 -ax map-pb ref.fa pacbio-reads.fq > aln.sam # for PacBio CLR reads
|
minimap2 -ax map-pb ref.fa pacbio-reads.fq > aln.sam # for PacBio CLR reads
|
||||||
minimap2 -ax map-ont ref.fa ont-reads.fq > aln.sam # for Oxford Nanopore reads
|
minimap2 -ax map-ont ref.fa ont-reads.fq > aln.sam # for Oxford Nanopore reads
|
||||||
|
minimap2 -ax map-iclr ref.fa iclr-reads.fq > aln.sam # for Illumina Complete Long Reads
|
||||||
```
|
```
|
||||||
The difference between `map-pb` and `map-ont` is that `map-pb` uses
|
The difference between `map-pb` and `map-ont` is that `map-pb` uses
|
||||||
homopolymer-compressed (HPC) minimizers as seeds, while `map-ont` uses ordinary
|
homopolymer-compressed (HPC) minimizers as seeds, while `map-ont` uses ordinary
|
||||||
minimizers as seeds. Emperical evaluation suggests HPC minimizers improve
|
minimizers as seeds. Empirical evaluation suggests HPC minimizers improve
|
||||||
performance and sensitivity when aligning PacBio CLR reads, but hurt when aligning
|
performance and sensitivity when aligning PacBio CLR reads, but hurt when aligning
|
||||||
Nanopore reads.
|
Nanopore reads. `map-iclr` uses an adjusted alignment scoring matrix that
|
||||||
|
accounts for the low overall error rate in the reads, with transversion errors
|
||||||
|
being less frequent than transitions.
|
||||||
|
|
||||||
#### <a name="map-long-splice"></a>Map long mRNA/cDNA reads
|
#### <a name="map-long-splice"></a>Map long mRNA/cDNA reads
|
||||||
|
|
||||||
@@ -168,9 +175,8 @@ or the last exons.
|
|||||||
|
|
||||||
Minimap2 rates an alignment by the score of the max-scoring sub-segment,
|
Minimap2 rates an alignment by the score of the max-scoring sub-segment,
|
||||||
*excluding* introns, and marks the best alignment as primary in SAM. When a
|
*excluding* introns, and marks the best alignment as primary in SAM. When a
|
||||||
spliced gene also has unspliced pseudogenes, minimap2 does not intentionally
|
spliced gene also has unspliced pseudogenes, minimap2 slightly prefers
|
||||||
prefer spliced alignment, though in practice it more often marks the spliced
|
the spliced alignment. By default, minimap2 outputs up to five secondary
|
||||||
alignment as the primary. By default, minimap2 outputs up to five secondary
|
|
||||||
alignments (i.e. likely pseudogenes in the context of RNA-seq mapping). This
|
alignments (i.e. likely pseudogenes in the context of RNA-seq mapping). This
|
||||||
can be tuned with option **-N**.
|
can be tuned with option **-N**.
|
||||||
|
|
||||||
@@ -201,6 +207,10 @@ bonus score (tuned by `--junc-bonus`) if an aligned junction matches a junction
|
|||||||
in the annotation. Option `--junc-bed` also takes 5-column BED, including the
|
in the annotation. Option `--junc-bed` also takes 5-column BED, including the
|
||||||
strand field. In this case, each line indicates an oriented junction.
|
strand field. In this case, each line indicates an oriented junction.
|
||||||
|
|
||||||
|
**Note:** `--junc-bed` is intended for long noisy RNA-seq reads only.
|
||||||
|
Applying the option to short RNA-seq reads would increase run time with little
|
||||||
|
improvement to junction accuracy.
|
||||||
|
|
||||||
#### <a name="long-overlap"></a>Find overlaps between long reads
|
#### <a name="long-overlap"></a>Find overlaps between long reads
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
@@ -213,7 +223,7 @@ the overlapping mode because it is slow and may produce false positive
|
|||||||
overlaps. However, if performance is not a concern, you may try to add `-a` or
|
overlaps. However, if performance is not a concern, you may try to add `-a` or
|
||||||
`-c` anyway.
|
`-c` anyway.
|
||||||
|
|
||||||
#### <a name="short-genomic"></a>Map short accurate genomic reads
|
#### <a name="short-genomic"></a>Map short genomic reads
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
minimap2 -ax sr ref.fa reads-se.fq > aln.sam # single-end alignment
|
minimap2 -ax sr ref.fa reads-se.fq > aln.sam # single-end alignment
|
||||||
@@ -226,8 +236,18 @@ be paired if they are adjacent in the input stream and have the same name (with
|
|||||||
the `/[0-9]` suffix trimmed if present). Single- and paired-end reads can be
|
the `/[0-9]` suffix trimmed if present). Single- and paired-end reads can be
|
||||||
mixed.
|
mixed.
|
||||||
|
|
||||||
Minimap2 does not work well with short spliced reads. There are many capable
|
#### <a name="short-rna-seq"></a>Map short RNA-seq reads
|
||||||
RNA-seq mappers for short reads.
|
|
||||||
|
```sh
|
||||||
|
minimap2 -ax splice:sr ref.fa reads-se.fq.gz > aln.sam # single-end
|
||||||
|
minimap2 -ax splice:sr ref.fa r1.fq.gz r2.fq.gz > aln.sam # paired-end
|
||||||
|
minimap2 -ax splice:sr -j anno.bed ref.fa r1.fq r2.fq > aln.sam # use annotation
|
||||||
|
# 2-pass alignment
|
||||||
|
minimap2 -x splice:sr -j anno.bed --write-junc ref.fa r1.fq r2.fq > junc.bed
|
||||||
|
minimap2 -ax splice:sr -j anno.bed --pass1=junc.bed ref.fa r1.fq r2.fq > aln.sam
|
||||||
|
```
|
||||||
|
The new preset `splice:sr` was added in v2.29. It functions similarly to `sr`
|
||||||
|
except that it performs spliced alignment.
|
||||||
|
|
||||||
#### <a name="full-genome"></a>Full genome/assembly alignment
|
#### <a name="full-genome"></a>Full genome/assembly alignment
|
||||||
|
|
||||||
@@ -350,6 +370,11 @@ If you use minimap2 in your work, please cite:
|
|||||||
> Li, H. (2018). Minimap2: pairwise alignment for nucleotide sequences.
|
> Li, H. (2018). Minimap2: pairwise alignment for nucleotide sequences.
|
||||||
> *Bioinformatics*, **34**:3094-3100. [doi:10.1093/bioinformatics/bty191][doi]
|
> *Bioinformatics*, **34**:3094-3100. [doi:10.1093/bioinformatics/bty191][doi]
|
||||||
|
|
||||||
|
and/or:
|
||||||
|
|
||||||
|
> Li, H. (2021). New strategies to improve minimap2 alignment accuracy.
|
||||||
|
> *Bioinformatics*, **37**:4572-4574. [doi:10.1093/bioinformatics/btab705][doi2]
|
||||||
|
|
||||||
## <a name="dguide"></a>Developers' Guide
|
## <a name="dguide"></a>Developers' Guide
|
||||||
|
|
||||||
Minimap2 is not only a command line tool, but also a programming library.
|
Minimap2 is not only a command line tool, but also a programming library.
|
||||||
@@ -399,5 +424,6 @@ mappy` or [from BioConda][mappyconda] via `conda install -c bioconda mappy`.
|
|||||||
[manpage]: https://lh3.github.io/minimap2/minimap2.html
|
[manpage]: https://lh3.github.io/minimap2/minimap2.html
|
||||||
[manpage-cs]: https://lh3.github.io/minimap2/minimap2.html#10
|
[manpage-cs]: https://lh3.github.io/minimap2/minimap2.html#10
|
||||||
[doi]: https://doi.org/10.1093/bioinformatics/bty191
|
[doi]: https://doi.org/10.1093/bioinformatics/bty191
|
||||||
[smide]: https://github.com/nemequ/simde
|
[doi2]: https://doi.org/10.1093/bioinformatics/btab705
|
||||||
|
[simde]: https://github.com/nemequ/simde
|
||||||
[unimap]: https://github.com/lh3/unimap
|
[unimap]: https://github.com/lh3/unimap
|
||||||
|
|||||||
@@ -6,6 +6,8 @@
|
|||||||
#include "mmpriv.h"
|
#include "mmpriv.h"
|
||||||
#include "ksw2.h"
|
#include "ksw2.h"
|
||||||
|
|
||||||
|
#define MM_MAX_QLEN_FLANK 100
|
||||||
|
|
||||||
static void ksw_gen_simple_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t sc_ambi)
|
static void ksw_gen_simple_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t sc_ambi)
|
||||||
{
|
{
|
||||||
int i, j;
|
int i, j;
|
||||||
@@ -21,6 +23,18 @@ static void ksw_gen_simple_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t sc
|
|||||||
mat[(m - 1) * m + j] = sc_ambi;
|
mat[(m - 1) * m + j] = sc_ambi;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static void ksw_gen_ts_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t transition, int8_t sc_ambi)
|
||||||
|
{
|
||||||
|
assert(m == 5);
|
||||||
|
ksw_gen_simple_mat(m, mat, a, b, sc_ambi);
|
||||||
|
if (transition == 0 || transition == b) return;
|
||||||
|
transition = transition > 0? -transition : transition;
|
||||||
|
mat[0 * m + 2] = transition; // A->G
|
||||||
|
mat[1 * m + 3] = transition; // C->T
|
||||||
|
mat[2 * m + 0] = transition; // G->A
|
||||||
|
mat[3 * m + 1] = transition; // T->C
|
||||||
|
}
|
||||||
|
|
||||||
static inline void mm_seq_rev(uint32_t len, uint8_t *seq)
|
static inline void mm_seq_rev(uint32_t len, uint8_t *seq)
|
||||||
{
|
{
|
||||||
uint32_t i;
|
uint32_t i;
|
||||||
@@ -246,7 +260,7 @@ static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *ts
|
|||||||
if (p == 0) return;
|
if (p == 0) return;
|
||||||
mm_fix_cigar(r, qseq, tseq, &qshift, &tshift);
|
mm_fix_cigar(r, qseq, tseq, &qshift, &tshift);
|
||||||
qseq += qshift, tseq += tshift; // qseq and tseq may be shifted due to the removal of leading I/D
|
qseq += qshift, tseq += tshift; // qseq and tseq may be shifted due to the removal of leading I/D
|
||||||
r->blen = r->mlen = 0;
|
r->blen = r->mlen = 0, r->is_spliced = 0;
|
||||||
for (k = 0; k < p->n_cigar; ++k) {
|
for (k = 0; k < p->n_cigar; ++k) {
|
||||||
uint32_t op = p->cigar[k]&0xf, len = p->cigar[k]>>4;
|
uint32_t op = p->cigar[k]&0xf, len = p->cigar[k]>>4;
|
||||||
if (op == MM_CIGAR_MATCH) {
|
if (op == MM_CIGAR_MATCH) {
|
||||||
@@ -280,17 +294,16 @@ static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *ts
|
|||||||
if (s < 0) s = 0;
|
if (s < 0) s = 0;
|
||||||
toff += len;
|
toff += len;
|
||||||
} else if (op == MM_CIGAR_N_SKIP) {
|
} else if (op == MM_CIGAR_N_SKIP) {
|
||||||
toff += len;
|
r->is_spliced = 1, toff += len;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
p->dp_max = (int32_t)(max + .499);
|
p->dp_max = p->dp_max0 = (int32_t)(max + .499);
|
||||||
assert(qoff == r->qe - r->qs && toff == r->re - r->rs);
|
assert(qoff == r->qe - r->qs && toff == r->re - r->rs);
|
||||||
if (is_eqx) mm_update_cigar_eqx(r, qseq, tseq); // NB: it has to be called here as changes to qseq and tseq are not returned
|
if (is_eqx) mm_update_cigar_eqx(r, qseq, tseq); // NB: it has to be called here as changes to qseq and tseq are not returned
|
||||||
}
|
}
|
||||||
|
|
||||||
static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, uint32_t *cigar) // TODO: this calls the libc realloc()
|
void mm_enlarge_cigar(mm_reg1_t *r, uint32_t n_cigar) // TODO: this calls the libc realloc()
|
||||||
{
|
{
|
||||||
mm_extra_t *p;
|
|
||||||
if (n_cigar == 0) return;
|
if (n_cigar == 0) return;
|
||||||
if (r->p == 0) {
|
if (r->p == 0) {
|
||||||
uint32_t capacity = n_cigar + sizeof(mm_extra_t)/4;
|
uint32_t capacity = n_cigar + sizeof(mm_extra_t)/4;
|
||||||
@@ -302,6 +315,13 @@ static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, uint32_t *cigar) //
|
|||||||
kroundup32(r->p->capacity);
|
kroundup32(r->p->capacity);
|
||||||
r->p = (mm_extra_t*)realloc(r->p, r->p->capacity * 4);
|
r->p = (mm_extra_t*)realloc(r->p, r->p->capacity * 4);
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, const uint32_t *cigar)
|
||||||
|
{
|
||||||
|
mm_extra_t *p;
|
||||||
|
if (n_cigar == 0) return;
|
||||||
|
mm_enlarge_cigar(r, n_cigar);
|
||||||
p = r->p;
|
p = r->p;
|
||||||
if (p->n_cigar > 0 && (p->cigar[p->n_cigar-1]&0xf) == (cigar[0]&0xf)) { // same CIGAR op at the boundary
|
if (p->n_cigar > 0 && (p->cigar[p->n_cigar-1]&0xf) == (cigar[0]&0xf)) { // same CIGAR op at the boundary
|
||||||
p->cigar[p->n_cigar-1] += cigar[0]>>4<<4;
|
p->cigar[p->n_cigar-1] += cigar[0]>>4<<4;
|
||||||
@@ -313,25 +333,31 @@ static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, uint32_t *cigar) //
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc, const int8_t *mat, int w, int end_bonus, int zdrop, int flag, ksw_extz_t *ez)
|
static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc,
|
||||||
|
const int8_t *mat, int w, int end_bonus, int zdrop, int ksw_flag, ksw_extz_t *ez)
|
||||||
{
|
{
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
||||||
int i;
|
int i;
|
||||||
fprintf(stderr, "===> q=(%d,%d), e=(%d,%d), bw=%d, flag=%d, zdrop=%d <===\n", opt->q, opt->q2, opt->e, opt->e2, w, flag, opt->zdrop);
|
fprintf(stderr, "===> q=(%d,%d), e=(%d,%d), bw=%d, ksw_flag=%d, zdrop=%d, end_bonus=%d <===\n", opt->q, opt->q2, opt->e, opt->e2, w, ksw_flag, opt->zdrop, end_bonus);
|
||||||
for (i = 0; i < tlen; ++i) fputc("ACGTN"[tseq[i]], stderr);
|
for (i = 0; i < tlen; ++i) fputc("ACGTN"[tseq[i]], stderr);
|
||||||
fputc('\n', stderr);
|
fputc('\n', stderr);
|
||||||
for (i = 0; i < qlen; ++i) fputc("ACGTN"[qseq[i]], stderr);
|
for (i = 0; i < qlen; ++i) fputc("ACGTN"[qseq[i]], stderr);
|
||||||
fputc('\n', stderr);
|
fputc('\n', stderr);
|
||||||
}
|
}
|
||||||
if (opt->max_sw_mat > 0 && (int64_t)tlen * qlen > opt->max_sw_mat) {
|
if (opt->transition != 0 && opt->b != opt->transition)
|
||||||
|
ksw_flag |= KSW_EZ_GENERIC_SC;
|
||||||
|
if (opt->max_sw_mat > 0 && (int64_t)tlen * qlen > opt->max_sw_mat) { // too much memory; skip alignment
|
||||||
ksw_reset_extz(ez);
|
ksw_reset_extz(ez);
|
||||||
ez->zdropped = 1;
|
ez->zdropped = 1;
|
||||||
} else if (opt->flag & MM_F_SPLICE)
|
} else if (opt->flag & MM_F_SPLICE) { // spliced alignment
|
||||||
ksw_exts2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->noncan, zdrop, opt->junc_bonus, flag, junc, ez);
|
assert((ksw_flag & KSW_EZ_SPLICE_FOR) == 0 || (ksw_flag & KSW_EZ_SPLICE_REV) == 0);
|
||||||
else if (opt->q == opt->q2 && opt->e == opt->e2)
|
if (!(opt->flag & MM_F_SPLICE_OLD)) ksw_flag |= KSW_EZ_SPLICE_CMPLX;
|
||||||
ksw_extz2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, w, zdrop, end_bonus, flag, ez);
|
ksw_exts2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->noncan, zdrop, end_bonus, opt->junc_bonus, opt->junc_pen, ksw_flag, junc, ez);
|
||||||
else
|
} else if (opt->q == opt->q2 && opt->e == opt->e2) { // affine gap
|
||||||
ksw_extd2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, flag, ez);
|
ksw_extz2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, w, zdrop, end_bonus, ksw_flag, ez);
|
||||||
|
} else { // dual affine gap
|
||||||
|
ksw_extd2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, ksw_flag, ez);
|
||||||
|
}
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
||||||
int i;
|
int i;
|
||||||
fprintf(stderr, "score=%d, cigar=", ez->score);
|
fprintf(stderr, "score=%d, cigar=", ez->score);
|
||||||
@@ -341,6 +367,45 @@ static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static int mm_align_sr_rna(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc, uint8_t *tseq2, uint8_t *junc2,
|
||||||
|
const int8_t *mat, int w, int end_bonus, int zdrop, int ksw_flag, ksw_extz_t *ez)
|
||||||
|
{
|
||||||
|
int32_t ilen = opt->q2 * 2, tlen2 = qlen * 2 + ilen;
|
||||||
|
int32_t i, ll = 0, lr = 0, nn = 0, n_ins = 0;
|
||||||
|
if (!(opt->flag & MM_F_SPLICE)) return 0; // only for spliced alignment
|
||||||
|
if (qlen > MM_MAX_QLEN_FLANK || qlen * 2 + ilen > tlen) return 0; // the query sequence can't be too long and the target sequence must be long enough
|
||||||
|
for (i = 0; i < qlen; ++i) // exact match length from the left
|
||||||
|
if (qseq[i] == tseq[i] && qseq[i] < 4)
|
||||||
|
++ll;
|
||||||
|
for (i = 0; i < qlen; ++i) // exact match length from the right
|
||||||
|
if (qseq[qlen - 1 - i] == tseq[tlen - 1 - i] && qseq[qlen - 1 - i] < 4)
|
||||||
|
++lr;
|
||||||
|
if (qlen - (ll + lr) > 9) return 0; // qlen may be smaller than ll+lr
|
||||||
|
memcpy(tseq2, tseq, qlen);
|
||||||
|
memset(&tseq2[qlen], 4, ilen);
|
||||||
|
memcpy(&tseq2[qlen + ilen], &tseq[tlen - qlen], qlen);
|
||||||
|
if (junc) {
|
||||||
|
memcpy(junc2, junc, qlen);
|
||||||
|
memset(&junc2[qlen], 0, ilen);
|
||||||
|
memcpy(&junc2[qlen + ilen], &junc[tlen - qlen], qlen);
|
||||||
|
}
|
||||||
|
if (!(opt->flag & MM_F_SPLICE_OLD)) ksw_flag |= KSW_EZ_SPLICE_CMPLX;
|
||||||
|
ksw_exts2_sse(km, qlen, qseq, tlen2, tseq2, 5, mat, opt->q, opt->e, opt->q2, opt->noncan, zdrop, end_bonus, opt->junc_bonus, opt->junc_pen, ksw_flag, junc2, ez);
|
||||||
|
if (ez->zdropped) return 0;
|
||||||
|
if ((ez->cigar[0]&0xf) != KSW_CIGAR_MATCH || (ez->cigar[ez->n_cigar-1]&0xf) != KSW_CIGAR_MATCH) return 0;
|
||||||
|
for (i = 0; i < ez->n_cigar; ++i) { // count the number of introns in the alignment
|
||||||
|
if ((ez->cigar[i]&0xf) == KSW_CIGAR_N_SKIP)
|
||||||
|
++nn;
|
||||||
|
else if ((ez->cigar[i]&0xf) == KSW_CIGAR_INS)
|
||||||
|
++n_ins;
|
||||||
|
}
|
||||||
|
if (nn != 1 || n_ins > 0) return 0; // the heuristic only works when there is exactly one intron
|
||||||
|
for (i = 0; i < ez->n_cigar; ++i)
|
||||||
|
if ((ez->cigar[i]&0xf) == KSW_CIGAR_N_SKIP)
|
||||||
|
ez->cigar[i] += (tlen - tlen2) << 4;
|
||||||
|
return 1;
|
||||||
|
}
|
||||||
|
|
||||||
static inline int mm_get_hplen_back(const mm_idx_t *mi, uint32_t rid, uint32_t x)
|
static inline int mm_get_hplen_back(const mm_idx_t *mi, uint32_t rid, uint32_t x)
|
||||||
{
|
{
|
||||||
int64_t i, off0 = mi->seq[rid].offset, off = off0 + x;
|
int64_t i, off0 = mi->seq[rid].offset, off = off0 + x;
|
||||||
@@ -570,12 +635,19 @@ static void mm_fix_bad_ends_splice(void *km, const mm_mapopt_t *opt, const mm_id
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static inline void mm_get_junc(const mm_idx_t *mi, int32_t ctg, int32_t st, int32_t en, int32_t rev, uint8_t *junc)
|
||||||
|
{
|
||||||
|
if (mi->spsc) mm_idx_spsc_get(mi, ctg, st, en, rev, junc);
|
||||||
|
else if (mi->I) mm_idx_bed_junc(mi, ctg, st, en, junc);
|
||||||
|
else memset(junc, 0, en - st);
|
||||||
|
}
|
||||||
|
|
||||||
static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, uint8_t *qseq0[2], mm_reg1_t *r, mm_reg1_t *r2, int n_a, mm128_t *a, ksw_extz_t *ez, int splice_flag)
|
static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, uint8_t *qseq0[2], mm_reg1_t *r, mm_reg1_t *r2, int n_a, mm128_t *a, ksw_extz_t *ez, int splice_flag)
|
||||||
{
|
{
|
||||||
int is_sr = !!(opt->flag & MM_F_SR), is_splice = !!(opt->flag & MM_F_SPLICE);
|
int is_sr = !!(opt->flag & MM_F_SR), is_splice = !!(opt->flag & MM_F_SPLICE), is_sr_rna = (!!(opt->flag & MM_F_SR_RNA) && is_splice);
|
||||||
int32_t rid = a[r->as].x<<1>>33, rev = a[r->as].x>>63, as1, cnt1;
|
int32_t rid = a[r->as].x<<1>>33, rev = a[r->as].x>>63, as1, cnt1;
|
||||||
uint8_t *tseq, *qseq, *junc;
|
uint8_t *tseq, *qseq, *junc, *tseq2 = 0, *junc2 = 0;
|
||||||
int32_t i, l, bw, bw_long, dropped = 0, extra_flag = 0, rs0, re0, qs0, qe0;
|
int32_t i, l, bw, bw_long, dropped = 0, ksw_flag = 0, rs0, re0, qs0, qe0;
|
||||||
int32_t rs, re, qs, qe;
|
int32_t rs, re, qs, qe;
|
||||||
int32_t rs1, qs1, re1, qe1;
|
int32_t rs1, qs1, re1, qe1;
|
||||||
int8_t mat[25];
|
int8_t mat[25];
|
||||||
@@ -584,7 +656,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
|
|
||||||
r2->cnt = 0;
|
r2->cnt = 0;
|
||||||
if (r->cnt == 0) return;
|
if (r->cnt == 0) return;
|
||||||
ksw_gen_simple_mat(5, mat, opt->a, opt->b, opt->sc_ambi);
|
ksw_gen_ts_mat(5, mat, opt->a, opt->b, opt->transition, opt->sc_ambi);
|
||||||
bw = (int)(opt->bw * 1.5 + 1.);
|
bw = (int)(opt->bw * 1.5 + 1.);
|
||||||
bw_long = (int)(opt->bw_long * 1.5 + 1.);
|
bw_long = (int)(opt->bw_long * 1.5 + 1.);
|
||||||
if (bw_long < bw) bw_long = bw;
|
if (bw_long < bw) bw_long = bw;
|
||||||
@@ -610,9 +682,10 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
assert(cnt1 > 0);
|
assert(cnt1 > 0);
|
||||||
|
|
||||||
if (is_splice) {
|
if (is_splice) {
|
||||||
if (splice_flag & MM_F_SPLICE_FOR) extra_flag |= rev? KSW_EZ_SPLICE_REV : KSW_EZ_SPLICE_FOR;
|
if (splice_flag & MM_F_SPLICE_FOR) ksw_flag |= rev? KSW_EZ_SPLICE_REV : KSW_EZ_SPLICE_FOR;
|
||||||
if (splice_flag & MM_F_SPLICE_REV) extra_flag |= rev? KSW_EZ_SPLICE_FOR : KSW_EZ_SPLICE_REV;
|
if (splice_flag & MM_F_SPLICE_REV) ksw_flag |= rev? KSW_EZ_SPLICE_FOR : KSW_EZ_SPLICE_REV;
|
||||||
if (opt->flag & MM_F_SPLICE_FLANK) extra_flag |= KSW_EZ_SPLICE_FLANK;
|
if (opt->flag & MM_F_SPLICE_FLANK) ksw_flag |= KSW_EZ_SPLICE_FLANK;
|
||||||
|
if (mi->spsc) ksw_flag |= KSW_EZ_SPLICE_SCORE;
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Look for the start and end of regions to perform DP. This sounds easy
|
/* Look for the start and end of regions to perform DP. This sounds easy
|
||||||
@@ -697,6 +770,12 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
tseq = (uint8_t*)kmalloc(km, re0 - rs0);
|
tseq = (uint8_t*)kmalloc(km, re0 - rs0);
|
||||||
junc = (uint8_t*)kmalloc(km, re0 - rs0);
|
junc = (uint8_t*)kmalloc(km, re0 - rs0);
|
||||||
|
|
||||||
|
if (is_sr_rna) {
|
||||||
|
int32_t max_tlen2 = MM_MAX_QLEN_FLANK * 2 + opt->q2 * 2;
|
||||||
|
tseq2 = Kmalloc(km, uint8_t, max_tlen2 * 2);
|
||||||
|
junc2 = tseq2 + max_tlen2;
|
||||||
|
}
|
||||||
|
|
||||||
if (qs > 0 && rs > 0) { // left extension; probably the condition can be changed to "qs > qs0 && rs > rs0"
|
if (qs > 0 && rs > 0) { // left extension; probably the condition can be changed to "qs > qs0 && rs > rs0"
|
||||||
if (opt->flag & MM_F_QSTRAND) {
|
if (opt->flag & MM_F_QSTRAND) {
|
||||||
qseq = &qseq0[0][qs0];
|
qseq = &qseq0[0][qs0];
|
||||||
@@ -705,11 +784,11 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
qseq = &qseq0[rev][qs0];
|
qseq = &qseq0[rev][qs0];
|
||||||
mm_idx_getseq(mi, rid, rs0, rs, tseq);
|
mm_idx_getseq(mi, rid, rs0, rs, tseq);
|
||||||
}
|
}
|
||||||
mm_idx_bed_junc(mi, rid, rs0, rs, junc);
|
mm_get_junc(mi, rid, rs0, rs, !!(ksw_flag&KSW_EZ_SPLICE_REV), junc);
|
||||||
mm_seq_rev(qs - qs0, qseq);
|
mm_seq_rev(qs - qs0, qseq);
|
||||||
mm_seq_rev(rs - rs0, tseq);
|
mm_seq_rev(rs - rs0, tseq);
|
||||||
mm_seq_rev(rs - rs0, junc);
|
mm_seq_rev(rs - rs0, junc);
|
||||||
mm_align_pair(km, opt, qs - qs0, qseq, rs - rs0, tseq, junc, mat, bw, opt->end_bonus, r->split_inv? opt->zdrop_inv : opt->zdrop, extra_flag|KSW_EZ_EXTZ_ONLY|KSW_EZ_RIGHT|KSW_EZ_REV_CIGAR, ez);
|
mm_align_pair(km, opt, qs - qs0, qseq, rs - rs0, tseq, junc, mat, bw, opt->end_bonus, r->split_inv? opt->zdrop_inv : opt->zdrop, ksw_flag|KSW_EZ_EXTZ_ONLY|KSW_EZ_RIGHT|KSW_EZ_REV_CIGAR, ez);
|
||||||
if (ez->n_cigar > 0) {
|
if (ez->n_cigar > 0) {
|
||||||
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
||||||
r->p->dp_score += ez->max;
|
r->p->dp_score += ez->max;
|
||||||
@@ -721,14 +800,14 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
re1 = rs, qe1 = qs;
|
re1 = rs, qe1 = qs;
|
||||||
assert(qs1 >= 0 && rs1 >= 0);
|
assert(qs1 >= 0 && rs1 >= 0);
|
||||||
|
|
||||||
for (i = is_sr? cnt1 - 1 : 1; i < cnt1; ++i) { // gap filling
|
for (i = is_sr? cnt1 - 1 : 1; i < cnt1; ++i) { // gap filling; for short genomic reads, fill from the first seed to the last
|
||||||
if ((a[as1+i].y & (MM_SEED_IGNORE|MM_SEED_TANDEM)) && i != cnt1 - 1) continue;
|
if ((a[as1+i].y & (MM_SEED_IGNORE|MM_SEED_TANDEM)) && i != cnt1 - 1) continue;
|
||||||
if (is_sr && !(mi->flag & MM_I_HPC)) {
|
if (is_sr && !(mi->flag & MM_I_HPC)) {
|
||||||
re = (int32_t)a[as1 + i].x + 1;
|
re = (int32_t)a[as1 + i].x + 1;
|
||||||
qe = (int32_t)a[as1 + i].y + 1;
|
qe = (int32_t)a[as1 + i].y + 1;
|
||||||
} else mm_adjust_minier(mi, qseq0, &a[as1 + i], &re, &qe);
|
} else mm_adjust_minier(mi, qseq0, &a[as1 + i], &re, &qe);
|
||||||
re1 = re, qe1 = qe;
|
re1 = re, qe1 = qe;
|
||||||
if (i == cnt1 - 1 || (a[as1+i].y&MM_SEED_LONG_JOIN) || (qe - qs >= opt->min_ksw_len && re - rs >= opt->min_ksw_len)) {
|
if (i == cnt1 - 1 || (a[as1+i].y&MM_SEED_LONG_JOIN) || (qe - qs >= opt->min_ksw_len && re - rs >= opt->min_ksw_len)) { // gap filling
|
||||||
int j, bw1 = bw_long, zdrop_code;
|
int j, bw1 = bw_long, zdrop_code;
|
||||||
if (a[as1+i].y & MM_SEED_LONG_JOIN)
|
if (a[as1+i].y & MM_SEED_LONG_JOIN)
|
||||||
bw1 = qe - qs > re - rs? qe - qs : re - rs;
|
bw1 = qe - qs > re - rs? qe - qs : re - rs;
|
||||||
@@ -740,21 +819,29 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
qseq = &qseq0[rev][qs];
|
qseq = &qseq0[rev][qs];
|
||||||
mm_idx_getseq(mi, rid, rs, re, tseq);
|
mm_idx_getseq(mi, rid, rs, re, tseq);
|
||||||
}
|
}
|
||||||
mm_idx_bed_junc(mi, rid, rs, re, junc);
|
mm_get_junc(mi, rid, rs, re, !!(ksw_flag&KSW_EZ_SPLICE_REV), junc);
|
||||||
if (is_sr) { // perform ungapped alignment
|
if (is_sr || (is_sr_rna && qe - qs == re - rs)) { // perform ungapped alignment
|
||||||
|
int32_t max_gapped_score = (qe - qs - 2) * opt->a - 2 * (opt->q + opt->e);
|
||||||
assert(qe - qs == re - rs);
|
assert(qe - qs == re - rs);
|
||||||
ksw_reset_extz(ez);
|
ksw_reset_extz(ez);
|
||||||
for (j = 0, ez->score = 0; j < qe - qs; ++j) {
|
for (j = 0, ez->score = 0; j < qe - qs; ++j) {
|
||||||
if (qseq[j] >= 4 || tseq[j] >= 4) ez->score += opt->e2;
|
if (qseq[j] >= 4 || tseq[j] >= 4) ez->score += opt->sc_ambi > 0? -opt->sc_ambi : opt->sc_ambi;
|
||||||
else ez->score += qseq[j] == tseq[j]? opt->a : -opt->b;
|
else ez->score += qseq[j] == tseq[j]? opt->a : -opt->b;
|
||||||
}
|
}
|
||||||
ez->cigar = ksw_push_cigar(km, &ez->n_cigar, &ez->m_cigar, ez->cigar, MM_CIGAR_MATCH, qe - qs);
|
if (ez->score > max_gapped_score)
|
||||||
|
ez->cigar = ksw_push_cigar(km, &ez->n_cigar, &ez->m_cigar, ez->cigar, MM_CIGAR_MATCH, qe - qs);
|
||||||
|
else
|
||||||
|
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, opt->zdrop, ksw_flag|KSW_EZ_APPROX_MAX, ez);
|
||||||
} else { // perform normal gapped alignment
|
} else { // perform normal gapped alignment
|
||||||
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, opt->zdrop, extra_flag|KSW_EZ_APPROX_MAX, ez); // first pass: with approximate Z-drop
|
int32_t skip_full = 0;
|
||||||
|
if (is_sr_rna)
|
||||||
|
skip_full = mm_align_sr_rna(km, opt, qe - qs, qseq, re - rs, tseq, junc, tseq2, junc2, mat, bw1, -1, opt->zdrop, ksw_flag|KSW_EZ_APPROX_MAX, ez);
|
||||||
|
if (!skip_full)
|
||||||
|
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, opt->zdrop, ksw_flag|KSW_EZ_APPROX_MAX, ez); // first pass: with approximate Z-drop
|
||||||
}
|
}
|
||||||
// test Z-drop and inversion Z-drop
|
// test Z-drop and inversion Z-drop
|
||||||
if ((zdrop_code = mm_test_zdrop(km, opt, qseq, tseq, ez->n_cigar, ez->cigar, mat)) != 0)
|
if ((zdrop_code = mm_test_zdrop(km, opt, qseq, tseq, ez->n_cigar, ez->cigar, mat)) != 0)
|
||||||
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, zdrop_code == 2? opt->zdrop_inv : opt->zdrop, extra_flag, ez); // second pass: lift approximate
|
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, zdrop_code == 2? opt->zdrop_inv : opt->zdrop, ksw_flag, ez); // second pass: lift approximate
|
||||||
// update CIGAR
|
// update CIGAR
|
||||||
if (ez->n_cigar > 0)
|
if (ez->n_cigar > 0)
|
||||||
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
||||||
@@ -792,8 +879,8 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
qseq = &qseq0[rev][qe];
|
qseq = &qseq0[rev][qe];
|
||||||
mm_idx_getseq(mi, rid, re, re0, tseq);
|
mm_idx_getseq(mi, rid, re, re0, tseq);
|
||||||
}
|
}
|
||||||
mm_idx_bed_junc(mi, rid, re, re0, junc);
|
mm_get_junc(mi, rid, re, re0, !!(ksw_flag&KSW_EZ_SPLICE_REV), junc);
|
||||||
mm_align_pair(km, opt, qe0 - qe, qseq, re0 - re, tseq, junc, mat, bw, opt->end_bonus, opt->zdrop, extra_flag|KSW_EZ_EXTZ_ONLY, ez);
|
mm_align_pair(km, opt, qe0 - qe, qseq, re0 - re, tseq, junc, mat, bw, opt->end_bonus, opt->zdrop, ksw_flag|KSW_EZ_EXTZ_ONLY, ez);
|
||||||
if (ez->n_cigar > 0) {
|
if (ez->n_cigar > 0) {
|
||||||
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
||||||
r->p->dp_score += ez->max;
|
r->p->dp_score += ez->max;
|
||||||
@@ -816,11 +903,12 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
|||||||
mm_idx_getseq(mi, rid, rs1, re1, tseq);
|
mm_idx_getseq(mi, rid, rs1, re1, tseq);
|
||||||
qseq = &qseq0[r->rev][qs1];
|
qseq = &qseq0[r->rev][qs1];
|
||||||
}
|
}
|
||||||
mm_update_extra(r, qseq, tseq, mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & MM_F_SR));
|
mm_update_extra(r, qseq, tseq, mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(is_sr || is_sr_rna));
|
||||||
if (rev && r->p->trans_strand)
|
if (rev && r->p->trans_strand)
|
||||||
r->p->trans_strand ^= 3; // flip to the read strand
|
r->p->trans_strand ^= 3; // flip to the read strand
|
||||||
}
|
}
|
||||||
|
|
||||||
|
if (tseq2) kfree(km, tseq2);
|
||||||
kfree(km, tseq);
|
kfree(km, tseq);
|
||||||
kfree(km, junc);
|
kfree(km, junc);
|
||||||
}
|
}
|
||||||
@@ -842,7 +930,7 @@ static int mm_align1_inv(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, i
|
|||||||
if (ql < opt->min_chain_score || ql > opt->max_gap) return 0;
|
if (ql < opt->min_chain_score || ql > opt->max_gap) return 0;
|
||||||
if (tl < opt->min_chain_score || tl > opt->max_gap) return 0;
|
if (tl < opt->min_chain_score || tl > opt->max_gap) return 0;
|
||||||
|
|
||||||
ksw_gen_simple_mat(5, mat, opt->a, opt->b, opt->sc_ambi);
|
ksw_gen_ts_mat(5, mat, opt->a, opt->b, opt->transition, opt->sc_ambi);
|
||||||
tseq = (uint8_t*)kmalloc(km, tl);
|
tseq = (uint8_t*)kmalloc(km, tl);
|
||||||
mm_idx_getseq(mi, r1->rid, r1->re, r2->rs, tseq);
|
mm_idx_getseq(mi, r1->rid, r1->re, r2->rs, tseq);
|
||||||
qseq = r1->rev? &qseq0[0][r2->qe] : &qseq0[1][qlen - r2->qs];
|
qseq = r1->rev? &qseq0[0][r2->qe] : &qseq0[1][qlen - r2->qs];
|
||||||
@@ -875,7 +963,7 @@ static int mm_align1_inv(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, i
|
|||||||
}
|
}
|
||||||
r_inv->rs = r1->re + t_off;
|
r_inv->rs = r1->re + t_off;
|
||||||
r_inv->re = r_inv->rs + ez->max_t + 1;
|
r_inv->re = r_inv->rs + ez->max_t + 1;
|
||||||
mm_update_extra(r_inv, &qseq[q_off], &tseq[t_off], mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & MM_F_SR));
|
mm_update_extra(r_inv, &qseq[q_off], &tseq[t_off], mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & (MM_F_SR|MM_F_SR_RNA)));
|
||||||
ret = 1;
|
ret = 1;
|
||||||
end_align1_inv:
|
end_align1_inv:
|
||||||
kfree(km, tseq);
|
kfree(km, tseq);
|
||||||
@@ -917,14 +1005,14 @@ double mm_event_identity(const mm_reg1_t *r)
|
|||||||
static int32_t mm_recal_max_dp(const mm_reg1_t *r, double b2, int32_t match_sc)
|
static int32_t mm_recal_max_dp(const mm_reg1_t *r, double b2, int32_t match_sc)
|
||||||
{
|
{
|
||||||
uint32_t i;
|
uint32_t i;
|
||||||
int32_t n_gap = 0, n_gapo = 0, n_mis;
|
int32_t n_gap = 0, n_mis;
|
||||||
double gap_cost = 0.0;
|
double gap_cost = 0.0;
|
||||||
if (r->p == 0) return -1;
|
if (r->p == 0) return -1;
|
||||||
for (i = 0; i < r->p->n_cigar; ++i) {
|
for (i = 0; i < r->p->n_cigar; ++i) {
|
||||||
int32_t op = r->p->cigar[i] & 0xf, len = r->p->cigar[i] >> 4;
|
int32_t op = r->p->cigar[i] & 0xf, len = r->p->cigar[i] >> 4;
|
||||||
if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL) {
|
if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL) {
|
||||||
gap_cost += b2 + (double)mg_log2(1.0 + len);
|
gap_cost += b2 + (double)mg_log2(1.0 + len);
|
||||||
++n_gapo, n_gap += len;
|
n_gap += len;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
n_mis = r->blen + r->p->n_ambi - r->mlen - n_gap;
|
n_mis = r->blen + r->p->n_ambi - r->mlen - n_gap;
|
||||||
@@ -976,24 +1064,36 @@ mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
|||||||
n_a = mm_squeeze_a(km, n_regs, regs, a);
|
n_a = mm_squeeze_a(km, n_regs, regs, a);
|
||||||
memset(&ez, 0, sizeof(ksw_extz_t));
|
memset(&ez, 0, sizeof(ksw_extz_t));
|
||||||
for (i = 0; i < n_regs; ++i) {
|
for (i = 0; i < n_regs; ++i) {
|
||||||
mm_reg1_t r2;
|
mm_reg1_t r2; // only used for inversion
|
||||||
if ((opt->flag&MM_F_SPLICE) && (opt->flag&MM_F_SPLICE_FOR) && (opt->flag&MM_F_SPLICE_REV)) { // then do two rounds of alignments for both strands
|
if ((opt->flag&MM_F_SPLICE) && (opt->flag&MM_F_SPLICE_FOR) && (opt->flag&MM_F_SPLICE_REV)) { // then do two rounds of alignments for both strands
|
||||||
mm_reg1_t s[2], s2[2];
|
mm_reg1_t s[2], s2[2], *r;
|
||||||
int which, trans_strand;
|
|
||||||
s[0] = s[1] = regs[i];
|
s[0] = s[1] = regs[i];
|
||||||
mm_align1(km, opt, mi, qlen, qseq0, &s[0], &s2[0], n_a, a, &ez, MM_F_SPLICE_FOR);
|
mm_align1(km, opt, mi, qlen, qseq0, &s[0], &s2[0], n_a, a, &ez, MM_F_SPLICE_FOR); // assume the transcript is on the + strand of the genome
|
||||||
mm_align1(km, opt, mi, qlen, qseq0, &s[1], &s2[1], n_a, a, &ez, MM_F_SPLICE_REV);
|
if ((opt->flag&MM_F_SR_RNA) && regs[i].qe - regs[i].qs == regs[i].re - regs[i].rs && s[0].qe - s[0].qs == s[0].re - s[0].rs && s[0].qs == 0 && s[0].qe == qlen) {
|
||||||
if (s[0].p->dp_score > s[1].p->dp_score) which = 0, trans_strand = 1;
|
|
||||||
else if (s[0].p->dp_score < s[1].p->dp_score) which = 1, trans_strand = 2;
|
|
||||||
else trans_strand = 3, which = (qlen + s[0].p->dp_score) & 1; // randomly choose a strand, effectively
|
|
||||||
if (which == 0) {
|
|
||||||
regs[i] = s[0], r2 = s2[0];
|
regs[i] = s[0], r2 = s2[0];
|
||||||
free(s[1].p);
|
regs[i].p->trans_strand = 0;
|
||||||
} else {
|
} else {
|
||||||
regs[i] = s[1], r2 = s2[1];
|
int which, trans_strand;
|
||||||
free(s[0].p);
|
mm_align1(km, opt, mi, qlen, qseq0, &s[1], &s2[1], n_a, a, &ez, MM_F_SPLICE_REV); // assume the transcript on the - strand
|
||||||
|
if (s[0].p->dp_score > s[1].p->dp_score) which = 0, trans_strand = 1;
|
||||||
|
else if (s[0].p->dp_score < s[1].p->dp_score) which = 1, trans_strand = 2;
|
||||||
|
else trans_strand = 3, which = (qlen + s[0].p->dp_score) & 1; // randomly choose a strand, effectively
|
||||||
|
if (which == 0) {
|
||||||
|
regs[i] = s[0], r2 = s2[0];
|
||||||
|
free(s[1].p);
|
||||||
|
} else {
|
||||||
|
regs[i] = s[1], r2 = s2[1];
|
||||||
|
free(s[0].p);
|
||||||
|
}
|
||||||
|
r = ®s[i];
|
||||||
|
r->p->trans_strand = trans_strand;
|
||||||
|
if (r->is_spliced) {
|
||||||
|
if (trans_strand == 1 || trans_strand == 2) // this is an *approximate* way to tell if there are splice signals.
|
||||||
|
r->p->dp_max += (opt->a + opt->b) + ((opt->a + opt->b) >> 1);
|
||||||
|
else if (trans_strand == 3)
|
||||||
|
r->p->dp_max -= opt->a + opt->b;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
regs[i].p->trans_strand = trans_strand;
|
|
||||||
} else { // one round of alignment
|
} else { // one round of alignment
|
||||||
mm_align1(km, opt, mi, qlen, qseq0, ®s[i], &r2, n_a, a, &ez, opt->flag);
|
mm_align1(km, opt, mi, qlen, qseq0, ®s[i], &r2, n_a, a, &ez, opt->flag);
|
||||||
if (opt->flag&MM_F_SPLICE)
|
if (opt->flag&MM_F_SPLICE)
|
||||||
@@ -1011,7 +1111,7 @@ mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
|||||||
kfree(km, qseq0[0]);
|
kfree(km, qseq0[0]);
|
||||||
kfree(km, ez.cigar);
|
kfree(km, ez.cigar);
|
||||||
mm_filter_regs(opt, qlen, n_regs_, regs);
|
mm_filter_regs(opt, qlen, n_regs_, regs);
|
||||||
if (!(opt->flag&MM_F_SR) && !opt->split_prefix && qlen >= opt->rank_min_len) {
|
if (!(opt->flag&(MM_F_SR|MM_F_SR_RNA|MM_F_ALL_CHAINS)) && !opt->split_prefix && qlen >= opt->rank_min_len) {
|
||||||
mm_update_dp_max(qlen, *n_regs_, regs, opt->rank_frac, opt->a, opt->b);
|
mm_update_dp_max(qlen, *n_regs_, regs, opt->rank_frac, opt->a, opt->b);
|
||||||
mm_filter_regs(opt, qlen, n_regs_, regs);
|
mm_filter_regs(opt, qlen, n_regs_, regs);
|
||||||
}
|
}
|
||||||
|
|||||||
+2
-2
@@ -31,8 +31,8 @@ To acquire the data used in this cookbook and to install minimap2 and paftools,
|
|||||||
please follow the command lines below:
|
please follow the command lines below:
|
||||||
```sh
|
```sh
|
||||||
# install minimap2 executables
|
# install minimap2 executables
|
||||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.22/minimap2-2.22_x64-linux.tar.bz2 | tar jxf -
|
curl -L https://github.com/lh3/minimap2/releases/download/v2.31/minimap2-2.31_x64-linux.tar.bz2 | tar jxf -
|
||||||
cp minimap2-2.22_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
cp minimap2-2.31_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
||||||
export PATH="$PATH:"`pwd` # put the current directory on PATH
|
export PATH="$PATH:"`pwd` # put the current directory on PATH
|
||||||
# download example datasets
|
# download example datasets
|
||||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.10/cookbook-data.tgz | tar zxf -
|
curl -L https://github.com/lh3/minimap2/releases/download/v2.10/cookbook-data.tgz | tar zxf -
|
||||||
|
|||||||
@@ -119,6 +119,7 @@ int mm_write_sam_hdr(const mm_idx_t *idx, const char *rg, const char *ver, int a
|
|||||||
{
|
{
|
||||||
kstring_t str = {0,0,0};
|
kstring_t str = {0,0,0};
|
||||||
int ret = 0;
|
int ret = 0;
|
||||||
|
mm_sprintf_lite(&str, "@HD\tVN:1.6\tSO:unsorted\tGO:query\n");
|
||||||
if (idx) {
|
if (idx) {
|
||||||
uint32_t i;
|
uint32_t i;
|
||||||
for (i = 0; i < idx->n_seq; ++i)
|
for (i = 0; i < idx->n_seq; ++i)
|
||||||
@@ -138,10 +139,48 @@ int mm_write_sam_hdr(const mm_idx_t *idx, const char *rg, const char *ver, int a
|
|||||||
return ret;
|
return ret;
|
||||||
}
|
}
|
||||||
|
|
||||||
static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq, const mm_reg1_t *r, char *tmp, int no_iden, int write_tag)
|
static void write_indel_ds(kstring_t *str, int64_t len, const uint8_t *seq, int64_t ll, int64_t lr) // write an indel to ds; adapted from minigraph
|
||||||
{
|
{
|
||||||
int i, q_off, t_off;
|
int64_t i;
|
||||||
if (write_tag) mm_sprintf_lite(s, "\tcs:Z:");
|
if (ll + lr >= len) {
|
||||||
|
mm_sprintf_lite(str, "[");
|
||||||
|
for (i = 0; i < len; ++i)
|
||||||
|
mm_sprintf_lite(str, "%c", "acgtn"[seq[i]]);
|
||||||
|
mm_sprintf_lite(str, "]");
|
||||||
|
} else {
|
||||||
|
int64_t k = 0;
|
||||||
|
if (ll > 0) {
|
||||||
|
mm_sprintf_lite(str, "[");
|
||||||
|
for (i = 0; i < ll; ++i)
|
||||||
|
mm_sprintf_lite(str, "%c", "acgtn"[seq[k+i]]);
|
||||||
|
mm_sprintf_lite(str, "]");
|
||||||
|
k += ll;
|
||||||
|
}
|
||||||
|
for (i = 0; i < len - lr - ll; ++i)
|
||||||
|
mm_sprintf_lite(str, "%c", "acgtn"[seq[k+i]]);
|
||||||
|
k += len - lr - ll;
|
||||||
|
if (lr > 0) {
|
||||||
|
mm_sprintf_lite(str, "[");
|
||||||
|
for (i = 0; i < lr; ++i)
|
||||||
|
mm_sprintf_lite(str, "%c", "acgtn"[seq[k+i]]);
|
||||||
|
mm_sprintf_lite(str, "]");
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static void write_cs_ds_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq, const mm_reg1_t *r, char *tmp, int no_iden, int is_ds, int write_tag)
|
||||||
|
{
|
||||||
|
int i, q_off, t_off, q_len = 0, t_len = 0;
|
||||||
|
if (write_tag) mm_sprintf_lite(s, "\t%cs:Z:", is_ds? 'd' : 'c');
|
||||||
|
for (i = 0; i < (int)r->p->n_cigar; ++i) {
|
||||||
|
int op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||||
|
if (op == MM_CIGAR_MATCH || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH)
|
||||||
|
q_len += len, t_len += len;
|
||||||
|
else if (op == MM_CIGAR_INS)
|
||||||
|
q_len += len;
|
||||||
|
else if (op == MM_CIGAR_DEL || op == MM_CIGAR_N_SKIP)
|
||||||
|
t_len += len;
|
||||||
|
}
|
||||||
for (i = q_off = t_off = 0; i < (int)r->p->n_cigar; ++i) {
|
for (i = q_off = t_off = 0; i < (int)r->p->n_cigar; ++i) {
|
||||||
int j, op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
int j, op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||||
assert((op >= MM_CIGAR_MATCH && op <= MM_CIGAR_N_SKIP) || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH);
|
assert((op >= MM_CIGAR_MATCH && op <= MM_CIGAR_N_SKIP) || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH);
|
||||||
@@ -167,14 +206,42 @@ static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
|||||||
}
|
}
|
||||||
q_off += len, t_off += len;
|
q_off += len, t_off += len;
|
||||||
} else if (op == MM_CIGAR_INS) {
|
} else if (op == MM_CIGAR_INS) {
|
||||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
if (is_ds) {
|
||||||
tmp[j] = "acgtn"[qseq[q_off + j]];
|
int z, ll, lr, y = q_off;
|
||||||
mm_sprintf_lite(s, "+%s", tmp);
|
for (z = 1; z <= len; ++z)
|
||||||
|
if (y - z < 0 || qseq[y + len - z] != qseq[y - z])
|
||||||
|
break;
|
||||||
|
lr = z - 1;
|
||||||
|
for (z = 0; z < len; ++z)
|
||||||
|
if (y + len + z >= q_len || qseq[y + len + z] != qseq[y + z])
|
||||||
|
break;
|
||||||
|
ll = z;
|
||||||
|
mm_sprintf_lite(s, "+");
|
||||||
|
write_indel_ds(s, len, &qseq[y], ll, lr);
|
||||||
|
} else {
|
||||||
|
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||||
|
tmp[j] = "acgtn"[qseq[q_off + j]];
|
||||||
|
mm_sprintf_lite(s, "+%s", tmp);
|
||||||
|
}
|
||||||
q_off += len;
|
q_off += len;
|
||||||
} else if (op == MM_CIGAR_DEL) {
|
} else if (op == MM_CIGAR_DEL) {
|
||||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
if (is_ds) {
|
||||||
tmp[j] = "acgtn"[tseq[t_off + j]];
|
int z, ll, lr, x = t_off;
|
||||||
mm_sprintf_lite(s, "-%s", tmp);
|
for (z = 1; z <= len; ++z)
|
||||||
|
if (x - z < 0 || tseq[x + len - z] != tseq[x - z])
|
||||||
|
break;
|
||||||
|
lr = z - 1;
|
||||||
|
for (z = 0; z < len; ++z)
|
||||||
|
if (x + len + z >= t_len || tseq[x + z] != tseq[x + len + z])
|
||||||
|
break;
|
||||||
|
ll = z;
|
||||||
|
mm_sprintf_lite(s, "-");
|
||||||
|
write_indel_ds(s, len, &tseq[x], ll, lr);
|
||||||
|
} else {
|
||||||
|
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||||
|
tmp[j] = "acgtn"[tseq[t_off + j]];
|
||||||
|
mm_sprintf_lite(s, "-%s", tmp);
|
||||||
|
}
|
||||||
t_off += len;
|
t_off += len;
|
||||||
} else { // intron
|
} else { // intron
|
||||||
assert(len >= 2);
|
assert(len >= 2);
|
||||||
@@ -186,6 +253,52 @@ static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
|||||||
assert(t_off == r->re - r->rs && q_off == r->qe - r->qs);
|
assert(t_off == r->re - r->rs && q_off == r->qe - r->qs);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static inline void revcomp_splice(uint8_t s[2])
|
||||||
|
{
|
||||||
|
uint8_t c = s[1] < 4? 3 - s[1] : 4;
|
||||||
|
s[1] = s[0] < 4? 3 - s[0] : 4;
|
||||||
|
s[0] = c;
|
||||||
|
}
|
||||||
|
|
||||||
|
void mm_write_junc(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r)
|
||||||
|
{
|
||||||
|
int32_t i, t_off, swritten = 0;
|
||||||
|
s->l = 0;
|
||||||
|
if (!r->is_spliced || r->p == 0) return; // no junctions
|
||||||
|
if (r->p->trans_strand != 1 && r->p->trans_strand != 2) return; // no preferred strand
|
||||||
|
for (i = 0, t_off = r->rs; i < (int)r->p->n_cigar; ++i) {
|
||||||
|
int op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||||
|
if (op == MM_CIGAR_MATCH || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH || op == MM_CIGAR_DEL) {
|
||||||
|
t_off += len;
|
||||||
|
} else if (op == MM_CIGAR_N_SKIP) { // intron
|
||||||
|
uint8_t donor[2], acceptor[2];
|
||||||
|
int32_t score1 = 0, score2 = 0, rev;
|
||||||
|
assert(len >= 2);
|
||||||
|
rev = (r->p->trans_strand == 2) ^ r->rev;
|
||||||
|
if (!rev) {
|
||||||
|
mm_idx_getseq(mi, r->rid, t_off, t_off + 2, donor);
|
||||||
|
mm_idx_getseq(mi, r->rid, t_off + len - 2, t_off + len, acceptor);
|
||||||
|
} else {
|
||||||
|
mm_idx_getseq(mi, r->rid, t_off, t_off + 2, acceptor);
|
||||||
|
mm_idx_getseq(mi, r->rid, t_off + len - 2, t_off + len, donor);
|
||||||
|
revcomp_splice(donor);
|
||||||
|
revcomp_splice(acceptor);
|
||||||
|
}
|
||||||
|
//fprintf(stderr, "%c%c-%c%c\n", "ACGTN"[donor[0]], "ACGTN"[donor[1]], "ACGTN"[acceptor[0]], "ACGTN"[acceptor[1]]);
|
||||||
|
if (donor[0] == 2 && donor[1] == 3) score1 = 3;
|
||||||
|
else if (donor[0] == 2 && donor[1] == 1) score1 = 2;
|
||||||
|
else if (donor[0] == 0 && donor[1] == 3) score1 = 1;
|
||||||
|
if (acceptor[0] == 0 && acceptor[1] == 2) score2 = 3;
|
||||||
|
else if (acceptor[0] == 0 && acceptor[1] == 1) score2 = 1;
|
||||||
|
if (swritten) mm_sprintf_lite(s, "\n");
|
||||||
|
else swritten = 1;
|
||||||
|
mm_sprintf_lite(s, "%s\t%d\t%d\t%s\t%d\t%c", mi->seq[r->rid].name, t_off, t_off + len, t->name, score1 + score2, "+-"[rev]);
|
||||||
|
t_off += len;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
assert(t_off == r->re);
|
||||||
|
}
|
||||||
|
|
||||||
static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq, const mm_reg1_t *r, char *tmp, int write_tag)
|
static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq, const mm_reg1_t *r, char *tmp, int write_tag)
|
||||||
{
|
{
|
||||||
int i, q_off, t_off, l_MD = 0;
|
int i, q_off, t_off, l_MD = 0;
|
||||||
@@ -217,7 +330,7 @@ static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
|||||||
assert(t_off == r->re - r->rs && q_off == r->qe - r->qs);
|
assert(t_off == r->re - r->rs && q_off == r->qe - r->qs);
|
||||||
}
|
}
|
||||||
|
|
||||||
static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int no_iden, int is_MD, int write_tag, int is_qstrand)
|
static void write_cs_ds_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int no_iden, int is_MD, int is_ds, int write_tag, int is_qstrand)
|
||||||
{
|
{
|
||||||
extern unsigned char seq_nt4_table[256];
|
extern unsigned char seq_nt4_table[256];
|
||||||
int i;
|
int i;
|
||||||
@@ -244,28 +357,38 @@ static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (is_MD) write_MD_core(s, tseq, qseq, r, tmp, write_tag);
|
if (is_MD) write_MD_core(s, tseq, qseq, r, tmp, write_tag);
|
||||||
else write_cs_core(s, tseq, qseq, r, tmp, no_iden, write_tag);
|
else write_cs_ds_core(s, tseq, qseq, r, tmp, no_iden, is_ds, write_tag);
|
||||||
kfree(km, qseq); kfree(km, tseq); kfree(km, tmp);
|
kfree(km, qseq); kfree(km, tseq); kfree(km, tmp);
|
||||||
}
|
}
|
||||||
|
|
||||||
int mm_gen_cs_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int is_MD, int no_iden, int is_qstrand)
|
int mm_gen_cs_ds_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int is_MD, int is_ds, int no_iden, int is_qstrand)
|
||||||
{
|
{
|
||||||
mm_bseq1_t t;
|
mm_bseq1_t t;
|
||||||
kstring_t str;
|
kstring_t str;
|
||||||
str.s = *buf, str.l = 0, str.m = *max_len;
|
str.s = *buf, str.l = 0, str.m = *max_len;
|
||||||
t.l_seq = strlen(seq);
|
t.l_seq = strlen(seq);
|
||||||
t.seq = (char*)seq;
|
t.seq = (char*)seq;
|
||||||
write_cs_or_MD(km, &str, mi, &t, r, no_iden, is_MD, 0, is_qstrand);
|
write_cs_ds_or_MD(km, &str, mi, &t, r, no_iden, is_MD, is_ds, 0, is_qstrand);
|
||||||
*max_len = str.m;
|
*max_len = str.m;
|
||||||
*buf = str.s;
|
*buf = str.s;
|
||||||
return str.l;
|
return str.l;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
int mm_gen_cs_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int is_MD, int no_iden, int is_qstrand)
|
||||||
|
{
|
||||||
|
return mm_gen_cs_ds_or_MD(km, buf, max_len, mi, r, seq, is_MD, 0, no_iden, is_qstrand);
|
||||||
|
}
|
||||||
|
|
||||||
int mm_gen_cs(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden)
|
int mm_gen_cs(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden)
|
||||||
{
|
{
|
||||||
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 0, no_iden, 0);
|
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 0, no_iden, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
int mm_gen_ds(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden)
|
||||||
|
{
|
||||||
|
return mm_gen_cs_ds_or_MD(km, buf, max_len, mi, r, seq, 0, 1, no_iden, 0);
|
||||||
|
}
|
||||||
|
|
||||||
int mm_gen_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq)
|
int mm_gen_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq)
|
||||||
{
|
{
|
||||||
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 1, 0, 0);
|
return mm_gen_cs_or_MD(km, buf, max_len, mi, r, seq, 1, 0, 0);
|
||||||
@@ -277,7 +400,7 @@ static inline void write_tags(kstring_t *s, const mm_reg1_t *r)
|
|||||||
if (r->id == r->parent) type = r->inv? 'I' : 'P';
|
if (r->id == r->parent) type = r->inv? 'I' : 'P';
|
||||||
else type = r->inv? 'i' : 'S';
|
else type = r->inv? 'i' : 'S';
|
||||||
if (r->p) {
|
if (r->p) {
|
||||||
mm_sprintf_lite(s, "\tNM:i:%d\tms:i:%d\tAS:i:%d\tnn:i:%d", r->blen - r->mlen + r->p->n_ambi, r->p->dp_max, r->p->dp_score, r->p->n_ambi);
|
mm_sprintf_lite(s, "\tNM:i:%d\tms:i:%d\tAS:i:%d\tnn:i:%d", r->blen - r->mlen + r->p->n_ambi, r->p->dp_max0, r->p->dp_score, r->p->n_ambi);
|
||||||
if (r->p->trans_strand == 1 || r->p->trans_strand == 2)
|
if (r->p->trans_strand == 1 || r->p->trans_strand == 2)
|
||||||
mm_sprintf_lite(s, "\tts:A:%c", "?+-?"[r->p->trans_strand]);
|
mm_sprintf_lite(s, "\tts:A:%c", "?+-?"[r->p->trans_strand]);
|
||||||
}
|
}
|
||||||
@@ -299,15 +422,18 @@ static inline void write_tags(kstring_t *s, const mm_reg1_t *r)
|
|||||||
if (r->split) mm_sprintf_lite(s, "\tzd:i:%d", r->split);
|
if (r->split) mm_sprintf_lite(s, "\tzd:i:%d", r->split);
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len)
|
void mm_write_paf4(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len, int n_seg, int seg_idx)
|
||||||
{
|
{
|
||||||
s->l = 0;
|
s->l = 0;
|
||||||
|
mm_sprintf_lite(s, "%s", t->name);
|
||||||
|
if ((opt_flag & MM_F_FRAG_MODE) && n_seg >= 2 && seg_idx >= 0)
|
||||||
|
mm_sprintf_lite(s, "/%d", seg_idx + 1);
|
||||||
if (r == 0) {
|
if (r == 0) {
|
||||||
mm_sprintf_lite(s, "%s\t%d\t0\t0\t*\t*\t0\t0\t0\t0\t0\t0", t->name, t->l_seq);
|
mm_sprintf_lite(s, "\t%d\t0\t0\t*\t*\t0\t0\t0\t0\t0\t0", t->l_seq);
|
||||||
if (rep_len >= 0) mm_sprintf_lite(s, "\trl:i:%d", rep_len);
|
if (rep_len >= 0) mm_sprintf_lite(s, "\trl:i:%d", rep_len);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
mm_sprintf_lite(s, "%s\t%d\t%d\t%d\t%c\t", t->name, t->l_seq, r->qs, r->qe, "+-"[r->rev]);
|
mm_sprintf_lite(s, "\t%d\t%d\t%d\t%c\t", t->l_seq, r->qs, r->qe, "+-"[r->rev]);
|
||||||
if (mi->seq[r->rid].name) mm_sprintf_lite(s, "%s", mi->seq[r->rid].name);
|
if (mi->seq[r->rid].name) mm_sprintf_lite(s, "%s", mi->seq[r->rid].name);
|
||||||
else mm_sprintf_lite(s, "%d", r->rid);
|
else mm_sprintf_lite(s, "%d", r->rid);
|
||||||
mm_sprintf_lite(s, "\t%d", mi->seq[r->rid].len);
|
mm_sprintf_lite(s, "\t%d", mi->seq[r->rid].len);
|
||||||
@@ -325,12 +451,17 @@ void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const
|
|||||||
for (k = 0; k < r->p->n_cigar; ++k)
|
for (k = 0; k < r->p->n_cigar; ++k)
|
||||||
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, MM_CIGAR_STR[r->p->cigar[k]&0xf]);
|
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, MM_CIGAR_STR[r->p->cigar[k]&0xf]);
|
||||||
}
|
}
|
||||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_DS|MM_F_OUT_MD)))
|
||||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1, !!(opt_flag&MM_F_QSTRAND));
|
write_cs_ds_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), !!(opt_flag&MM_F_OUT_MD), !!(opt_flag&MM_F_OUT_DS), 1, !!(opt_flag&MM_F_QSTRAND));
|
||||||
if ((opt_flag & MM_F_COPY_COMMENT) && t->comment)
|
if ((opt_flag & MM_F_COPY_COMMENT) && t->comment)
|
||||||
mm_sprintf_lite(s, "\t%s", t->comment);
|
mm_sprintf_lite(s, "\t%s", t->comment);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len)
|
||||||
|
{
|
||||||
|
mm_write_paf4(s, mi, t, r, km, opt_flag, rep_len, 0, 0);
|
||||||
|
}
|
||||||
|
|
||||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag)
|
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag)
|
||||||
{
|
{
|
||||||
mm_write_paf3(s, mi, t, r, km, opt_flag, -1);
|
mm_write_paf3(s, mi, t, r, km, opt_flag, -1);
|
||||||
@@ -369,14 +500,16 @@ static void write_sam_cigar(kstring_t *s, int sam_flag, int in_tag, int qlen, co
|
|||||||
clip_len[0] = r->rev? qlen - r->qe : r->qs;
|
clip_len[0] = r->rev? qlen - r->qe : r->qs;
|
||||||
clip_len[1] = r->rev? r->qs : qlen - r->qe;
|
clip_len[1] = r->rev? r->qs : qlen - r->qe;
|
||||||
if (in_tag) {
|
if (in_tag) {
|
||||||
int clip_char = (sam_flag&0x800) && !(opt_flag&MM_F_SOFTCLIP)? 5 : 4;
|
int clip_char = (((sam_flag&0x800) || ((sam_flag&0x100) && (opt_flag&MM_F_SECONDARY_SEQ))) &&
|
||||||
|
!(opt_flag&MM_F_SOFTCLIP)) ? 5 : 4;
|
||||||
mm_sprintf_lite(s, "\tCG:B:I");
|
mm_sprintf_lite(s, "\tCG:B:I");
|
||||||
if (clip_len[0]) mm_sprintf_lite(s, ",%u", clip_len[0]<<4|clip_char);
|
if (clip_len[0]) mm_sprintf_lite(s, ",%u", clip_len[0]<<4|clip_char);
|
||||||
for (k = 0; k < r->p->n_cigar; ++k)
|
for (k = 0; k < r->p->n_cigar; ++k)
|
||||||
mm_sprintf_lite(s, ",%u", r->p->cigar[k]);
|
mm_sprintf_lite(s, ",%u", r->p->cigar[k]);
|
||||||
if (clip_len[1]) mm_sprintf_lite(s, ",%u", clip_len[1]<<4|clip_char);
|
if (clip_len[1]) mm_sprintf_lite(s, ",%u", clip_len[1]<<4|clip_char);
|
||||||
} else {
|
} else {
|
||||||
int clip_char = (sam_flag&0x800) && !(opt_flag&MM_F_SOFTCLIP)? 'H' : 'S';
|
int clip_char = (((sam_flag&0x800) || ((sam_flag&0x100) && (opt_flag&MM_F_SECONDARY_SEQ))) &&
|
||||||
|
!(opt_flag&MM_F_SOFTCLIP)) ? 'H' : 'S';
|
||||||
assert(clip_len[0] < qlen && clip_len[1] < qlen);
|
assert(clip_len[0] < qlen && clip_len[1] < qlen);
|
||||||
if (clip_len[0]) mm_sprintf_lite(s, "%d%c", clip_len[0], clip_char);
|
if (clip_len[0]) mm_sprintf_lite(s, "%d%c", clip_len[0], clip_char);
|
||||||
for (k = 0; k < r->p->n_cigar; ++k)
|
for (k = 0; k < r->p->n_cigar; ++k)
|
||||||
@@ -451,7 +584,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
|||||||
if (cigar_in_tag) {
|
if (cigar_in_tag) {
|
||||||
int slen;
|
int slen;
|
||||||
if ((flag & 0x900) == 0 || (opt_flag & MM_F_SOFTCLIP)) slen = t->l_seq;
|
if ((flag & 0x900) == 0 || (opt_flag & MM_F_SOFTCLIP)) slen = t->l_seq;
|
||||||
else if (flag & 0x100) slen = 0;
|
else if ((flag & 0x100) && !(opt_flag & MM_F_SECONDARY_SEQ)) slen = 0;
|
||||||
else slen = r->qe - r->qs;
|
else slen = r->qe - r->qs;
|
||||||
mm_sprintf_lite(s, "%dS%dN", slen, r->re - r->rs);
|
mm_sprintf_lite(s, "%dS%dN", slen, r->re - r->rs);
|
||||||
} else write_sam_cigar(s, flag, 0, t->l_seq, r, opt_flag);
|
} else write_sam_cigar(s, flag, 0, t->l_seq, r, opt_flag);
|
||||||
@@ -492,7 +625,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
|||||||
mm_sprintf_lite(s, "\t");
|
mm_sprintf_lite(s, "\t");
|
||||||
if (t->qual) sam_write_sq(s, t->qual, t->l_seq, r->rev, 0);
|
if (t->qual) sam_write_sq(s, t->qual, t->l_seq, r->rev, 0);
|
||||||
else mm_sprintf_lite(s, "*");
|
else mm_sprintf_lite(s, "*");
|
||||||
} else if (flag & 0x100) {
|
} else if ((flag & 0x100) && !(opt_flag & MM_F_SECONDARY_SEQ)){
|
||||||
mm_sprintf_lite(s, "*\t*");
|
mm_sprintf_lite(s, "*\t*");
|
||||||
} else {
|
} else {
|
||||||
sam_write_sq(s, t->seq + r->qs, r->qe - r->qs, r->rev, r->rev);
|
sam_write_sq(s, t->seq + r->qs, r->qe - r->qs, r->rev, r->rev);
|
||||||
@@ -532,8 +665,8 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_DS|MM_F_OUT_MD)))
|
||||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1, 0);
|
write_cs_ds_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, !!(opt_flag&MM_F_OUT_DS), 1, 0);
|
||||||
if (cigar_in_tag)
|
if (cigar_in_tag)
|
||||||
write_sam_cigar(s, flag, 1, t->l_seq, r, opt_flag);
|
write_sam_cigar(s, flag, 1, t->l_seq, r, opt_flag);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -55,7 +55,7 @@ mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u,
|
|||||||
mm_reg1_t *r;
|
mm_reg1_t *r;
|
||||||
int i, k;
|
int i, k;
|
||||||
|
|
||||||
if (n_u == 0) return 0;
|
if (n_u <= 0) return 0;
|
||||||
|
|
||||||
// sort by score
|
// sort by score
|
||||||
z = (mm128_t*)kmalloc(km, n_u * 16);
|
z = (mm128_t*)kmalloc(km, n_u * 16);
|
||||||
@@ -252,25 +252,52 @@ void mm_sync_regs(void *km, int n_regs, mm_reg1_t *regs) // keep mm_reg1_t::{id,
|
|||||||
mm_set_sam_pri(n_regs, regs);
|
mm_set_sam_pri(n_regs, regs);
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int *n_, mm_reg1_t *r)
|
void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int check_strand, int min_strand_sc, int *n_, mm_reg1_t *r)
|
||||||
{
|
{
|
||||||
if (pri_ratio > 0.0f && *n_ > 0) {
|
if (pri_ratio > 0.0f && *n_ > 0) {
|
||||||
int i, k, n = *n_, n_2nd = 0;
|
int i, k, n = *n_, n_2nd = 0;
|
||||||
for (i = k = 0; i < n; ++i) {
|
uint8_t *keep = (uint8_t*)kmalloc(km, n);
|
||||||
|
for (i = 0; i < n; ++i) {
|
||||||
int p = r[i].parent;
|
int p = r[i].parent;
|
||||||
|
keep[i] = 0;
|
||||||
if (p == i || r[i].inv) { // primary or inversion
|
if (p == i || r[i].inv) { // primary or inversion
|
||||||
r[k++] = r[i];
|
keep[i] = 1;
|
||||||
} else if ((r[i].score >= r[p].score * pri_ratio || r[i].score + min_diff >= r[p].score) && n_2nd < best_n) {
|
} else if ((r[i].score >= r[p].score * pri_ratio || r[i].score + min_diff >= r[p].score) && n_2nd < best_n) {
|
||||||
if (!(r[i].qs == r[p].qs && r[i].qe == r[p].qe && r[i].rid == r[p].rid && r[i].rs == r[p].rs && r[i].re == r[p].re)) // not identical hits
|
if (!(r[i].qs == r[p].qs && r[i].qe == r[p].qe && r[i].rid == r[p].rid && r[i].rs == r[p].rs && r[i].re == r[p].re)) // not identical hits
|
||||||
r[k++] = r[i], ++n_2nd;
|
keep[i] = 1, ++n_2nd;
|
||||||
else if (r[i].p) free(r[i].p);
|
} else if (check_strand && n_2nd < best_n && r[i].score > min_strand_sc && r[i].rev != r[p].rev) {
|
||||||
} else if (r[i].p) free(r[i].p);
|
r[i].strand_retained = 1;
|
||||||
|
keep[i] = 1, ++n_2nd;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
for (i = k = 0; i < n; ++i) {
|
||||||
|
if (keep[i]) r[k++] = r[i];
|
||||||
|
else if (r[i].p) free(r[i].p);
|
||||||
|
}
|
||||||
|
kfree(km, keep);
|
||||||
if (k != n) mm_sync_regs(km, k, r); // removing hits requires sync()
|
if (k != n) mm_sync_regs(km, k, r); // removing hits requires sync()
|
||||||
*n_ = k;
|
*n_ = k;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
int mm_filter_strand_retained(int n_regs, mm_reg1_t *r)
|
||||||
|
{
|
||||||
|
int i, k;
|
||||||
|
uint8_t *keep = (uint8_t*)malloc(n_regs);
|
||||||
|
for (i = 0; i < n_regs; ++i) {
|
||||||
|
int p = r[i].parent;
|
||||||
|
keep[i] = (!r[i].strand_retained || r[i].div < r[p].div * 5.0f || r[i].div < 0.01f);
|
||||||
|
}
|
||||||
|
for (i = k = 0; i < n_regs; ++i) {
|
||||||
|
if (keep[i]) {
|
||||||
|
if (k < i) r[k++] = r[i];
|
||||||
|
else ++k;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
free(keep);
|
||||||
|
return k;
|
||||||
|
}
|
||||||
|
|
||||||
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs)
|
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs)
|
||||||
{ // NB: after this call, mm_reg1_t::parent can be -1 if its parent filtered out
|
{ // NB: after this call, mm_reg1_t::parent can be -1 if its parent filtered out
|
||||||
int i, k;
|
int i, k;
|
||||||
@@ -402,16 +429,19 @@ static void mm_set_inv_mapq(void *km, int n_regs, mm_reg1_t *regs)
|
|||||||
kfree(km, aux);
|
kfree(km, aux);
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_set_mapq(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr)
|
void mm_set_mapq2(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr, int is_splice)
|
||||||
{
|
{
|
||||||
static const float q_coef = 40.0f;
|
static const float q_coef = 40.0f;
|
||||||
int64_t sum_sc = 0;
|
int64_t sum_sc = 0;
|
||||||
float uniq_ratio;
|
float uniq_ratio;
|
||||||
int i;
|
int i, n_2nd_splice = 0;
|
||||||
if (n_regs == 0) return;
|
if (n_regs == 0) return;
|
||||||
for (i = 0; i < n_regs; ++i)
|
for (i = 0; i < n_regs; ++i) {
|
||||||
if (regs[i].parent == regs[i].id)
|
if (regs[i].parent == regs[i].id)
|
||||||
sum_sc += regs[i].score;
|
sum_sc += regs[i].score;
|
||||||
|
else if (regs[i].is_spliced)
|
||||||
|
++n_2nd_splice;
|
||||||
|
}
|
||||||
uniq_ratio = (float)sum_sc / (sum_sc + rep_len);
|
uniq_ratio = (float)sum_sc / (sum_sc + rep_len);
|
||||||
for (i = 0; i < n_regs; ++i) {
|
for (i = 0; i < n_regs; ++i) {
|
||||||
mm_reg1_t *r = ®s[i];
|
mm_reg1_t *r = ®s[i];
|
||||||
@@ -424,13 +454,18 @@ void mm_set_mapq(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int ma
|
|||||||
pen_cm = pen_s1 < pen_cm? pen_s1 : pen_cm;
|
pen_cm = pen_s1 < pen_cm? pen_s1 : pen_cm;
|
||||||
subsc = r->subsc > min_chain_sc? r->subsc : min_chain_sc;
|
subsc = r->subsc > min_chain_sc? r->subsc : min_chain_sc;
|
||||||
if (r->p && r->p->dp_max2 > 0 && r->p->dp_max > 0) {
|
if (r->p && r->p->dp_max2 > 0 && r->p->dp_max > 0) {
|
||||||
float identity = (float)r->mlen / r->blen;
|
float x, identity = (float)r->mlen / r->blen;
|
||||||
float x = (float)r->p->dp_max2 * subsc / r->p->dp_max / r->score0;
|
if (is_sr && is_splice)
|
||||||
|
x = (float)r->p->dp_max2 / r->p->dp_max; // ignore chaining score; for short RNA-seq reads, unspliced chaining score tends to be higher
|
||||||
|
else
|
||||||
|
x = (float)r->p->dp_max2 * subsc / r->p->dp_max / r->score0;
|
||||||
mapq = (int)(identity * pen_cm * q_coef * (1.0f - x * x) * logf((float)r->p->dp_max / match_sc));
|
mapq = (int)(identity * pen_cm * q_coef * (1.0f - x * x) * logf((float)r->p->dp_max / match_sc));
|
||||||
if (!is_sr) {
|
if (!is_sr) {
|
||||||
int mapq_alt = (int)(6.02f * identity * identity * (r->p->dp_max - r->p->dp_max2) / match_sc + .499f); // BWA-MEM like mapQ, mostly for short reads
|
int mapq_alt = (int)(6.02f * identity * identity * (r->p->dp_max - r->p->dp_max2) / match_sc + .499f); // BWA-MEM like mapQ, mostly for short reads
|
||||||
mapq = mapq < mapq_alt? mapq : mapq_alt; // in case the long-read heuristic fails
|
mapq = mapq < mapq_alt? mapq : mapq_alt; // in case the long-read heuristic fails
|
||||||
}
|
}
|
||||||
|
if (is_splice && is_sr && r->is_spliced && n_2nd_splice == 0)
|
||||||
|
mapq += 10;
|
||||||
} else {
|
} else {
|
||||||
float x = (float)subsc / r->score0;
|
float x = (float)subsc / r->score0;
|
||||||
if (r->p) {
|
if (r->p) {
|
||||||
|
|||||||
@@ -12,6 +12,7 @@
|
|||||||
#include "bseq.h"
|
#include "bseq.h"
|
||||||
#include "minimap.h"
|
#include "minimap.h"
|
||||||
#include "mmpriv.h"
|
#include "mmpriv.h"
|
||||||
|
#include "ksw2.h"
|
||||||
#include "kvec.h"
|
#include "kvec.h"
|
||||||
#include "khash.h"
|
#include "khash.h"
|
||||||
|
|
||||||
@@ -32,7 +33,7 @@ typedef struct mm_idx_bucket_s {
|
|||||||
} mm_idx_bucket_t;
|
} mm_idx_bucket_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int32_t st, en, max; // max is not used for now
|
int32_t st, en, cnt;
|
||||||
int32_t score:30, strand:2;
|
int32_t score:30, strand:2;
|
||||||
} mm_idx_intv1_t;
|
} mm_idx_intv1_t;
|
||||||
|
|
||||||
@@ -41,6 +42,11 @@ typedef struct mm_idx_intv_s {
|
|||||||
mm_idx_intv1_t *a;
|
mm_idx_intv1_t *a;
|
||||||
} mm_idx_intv_t;
|
} mm_idx_intv_t;
|
||||||
|
|
||||||
|
typedef struct mm_idx_jjump_s {
|
||||||
|
int32_t n, m;
|
||||||
|
mm_idx_jjump1_t *a;
|
||||||
|
} mm_idx_jjump_t;
|
||||||
|
|
||||||
mm_idx_t *mm_idx_init(int w, int k, int b, int flag)
|
mm_idx_t *mm_idx_init(int w, int k, int b, int flag)
|
||||||
{
|
{
|
||||||
mm_idx_t *mi;
|
mm_idx_t *mi;
|
||||||
@@ -65,11 +71,17 @@ void mm_idx_destroy(mm_idx_t *mi)
|
|||||||
kh_destroy(idx, (idxhash_t*)mi->B[i].h);
|
kh_destroy(idx, (idxhash_t*)mi->B[i].h);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if (mi->spsc) free(mi->spsc);
|
||||||
if (mi->I) {
|
if (mi->I) {
|
||||||
for (i = 0; i < mi->n_seq; ++i)
|
for (i = 0; i < mi->n_seq; ++i)
|
||||||
free(mi->I[i].a);
|
free(mi->I[i].a);
|
||||||
free(mi->I);
|
free(mi->I);
|
||||||
}
|
}
|
||||||
|
if (mi->J) {
|
||||||
|
for (i = 0; i < mi->n_seq; ++i)
|
||||||
|
free(mi->J[i].a);
|
||||||
|
free(mi->J);
|
||||||
|
}
|
||||||
if (!mi->km) {
|
if (!mi->km) {
|
||||||
for (i = 0; i < mi->n_seq; ++i)
|
for (i = 0; i < mi->n_seq; ++i)
|
||||||
free(mi->seq[i].name);
|
free(mi->seq[i].name);
|
||||||
@@ -99,7 +111,7 @@ const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n)
|
|||||||
|
|
||||||
void mm_idx_stat(const mm_idx_t *mi)
|
void mm_idx_stat(const mm_idx_t *mi)
|
||||||
{
|
{
|
||||||
int n = 0, n1 = 0;
|
int64_t n = 0, n1 = 0;
|
||||||
uint32_t i;
|
uint32_t i;
|
||||||
uint64_t sum = 0, len = 0;
|
uint64_t sum = 0, len = 0;
|
||||||
fprintf(stderr, "[M::%s] kmer size: %d; skip: %d; is_hpc: %d; #seq: %d\n", __func__, mi->k, mi->w, mi->flag&MM_I_HPC, mi->n_seq);
|
fprintf(stderr, "[M::%s] kmer size: %d; skip: %d; is_hpc: %d; #seq: %d\n", __func__, mi->k, mi->w, mi->flag&MM_I_HPC, mi->n_seq);
|
||||||
@@ -117,8 +129,8 @@ void mm_idx_stat(const mm_idx_t *mi)
|
|||||||
if (kh_key(h, k)&1) ++n1;
|
if (kh_key(h, k)&1) ++n1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
fprintf(stderr, "[M::%s::%.3f*%.2f] distinct minimizers: %d (%.2f%% are singletons); average occurrences: %.3lf; average spacing: %.3lf; total length: %ld\n",
|
fprintf(stderr, "[M::%s::%.3f*%.2f] distinct minimizers: %ld (%.2f%% are singletons); average occurrences: %.3lf; average spacing: %.3lf; total length: %ld\n",
|
||||||
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), n, 100.0*n1/n, (double)sum / n, (double)len / sum, (long)len);
|
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), (long)n, 100.0*n1/n, (double)sum / n, (double)len / sum, (long)len);
|
||||||
}
|
}
|
||||||
|
|
||||||
int mm_idx_index_name(mm_idx_t *mi)
|
int mm_idx_index_name(mm_idx_t *mi)
|
||||||
@@ -192,6 +204,7 @@ int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f)
|
|||||||
if (f <= 0.) return INT32_MAX;
|
if (f <= 0.) return INT32_MAX;
|
||||||
for (i = 0; i < 1<<mi->b; ++i)
|
for (i = 0; i < 1<<mi->b; ++i)
|
||||||
if (mi->B[i].h) n += kh_size((idxhash_t*)mi->B[i].h);
|
if (mi->B[i].h) n += kh_size((idxhash_t*)mi->B[i].h);
|
||||||
|
if (n == 0) return INT32_MAX;
|
||||||
a = (uint32_t*)malloc(n * 4);
|
a = (uint32_t*)malloc(n * 4);
|
||||||
for (i = n = 0; i < 1<<mi->b; ++i) {
|
for (i = n = 0; i < 1<<mi->b; ++i) {
|
||||||
idxhash_t *h = (idxhash_t*)mi->B[i].h;
|
idxhash_t *h = (idxhash_t*)mi->B[i].h;
|
||||||
@@ -656,10 +669,17 @@ int mm_idx_alt_read(mm_idx_t *mi, const char *fn)
|
|||||||
return n_alt;
|
return n_alt;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/***************
|
||||||
|
* BED reading *
|
||||||
|
***************/
|
||||||
|
|
||||||
#define sort_key_bed(a) ((a).st)
|
#define sort_key_bed(a) ((a).st)
|
||||||
KRADIX_SORT_INIT(bed, mm_idx_intv1_t, sort_key_bed, 4)
|
KRADIX_SORT_INIT(bed, mm_idx_intv1_t, sort_key_bed, 4)
|
||||||
|
|
||||||
mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc)
|
#define sort_key_end(a) ((a).en)
|
||||||
|
KRADIX_SORT_INIT(end, mm_idx_intv1_t, sort_key_end, 4)
|
||||||
|
|
||||||
|
static mm_idx_intv_t *mm_idx_bed_read_core(const mm_idx_t *mi, const char *fn, int read_junc, int min_sc)
|
||||||
{
|
{
|
||||||
gzFile fp;
|
gzFile fp;
|
||||||
kstream_t *ks;
|
kstream_t *ks;
|
||||||
@@ -668,7 +688,7 @@ mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc
|
|||||||
|
|
||||||
fp = fn && strcmp(fn, "-")? gzopen(fn, "r") : gzdopen(fileno(stdin), "r");
|
fp = fn && strcmp(fn, "-")? gzopen(fn, "r") : gzdopen(fileno(stdin), "r");
|
||||||
if (fp == 0) return 0;
|
if (fp == 0) return 0;
|
||||||
I = (mm_idx_intv_t*)calloc(mi->n_seq, sizeof(*I));
|
I = CALLOC(mm_idx_intv_t, mi->n_seq);
|
||||||
ks = ks_init(fp);
|
ks = ks_init(fp);
|
||||||
while (ks_getuntil(ks, KS_SEP_LINE, &str, 0) >= 0) {
|
while (ks_getuntil(ks, KS_SEP_LINE, &str, 0) >= 0) {
|
||||||
mm_idx_intv_t *r;
|
mm_idx_intv_t *r;
|
||||||
@@ -689,7 +709,7 @@ mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc
|
|||||||
t.en = atol(q);
|
t.en = atol(q);
|
||||||
if (t.en < 0) break;
|
if (t.en < 0) break;
|
||||||
} else if (i == 4) { // BED score
|
} else if (i == 4) { // BED score
|
||||||
t.score = atol(q);
|
t.score = *q >= '0' && *q <= '9'? atol(q) : -1;
|
||||||
} else if (i == 5) { // strand
|
} else if (i == 5) { // strand
|
||||||
t.strand = *q == '+'? 1 : *q == '-'? -1 : 0;
|
t.strand = *q == '+'? 1 : *q == '-'? -1 : 0;
|
||||||
} else if (i == 9) {
|
} else if (i == 9) {
|
||||||
@@ -705,7 +725,8 @@ mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc
|
|||||||
++i, q = p + 1;
|
++i, q = p + 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (id < 0 || t.st < 0 || t.st >= t.en) continue;
|
if (id < 0 || t.st < 0 || t.st >= t.en) continue; // contig ID not found, or other problems
|
||||||
|
if (min_sc > 0 && t.score < min_sc) continue;
|
||||||
r = &I[id];
|
r = &I[id];
|
||||||
if (i >= 11 && read_junc) { // BED12
|
if (i >= 11 && read_junc) { // BED12
|
||||||
int32_t st, sz, en;
|
int32_t st, sz, en;
|
||||||
@@ -738,14 +759,44 @@ mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc
|
|||||||
return I;
|
return I;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static mm_idx_intv_t *mm_idx_bed_read_merge(const mm_idx_t *mi, const char *fn, int read_junc, int min_sc)
|
||||||
|
{
|
||||||
|
long n = 0, n0 = 0;
|
||||||
|
int32_t i;
|
||||||
|
mm_idx_intv_t *I;
|
||||||
|
I = mm_idx_bed_read_core(mi, fn, read_junc, min_sc);
|
||||||
|
if (I == 0) return 0;
|
||||||
|
for (i = 0; i < mi->n_seq; ++i) {
|
||||||
|
int32_t j, j0, k;
|
||||||
|
mm_idx_intv_t *intv = &I[i];
|
||||||
|
n0 += intv->n;
|
||||||
|
radix_sort_bed(intv->a, intv->a + intv->n); // sort by st
|
||||||
|
for (j = 1, j0 = 0; j <= intv->n; ++j) { // sort by st and then by end
|
||||||
|
if (j == intv->n || intv->a[j].st != intv->a[j0].st) {
|
||||||
|
radix_sort_end(intv->a + j0, intv->a + j);
|
||||||
|
j0 = j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for (j = 1, j0 = 0, k = 0; j <= intv->n; ++j) { // merge intervals with the same (st, en)
|
||||||
|
if (j == intv->n || intv->a[j].st != intv->a[j0].st || intv->a[j].en != intv->a[j0].en) {
|
||||||
|
intv->a[k] = intv->a[j0];
|
||||||
|
intv->a[k++].cnt = j - j0;
|
||||||
|
j0 = j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
intv->a = REALLOC(mm_idx_intv1_t, intv->a, k);
|
||||||
|
intv->n = intv->m = k;
|
||||||
|
n += k;
|
||||||
|
}
|
||||||
|
if (mm_verbose >= 3)
|
||||||
|
fprintf(stderr, "[%s] read %ld introns, %ld of which are non-redundant\n", __func__, n0, n);
|
||||||
|
return I;
|
||||||
|
}
|
||||||
|
|
||||||
int mm_idx_bed_read(mm_idx_t *mi, const char *fn, int read_junc)
|
int mm_idx_bed_read(mm_idx_t *mi, const char *fn, int read_junc)
|
||||||
{
|
{
|
||||||
int32_t i;
|
|
||||||
if (mi->h == 0) mm_idx_index_name(mi);
|
if (mi->h == 0) mm_idx_index_name(mi);
|
||||||
mi->I = mm_idx_read_bed(mi, fn, read_junc);
|
mi->I = mm_idx_bed_read_merge(mi, fn, read_junc, -1);
|
||||||
if (mi->I == 0) return -1;
|
|
||||||
for (i = 0; i < mi->n_seq; ++i) // TODO: eliminate redundant intervals
|
|
||||||
radix_sort_bed(mi->I[i].a, mi->I[i].a + mi->I[i].n);
|
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -773,3 +824,251 @@ int mm_idx_bed_junc(const mm_idx_t *mi, int32_t ctg, int32_t st, int32_t en, uin
|
|||||||
}
|
}
|
||||||
return left;
|
return left;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/*********************************
|
||||||
|
* Reading junctions for jumping *
|
||||||
|
*********************************/
|
||||||
|
|
||||||
|
#define sort_key_jj(a) ((a).off)
|
||||||
|
KRADIX_SORT_INIT(jj, mm_idx_jjump1_t, sort_key_jj, 4)
|
||||||
|
|
||||||
|
#define sort_key_jj2(a) ((a).off2)
|
||||||
|
KRADIX_SORT_INIT(jj2, mm_idx_jjump1_t, sort_key_jj2, 4)
|
||||||
|
|
||||||
|
static void sort_jjump(mm_idx_jjump_t *jj2)
|
||||||
|
{
|
||||||
|
int32_t j0, j, k;
|
||||||
|
if (jj2 == 0 || jj2->n == 0) return;
|
||||||
|
radix_sort_jj(jj2->a, jj2->a + jj2->n);
|
||||||
|
for (j0 = 0, j = 1; j <= jj2->n; ++j) {
|
||||||
|
if (j == jj2->n || jj2->a[j0].off != jj2->a[j].off) {
|
||||||
|
radix_sort_jj2(jj2->a + j0, jj2->a + j);
|
||||||
|
j0 = j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
// the actual merge
|
||||||
|
for (j0 = 0, j = 1, k = 0; j <= jj2->n; ++j) {
|
||||||
|
if (j == jj2->n || jj2->a[j0].off != jj2->a[j].off || jj2->a[j0].off2 != jj2->a[j].off2) {
|
||||||
|
int32_t t, cnt = 0;
|
||||||
|
uint16_t flag = 0;
|
||||||
|
for (t = j0; t < j; ++t) cnt += jj2->a[t].cnt, flag |= jj2->a[t].flag;
|
||||||
|
jj2->a[k] = jj2->a[j0];
|
||||||
|
jj2->a[k].cnt = cnt;
|
||||||
|
jj2->a[k++].flag = flag;
|
||||||
|
j0 = j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
jj2->n = k;
|
||||||
|
jj2->a = REALLOC(mm_idx_jjump1_t, jj2->a, k);
|
||||||
|
}
|
||||||
|
|
||||||
|
static mm_idx_jjump_t *mm_idx_bed2jjump(const mm_idx_t *mi, const mm_idx_intv_t *I, uint16_t flag)
|
||||||
|
{
|
||||||
|
int32_t i;
|
||||||
|
mm_idx_jjump_t *J;
|
||||||
|
J = CALLOC(mm_idx_jjump_t, mi->n_seq);
|
||||||
|
for (i = 0; i < mi->n_seq; ++i) {
|
||||||
|
int32_t j, k;
|
||||||
|
const mm_idx_intv_t *intv = &I[i];
|
||||||
|
mm_idx_jjump_t *jj = &J[i];
|
||||||
|
jj->n = intv->n * 2;
|
||||||
|
jj->a = CALLOC(mm_idx_jjump1_t, jj->n);
|
||||||
|
for (j = k = 0; j < intv->n; ++j) {
|
||||||
|
jj->a[k].off = intv->a[j].st, jj->a[k].off2 = intv->a[j].en, jj->a[k].cnt = intv->a[j].cnt, jj->a[k].strand = intv->a[j].strand, jj->a[k++].flag = flag;
|
||||||
|
jj->a[k].off = intv->a[j].en, jj->a[k].off2 = intv->a[j].st, jj->a[k].cnt = intv->a[j].cnt, jj->a[k].strand = intv->a[j].strand, jj->a[k++].flag = flag;
|
||||||
|
}
|
||||||
|
sort_jjump(jj);
|
||||||
|
}
|
||||||
|
return J;
|
||||||
|
}
|
||||||
|
|
||||||
|
static mm_idx_jjump_t *mm_idx_jjump_merge(const mm_idx_t *mi, const mm_idx_jjump_t *J0, const mm_idx_jjump_t *J1)
|
||||||
|
{
|
||||||
|
int32_t i;
|
||||||
|
mm_idx_jjump_t *J2;
|
||||||
|
J2 = CALLOC(mm_idx_jjump_t, mi->n_seq);
|
||||||
|
for (i = 0; i < mi->n_seq; ++i) {
|
||||||
|
int32_t j, k;
|
||||||
|
const mm_idx_jjump_t *jj0 = &J0[i], *jj1 = &J1[i];
|
||||||
|
mm_idx_jjump_t *jj2 = &J2[i];
|
||||||
|
jj2->n = jj0->n + jj1->n;
|
||||||
|
jj2->a = CALLOC(mm_idx_jjump1_t, jj2->n);
|
||||||
|
for (j = k = 0; j < jj0->n; ++j) jj2->a[k++] = jj0->a[j];
|
||||||
|
for (j = 0; j < jj1->n; ++j) jj2->a[k++] = jj1->a[j];
|
||||||
|
sort_jjump(jj2);
|
||||||
|
}
|
||||||
|
return J2;
|
||||||
|
}
|
||||||
|
|
||||||
|
int mm_idx_jjump_read(mm_idx_t *mi, const char *fn, int flag, int min_sc)
|
||||||
|
{
|
||||||
|
int32_t i, j, n_anno = 0, n_misc = 0;
|
||||||
|
mm_idx_intv_t *I;
|
||||||
|
mm_idx_jjump_t *J;
|
||||||
|
if (mi->h == 0) mm_idx_index_name(mi);
|
||||||
|
I = mm_idx_bed_read_merge(mi, fn, 1, min_sc);
|
||||||
|
J = mm_idx_bed2jjump(mi, I, flag);
|
||||||
|
for (i = 0; i < mi->n_seq; ++i) free(I[i].a);
|
||||||
|
free(I);
|
||||||
|
if (mi->J) {
|
||||||
|
mm_idx_jjump_t *J2;
|
||||||
|
J2 = mm_idx_jjump_merge(mi, mi->J, J);
|
||||||
|
for (i = 0; i < mi->n_seq; ++i) {
|
||||||
|
free(mi->J[i].a); free(J[i].a);
|
||||||
|
}
|
||||||
|
free(mi->J); free(J);
|
||||||
|
mi->J = J2;
|
||||||
|
} else mi->J = J;
|
||||||
|
for (i = 0; i < mi->n_seq; ++i) {
|
||||||
|
for (j = 0; j < mi->J[i].n; ++j)
|
||||||
|
if (mi->J[i].a[j].flag & MM_JUNC_ANNO) ++n_anno;
|
||||||
|
else ++n_misc;
|
||||||
|
}
|
||||||
|
if (mm_verbose >= 3)
|
||||||
|
fprintf(stderr, "[%s] there are %d annotated and %d other splice positions in the index\n", __func__, n_anno, n_misc);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
static int32_t mm_idx_jump_get_core(int32_t n, const mm_idx_jjump1_t *a, int32_t x) // similar to mm_idx_find_intv()
|
||||||
|
{
|
||||||
|
int32_t s = 0, e = n;
|
||||||
|
if (n == 0) return -1;
|
||||||
|
if (x < a[0].off) return -1;
|
||||||
|
while (s < e) {
|
||||||
|
int32_t mid = s + (e - s) / 2;
|
||||||
|
if (x >= a[mid].off && (mid + 1 >= n || x < a[mid+1].off)) return mid;
|
||||||
|
else if (x < a[mid].off) e = mid;
|
||||||
|
else s = mid + 1;
|
||||||
|
}
|
||||||
|
assert(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
const mm_idx_jjump1_t *mm_idx_jump_get(const mm_idx_t *db, int32_t cid, int32_t st, int32_t en, int32_t *n)
|
||||||
|
{
|
||||||
|
mm_idx_jjump_t *s;
|
||||||
|
int32_t l, r;
|
||||||
|
*n = 0;
|
||||||
|
if (cid >= db->n_seq || cid < 0 || db->J == 0) return 0;
|
||||||
|
if (en < 0 || en > db->seq[cid].len) en = db->seq[cid].len;
|
||||||
|
s = &db->J[cid];
|
||||||
|
if (s->n == 0) return 0;
|
||||||
|
l = mm_idx_jump_get_core(s->n, s->a, st);
|
||||||
|
r = mm_idx_jump_get_core(s->n, s->a, en);
|
||||||
|
*n = r - l;
|
||||||
|
return &s->a[l + 1];
|
||||||
|
}
|
||||||
|
|
||||||
|
/****************
|
||||||
|
* splice score *
|
||||||
|
****************/
|
||||||
|
|
||||||
|
typedef struct mm_idx_spsc_s {
|
||||||
|
uint32_t n, m;
|
||||||
|
uint64_t *a; // pos<<56 | score<<1 | acceptor
|
||||||
|
} mm_idx_spsc_t;
|
||||||
|
|
||||||
|
int32_t mm_idx_spsc_read2(mm_idx_t *idx, const char *fn, int32_t max_sc, float scale)
|
||||||
|
{
|
||||||
|
gzFile fp;
|
||||||
|
kstring_t str = {0,0,0};
|
||||||
|
kstream_t *ks;
|
||||||
|
int32_t dret, j;
|
||||||
|
int64_t n_read = 0;
|
||||||
|
|
||||||
|
fp = fn && strcmp(fn, "-") != 0? gzopen(fn, "rb") : gzdopen(0, "rb");
|
||||||
|
if (fp == 0) return -1;
|
||||||
|
if (idx->h == 0) mm_idx_index_name(idx);
|
||||||
|
if (max_sc > 63) max_sc = 63;
|
||||||
|
idx->spsc = Kcalloc(0, mm_idx_spsc_t, idx->n_seq * 2);
|
||||||
|
ks = ks_init(fp);
|
||||||
|
while (ks_getuntil(ks, KS_SEP_LINE, &str, &dret) >= 0) {
|
||||||
|
mm_idx_spsc_t *s;
|
||||||
|
char *p, *q, *name = 0;
|
||||||
|
int32_t i, type = -1, strand = 0, cid = -1, score = -1;
|
||||||
|
int64_t pos = -1;
|
||||||
|
for (i = 0, p = q = str.s;; ++p) {
|
||||||
|
if (*p == '\t' || *p == 0) {
|
||||||
|
int c = *p;
|
||||||
|
*p = 0;
|
||||||
|
if (i == 0) {
|
||||||
|
name = q;
|
||||||
|
} else if (i == 1) {
|
||||||
|
pos = atol(q);
|
||||||
|
} else if (i == 2) {
|
||||||
|
strand = *q == '+'? 1 : '-'? -1 : 0;
|
||||||
|
} else if (i == 3) {
|
||||||
|
type = *q == 'D'? 0 : *q == 'A'? 1 : -1;
|
||||||
|
} else if (i == 4) {
|
||||||
|
score = atoi(q);
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
if (c == 0) break;
|
||||||
|
q = p + 1, ++i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (i < 4) continue; // not enough fields
|
||||||
|
if (scale > 0.0f && scale < 1.0f)
|
||||||
|
score = score > 0.0f? (int)(score * scale + .499) : (int)(score * scale - .499);
|
||||||
|
if (score > max_sc) score = max_sc;
|
||||||
|
if (score < -max_sc) score = -max_sc;
|
||||||
|
cid = mm_idx_name2id(idx, name);
|
||||||
|
if (cid < 0 || type < 0 || strand == 0 || pos < 0) continue; // FIXME: give a warning!
|
||||||
|
s = &idx->spsc[cid << 1 | (strand > 0? 0 : 1)];
|
||||||
|
Kgrow(0, uint64_t, s->a, s->n, s->m);
|
||||||
|
if (pos > 0 && pos < idx->seq[cid].len) { // ignore scores at the ends
|
||||||
|
s->a[s->n++] = (uint64_t)pos << 8 | (score + KSW_SPSC_OFFSET) << 1 | type;
|
||||||
|
++n_read;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ks_destroy(ks);
|
||||||
|
gzclose(fp);
|
||||||
|
for (j = 0; j < idx->n_seq * 2; ++j) {
|
||||||
|
mm_idx_spsc_t *s = &idx->spsc[j];
|
||||||
|
if (s->n > 0)
|
||||||
|
radix_sort_64(s->a, s->a + s->n);
|
||||||
|
}
|
||||||
|
if (mm_verbose >= 3)
|
||||||
|
fprintf(stderr, "[M::%s] read %ld splice scores\n", __func__, (long)n_read);
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
int32_t mm_idx_spsc_read(mm_idx_t *idx, const char *fn, int32_t max_sc)
|
||||||
|
{
|
||||||
|
return mm_idx_spsc_read2(idx, fn, max_sc, 1.0f);
|
||||||
|
}
|
||||||
|
|
||||||
|
static int32_t mm_idx_find_intv(int32_t n, const uint64_t *a, int64_t x)
|
||||||
|
{
|
||||||
|
int32_t s = 0, e = n;
|
||||||
|
if (n == 0) return -1;
|
||||||
|
if (x < a[0]>>8) return -1;
|
||||||
|
while (s < e) {
|
||||||
|
int32_t mid = s + (e - s) / 2;
|
||||||
|
if (x >= a[mid]>>8 && (mid + 1 >= n || x < a[mid+1]>>8)) return mid;
|
||||||
|
else if (x < a[mid]>>8) e = mid;
|
||||||
|
else s = mid + 1;
|
||||||
|
}
|
||||||
|
assert(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
int64_t mm_idx_spsc_get(const mm_idx_t *db, int32_t cid, int64_t st, int64_t en, int32_t rev, uint8_t *sc)
|
||||||
|
{
|
||||||
|
const mm_idx_spsc_t *s;
|
||||||
|
if (cid >= db->n_seq || cid < 0 || db->spsc == 0) return -1;
|
||||||
|
if (en < 0 || en > db->seq[cid].len) en = db->seq[cid].len;
|
||||||
|
memset(sc, 0xff, en - st);
|
||||||
|
s = &db->spsc[cid << 1 | (!!rev)];
|
||||||
|
if (s->n > 0) {
|
||||||
|
int32_t j, l, r;
|
||||||
|
l = mm_idx_find_intv(s->n, s->a, st);
|
||||||
|
r = mm_idx_find_intv(s->n, s->a, en);
|
||||||
|
for (j = l + 1; j <= r; ++j) {
|
||||||
|
int64_t x = (s->a[j]>>8) - st;
|
||||||
|
uint8_t score = s->a[j] & 0xff;
|
||||||
|
assert(x <= en - st);
|
||||||
|
if (x == en - st) continue;
|
||||||
|
if (sc[x] == 0xff || sc[x] < score) sc[x] = score;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return en - st;
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,201 @@
|
|||||||
|
#include <stdio.h>
|
||||||
|
#include "mmpriv.h"
|
||||||
|
#include "kalloc.h"
|
||||||
|
|
||||||
|
#define MM_MIN_EXON_LEN 20
|
||||||
|
|
||||||
|
static int32_t mm_jump_check(void *km, const mm_idx_t *mi, int32_t qlen, const uint8_t *qseq0, const mm_reg1_t *r, int32_t ext, int32_t is_left) // TODO: check close N
|
||||||
|
{
|
||||||
|
int32_t clip, clen, e = !r->rev ^ !is_left; // 0 for left of the alignment; 1 for right
|
||||||
|
uint32_t cigar;
|
||||||
|
if (!r->p || r->p->n_cigar <= 0) return -1; // only working with CIGAR
|
||||||
|
clip = e == 0? r->qs : qlen - r->qe;
|
||||||
|
cigar = r->p->cigar[is_left? 0 : r->p->n_cigar - 1];
|
||||||
|
clen = (cigar&0xf) == MM_CIGAR_MATCH? cigar>>4 : 0;
|
||||||
|
if (clen <= ext) return -1;
|
||||||
|
if (is_left) {
|
||||||
|
if (clip >= r->rs) return -1; // no space to jump
|
||||||
|
} else {
|
||||||
|
if (clip >= mi->seq[r->rid].len - r->re) return -1; // no space to jump
|
||||||
|
}
|
||||||
|
return 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
static uint8_t *mm_jump_get_qseq_seq(void *km, int32_t qlen, const uint8_t *qseq0, const mm_reg1_t *r, int32_t is_left, int32_t ql0, uint8_t *qseq)
|
||||||
|
{
|
||||||
|
extern unsigned char seq_nt4_table[256];
|
||||||
|
int32_t i, k = 0;
|
||||||
|
if (!r->rev) {
|
||||||
|
if (is_left)
|
||||||
|
for (i = 0; i < ql0; ++i)
|
||||||
|
qseq[k++] = seq_nt4_table[(uint8_t)qseq0[i]];
|
||||||
|
else
|
||||||
|
for (i = qlen - ql0; i < qlen; ++i)
|
||||||
|
qseq[k++] = seq_nt4_table[(uint8_t)qseq0[i]];
|
||||||
|
} else {
|
||||||
|
if (is_left)
|
||||||
|
for (i = qlen - 1; i >= qlen - ql0; --i) {
|
||||||
|
uint8_t c = seq_nt4_table[(uint8_t)qseq0[i]];
|
||||||
|
qseq[k++] = c >= 4? c : 3 - c;
|
||||||
|
}
|
||||||
|
else
|
||||||
|
for (i = ql0 - 1; i >= 0; --i) {
|
||||||
|
uint8_t c = seq_nt4_table[(uint8_t)qseq0[i]];
|
||||||
|
qseq[k++] = c >= 4? c : 3 - c;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return qseq;
|
||||||
|
}
|
||||||
|
|
||||||
|
static void mm_jump_split_left(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq0, mm_reg1_t *r, int32_t ts_strand)
|
||||||
|
{
|
||||||
|
uint8_t *tseq = 0, *qseq = 0;
|
||||||
|
int32_t i, n, l, i0, m, mm0;
|
||||||
|
int32_t i0_anno = -1, n_anno = 0, mm0_anno = 0, i0_misc = -1, n_misc = 0, mm0_misc = 0;
|
||||||
|
int32_t ext = 1 + (opt->b + opt->a - 1) / opt->a + 1;
|
||||||
|
int32_t clip = !r->rev? r->qs : qlen - r->qe;
|
||||||
|
int32_t extt = clip < ext? clip : ext;
|
||||||
|
const mm_idx_jjump1_t *a;
|
||||||
|
|
||||||
|
if (mm_jump_check(km, mi, qlen, qseq0, r, ext + MM_MIN_EXON_LEN, 1) < 0) return;
|
||||||
|
a = mm_idx_jump_get(mi, r->rid, r->rs - extt, r->rs + ext, &n);
|
||||||
|
if (n == 0) return;
|
||||||
|
|
||||||
|
for (i = 0; i < n; ++i) { // traverse possible jumps
|
||||||
|
const mm_idx_jjump1_t *ai = &a[i];
|
||||||
|
int32_t tlen, tl1, j, mm1, mm2;
|
||||||
|
assert(ai->off >= r->rs - extt && ai->off <= r->rs + ext);
|
||||||
|
if (ts_strand * ai->strand < 0) continue; // wrong strand
|
||||||
|
if (ai->off2 >= ai->off) continue; // wrong direction
|
||||||
|
if (ai->off - ai->off2 < 6) continue; // intron too small
|
||||||
|
if (ai->off2 < clip + ext) continue; // not long enough
|
||||||
|
if (tseq == 0) {
|
||||||
|
tseq = Kcalloc(km, uint8_t, (clip + ext) * 2); // tseq and qseq are allocated together
|
||||||
|
qseq = tseq + clip + ext;
|
||||||
|
mm_jump_get_qseq_seq(km, qlen, qseq0, r, 1, clip + ext, qseq);
|
||||||
|
}
|
||||||
|
tl1 = clip + (ai->off - r->rs);
|
||||||
|
tlen = mm_idx_getseq2(mi, 0, r->rid, ai->off, r->rs + ext, &tseq[tl1]);
|
||||||
|
assert(tlen == r->rs + ext - ai->off);
|
||||||
|
tlen = mm_idx_getseq2(mi, 0, r->rid, ai->off2 - tl1, ai->off2, tseq);
|
||||||
|
assert(tlen == tl1);
|
||||||
|
for (j = 0, mm1 = 0; j < tl1; ++j)
|
||||||
|
if (qseq[j] != tseq[j] || qseq[j] > 3 || tseq[j] > 3)
|
||||||
|
++mm1;
|
||||||
|
for (mm2 = 0; j < clip + ext; ++j)
|
||||||
|
if (qseq[j] != tseq[j] || qseq[j] > 3 || tseq[j] > 3)
|
||||||
|
++mm2;
|
||||||
|
if (mm1 == 0 && mm2 <= 1) {
|
||||||
|
if (ai->flag & MM_JUNC_ANNO)
|
||||||
|
i0_anno = i, mm0_anno = mm1 + mm2, ++n_anno; // i0 points to the rightmost i
|
||||||
|
else
|
||||||
|
i0_misc = i, mm0_misc = mm1 + mm2, ++n_misc;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (n_anno > 0) m = n_anno, i0 = i0_anno, mm0 = mm0_anno;
|
||||||
|
else m = n_misc, i0 = i0_misc, mm0 = mm0_misc;
|
||||||
|
kfree(km, tseq);
|
||||||
|
|
||||||
|
l = m > 0? a[i0].off - r->rs : 0; // may be negative
|
||||||
|
if (m == 1 && clip + l >= opt->jump_min_match) { // add one more exon
|
||||||
|
mm_enlarge_cigar(r, 2);
|
||||||
|
memmove(r->p->cigar + 2, r->p->cigar, r->p->n_cigar * 4);
|
||||||
|
r->p->cigar[0] = (clip + l) << 4 | MM_CIGAR_MATCH;
|
||||||
|
r->p->cigar[1] = (a[i0].off - a[i0].off2) << 4 | MM_CIGAR_N_SKIP;
|
||||||
|
r->p->cigar[2] = ((r->p->cigar[2]>>4) - l) << 4 | MM_CIGAR_MATCH;
|
||||||
|
r->p->n_cigar += 2;
|
||||||
|
r->rs = a[i0].off2 - (clip + l);
|
||||||
|
if (!r->rev) r->qs = 0;
|
||||||
|
else r->qe = qlen;
|
||||||
|
r->blen += clip, r->mlen += clip - mm0;
|
||||||
|
r->p->dp_max0 += (clip - mm0) * opt->a - mm0 * opt->b;
|
||||||
|
r->p->dp_max += (clip - mm0) * opt->a - mm0 * opt->b;
|
||||||
|
if (!r->is_spliced) r->is_spliced = 1, r->p->dp_max += (opt->a + opt->b) + ((opt->a + opt->b) >> 1);
|
||||||
|
} else if (m > 0 && a[i0].off > r->rs) { // trim by l; l is always positive
|
||||||
|
r->p->cigar[0] -= l << 4 | MM_CIGAR_MATCH;
|
||||||
|
r->rs += l;
|
||||||
|
if (!r->rev) r->qs += l;
|
||||||
|
else r->qe -= l;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
static void mm_jump_split_right(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq0, mm_reg1_t *r, int32_t ts_strand)
|
||||||
|
{
|
||||||
|
uint8_t *tseq = 0, *qseq = 0;
|
||||||
|
int32_t i, n, l, i0, m, mm0;
|
||||||
|
int32_t i0_anno = -1, n_anno = 0, mm0_anno = 0, i0_misc = -1, n_misc = 0, mm0_misc = 0;
|
||||||
|
int32_t ext = 1 + (opt->b + opt->a - 1) / opt->a + 1;
|
||||||
|
int32_t clip = !r->rev? qlen - r->qe : r->qs;
|
||||||
|
int32_t extt = clip < ext? clip : ext;
|
||||||
|
const mm_idx_jjump1_t *a;
|
||||||
|
|
||||||
|
if (mm_jump_check(km, mi, qlen, qseq0, r, ext + MM_MIN_EXON_LEN, 0) < 0) return;
|
||||||
|
a = mm_idx_jump_get(mi, r->rid, r->re - ext, r->re + extt, &n);
|
||||||
|
if (n == 0) return;
|
||||||
|
|
||||||
|
for (i = 0; i < n; ++i) { // traverse possible jumps
|
||||||
|
const mm_idx_jjump1_t *ai = &a[i];
|
||||||
|
int32_t tlen, tl1, j, mm1, mm2;
|
||||||
|
assert(ai->off >= r->re - ext && ai->off <= r->re + extt);
|
||||||
|
if (ts_strand * ai->strand < 0) continue; // wrong strand
|
||||||
|
if (ai->off2 <= ai->off) continue; // wrong direction
|
||||||
|
if (ai->off2 - ai->off < 6) continue; // intron too small
|
||||||
|
if (ai->off2 + clip + ext > mi->seq[r->rid].len) continue; // not long enough
|
||||||
|
if (tseq == 0) {
|
||||||
|
tseq = Kcalloc(km, uint8_t, (clip + ext) * 2); // tseq and qseq are allocated together
|
||||||
|
qseq = tseq + clip + ext;
|
||||||
|
mm_jump_get_qseq_seq(km, qlen, qseq0, r, 0, clip + ext, qseq);
|
||||||
|
}
|
||||||
|
tl1 = clip + (r->re - ai->off);
|
||||||
|
tlen = mm_idx_getseq2(mi, 0, r->rid, r->re - ext, ai->off, tseq);
|
||||||
|
assert(tlen == ai->off - (r->re - ext));
|
||||||
|
tlen = mm_idx_getseq2(mi, 0, r->rid, ai->off2, ai->off2 + tl1, &tseq[clip + ext - tl1]);
|
||||||
|
assert(tlen == tl1);
|
||||||
|
for (j = 0, mm2 = 0; j < clip + ext - tl1; ++j)
|
||||||
|
if (qseq[j] != tseq[j] || qseq[j] > 3 || tseq[j] > 3)
|
||||||
|
++mm2;
|
||||||
|
for (mm1 = 0; j < clip + ext; ++j)
|
||||||
|
if (qseq[j] != tseq[j] || qseq[j] > 3 || tseq[j] > 3)
|
||||||
|
++mm1;
|
||||||
|
if (mm1 == 0 && mm2 <= 1) {
|
||||||
|
if (ai->flag & MM_JUNC_ANNO) {
|
||||||
|
if (i0_anno < 0) i0_anno = i, mm0_anno = mm1 + mm2;
|
||||||
|
++n_anno;
|
||||||
|
} else {
|
||||||
|
if (i0_misc < 0) i0_misc = i, mm0_misc = mm1 + mm2;
|
||||||
|
++n_misc;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (n_anno > 0) m = n_anno, i0 = i0_anno, mm0 = mm0_anno;
|
||||||
|
else m = n_misc, i0 = i0_misc, mm0 = mm0_misc;
|
||||||
|
kfree(km, tseq);
|
||||||
|
|
||||||
|
l = m > 0? r->re - a[i0].off : 0; // may be negative
|
||||||
|
if (m == 1 && clip + l >= opt->jump_min_match) { // add one more exon
|
||||||
|
mm_enlarge_cigar(r, 2);
|
||||||
|
r->p->cigar[r->p->n_cigar - 1] = ((r->p->cigar[r->p->n_cigar - 1]>>4) - l) << 4 | MM_CIGAR_MATCH;
|
||||||
|
r->p->cigar[r->p->n_cigar] = (a[i0].off2 - a[i0].off) << 4 | MM_CIGAR_N_SKIP;
|
||||||
|
r->p->cigar[r->p->n_cigar + 1] = (clip + l) << 4 | MM_CIGAR_MATCH;
|
||||||
|
r->p->n_cigar += 2;
|
||||||
|
r->re = a[i0].off2 + (clip + l);
|
||||||
|
if (!r->rev) r->qe = qlen;
|
||||||
|
else r->qs = 0;
|
||||||
|
r->blen += clip, r->mlen += clip - mm0;
|
||||||
|
r->p->dp_max0 += (clip - mm0) * opt->a - mm0 * opt->b;
|
||||||
|
r->p->dp_max += (clip - mm0) * opt->a - mm0 * opt->b;
|
||||||
|
if (!r->is_spliced) r->is_spliced = 1, r->p->dp_max += (opt->a + opt->b) + ((opt->a + opt->b) >> 1);
|
||||||
|
} else if (m > 0 && r->re > a[i0].off) { // trim by l; l is always positive
|
||||||
|
r->p->cigar[r->p->n_cigar - 1] -= l << 4 | MM_CIGAR_MATCH;
|
||||||
|
r->re -= l;
|
||||||
|
if (!r->rev) r->qe -= l;
|
||||||
|
else r->qs += l;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void mm_jump_split(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq, mm_reg1_t *r, int32_t ts_strand)
|
||||||
|
{
|
||||||
|
assert((opt->flag & MM_F_EQX) == 0);
|
||||||
|
mm_jump_split_left(km, mi, opt, qlen, qseq, r, ts_strand);
|
||||||
|
mm_jump_split_right(km, mi, opt, qlen, qseq, r, ts_strand);
|
||||||
|
}
|
||||||
@@ -40,7 +40,8 @@ void *km_init2(void *km_par, size_t min_core_size)
|
|||||||
kmem_t *km;
|
kmem_t *km;
|
||||||
km = (kmem_t*)kcalloc(km_par, 1, sizeof(kmem_t));
|
km = (kmem_t*)kcalloc(km_par, 1, sizeof(kmem_t));
|
||||||
km->par = km_par;
|
km->par = km_par;
|
||||||
km->min_core_size = min_core_size > 0? min_core_size : 0x80000;
|
if (km_par) km->min_core_size = min_core_size > 0? min_core_size : ((kmem_t*)km_par)->min_core_size - 2;
|
||||||
|
else km->min_core_size = min_core_size > 0? min_core_size : 0x80000;
|
||||||
return (void*)km;
|
return (void*)km;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -183,6 +184,16 @@ void *krealloc(void *_km, void *ap, size_t n_bytes) // TODO: this can be made mo
|
|||||||
return q;
|
return q;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void *krelocate(void *km, void *ap, size_t n_bytes)
|
||||||
|
{
|
||||||
|
void *p;
|
||||||
|
if (km == 0 || ap == 0) return ap;
|
||||||
|
p = kmalloc(km, n_bytes);
|
||||||
|
memcpy(p, ap, n_bytes);
|
||||||
|
kfree(km, ap);
|
||||||
|
return p;
|
||||||
|
}
|
||||||
|
|
||||||
void km_stat(const void *_km, km_stat_t *s)
|
void km_stat(const void *_km, km_stat_t *s)
|
||||||
{
|
{
|
||||||
kmem_t *km = (kmem_t*)_km;
|
kmem_t *km = (kmem_t*)_km;
|
||||||
@@ -203,3 +214,11 @@ void km_stat(const void *_km, km_stat_t *s)
|
|||||||
s->largest = s->largest > size? s->largest : size;
|
s->largest = s->largest > size? s->largest : size;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void km_stat_print(const void *km)
|
||||||
|
{
|
||||||
|
km_stat_t st;
|
||||||
|
km_stat(km, &st);
|
||||||
|
fprintf(stderr, "[km_stat] cap=%ld, avail=%ld, largest=%ld, n_core=%ld, n_block=%ld\n",
|
||||||
|
st.capacity, st.available, st.largest, st.n_blocks, st.n_cores);
|
||||||
|
}
|
||||||
|
|||||||
@@ -13,6 +13,7 @@ typedef struct {
|
|||||||
|
|
||||||
void *kmalloc(void *km, size_t size);
|
void *kmalloc(void *km, size_t size);
|
||||||
void *krealloc(void *km, void *ptr, size_t size);
|
void *krealloc(void *km, void *ptr, size_t size);
|
||||||
|
void *krelocate(void *km, void *ap, size_t n_bytes);
|
||||||
void *kcalloc(void *km, size_t count, size_t size);
|
void *kcalloc(void *km, size_t count, size_t size);
|
||||||
void kfree(void *km, void *ptr);
|
void kfree(void *km, void *ptr);
|
||||||
|
|
||||||
@@ -20,11 +21,29 @@ void *km_init(void);
|
|||||||
void *km_init2(void *km_par, size_t min_core_size);
|
void *km_init2(void *km_par, size_t min_core_size);
|
||||||
void km_destroy(void *km);
|
void km_destroy(void *km);
|
||||||
void km_stat(const void *_km, km_stat_t *s);
|
void km_stat(const void *_km, km_stat_t *s);
|
||||||
|
void km_stat_print(const void *km);
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
|
#define Kmalloc(km, type, cnt) ((type*)kmalloc((km), (cnt) * sizeof(type)))
|
||||||
|
#define Kcalloc(km, type, cnt) ((type*)kcalloc((km), (cnt), sizeof(type)))
|
||||||
|
#define Krealloc(km, type, ptr, cnt) ((type*)krealloc((km), (ptr), (cnt) * sizeof(type)))
|
||||||
|
|
||||||
|
#define Kgrow(km, type, ptr, __i, __m) do { \
|
||||||
|
if ((__i) >= (__m)) { \
|
||||||
|
(__m) = (__i) + 1; \
|
||||||
|
(__m) += ((__m)>>1) + 16; \
|
||||||
|
(ptr) = Krealloc(km, type, ptr, (__m)); \
|
||||||
|
} \
|
||||||
|
} while (0)
|
||||||
|
|
||||||
|
#define Kexpand(km, type, a, m) do { \
|
||||||
|
(m) = (m) >= 4? (m) + ((m)>>1) : 16; \
|
||||||
|
(a) = Krealloc(km, type, (a), (m)); \
|
||||||
|
} while (0)
|
||||||
|
|
||||||
#define KMALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kmalloc((km), (len) * sizeof(*(ptr))))
|
#define KMALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kmalloc((km), (len) * sizeof(*(ptr))))
|
||||||
#define KCALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kcalloc((km), (len), sizeof(*(ptr))))
|
#define KCALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kcalloc((km), (len), sizeof(*(ptr))))
|
||||||
#define KREALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))krealloc((km), (ptr), (len) * sizeof(*(ptr))))
|
#define KREALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))krealloc((km), (ptr), (len) * sizeof(*(ptr))))
|
||||||
@@ -50,7 +69,7 @@ void km_stat(const void *_km, km_stat_t *s);
|
|||||||
} kmp_##name##_t; \
|
} kmp_##name##_t; \
|
||||||
SCOPE kmp_##name##_t *kmp_init_##name(void *km) { \
|
SCOPE kmp_##name##_t *kmp_init_##name(void *km) { \
|
||||||
kmp_##name##_t *mp; \
|
kmp_##name##_t *mp; \
|
||||||
KCALLOC(km, mp, 1); \
|
mp = Kcalloc(km, kmp_##name##_t, 1); \
|
||||||
mp->km = km; \
|
mp->km = km; \
|
||||||
return mp; \
|
return mp; \
|
||||||
} \
|
} \
|
||||||
@@ -66,7 +85,7 @@ void km_stat(const void *_km, km_stat_t *s);
|
|||||||
} \
|
} \
|
||||||
SCOPE void kmp_free_##name(kmp_##name##_t *mp, kmptype_t *p) { \
|
SCOPE void kmp_free_##name(kmp_##name##_t *mp, kmptype_t *p) { \
|
||||||
--mp->cnt; \
|
--mp->cnt; \
|
||||||
if (mp->n == mp->max) KEXPAND(mp->km, mp->buf, mp->max); \
|
if (mp->n == mp->max) Kexpand(mp->km, kmptype_t*, mp->buf, mp->max); \
|
||||||
mp->buf[mp->n++] = p; \
|
mp->buf[mp->n++] = p; \
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -15,6 +15,8 @@
|
|||||||
#define KSW_EZ_SPLICE_FOR 0x100
|
#define KSW_EZ_SPLICE_FOR 0x100
|
||||||
#define KSW_EZ_SPLICE_REV 0x200
|
#define KSW_EZ_SPLICE_REV 0x200
|
||||||
#define KSW_EZ_SPLICE_FLANK 0x400
|
#define KSW_EZ_SPLICE_FLANK 0x400
|
||||||
|
#define KSW_EZ_SPLICE_CMPLX 0x800 // use the miniprot splice model
|
||||||
|
#define KSW_EZ_SPLICE_SCORE 0x1000 // use splice score
|
||||||
|
|
||||||
// The subset of CIGAR operators used by ksw code.
|
// The subset of CIGAR operators used by ksw code.
|
||||||
// Use MM_CIGAR_* from minimap.h if you need the full list.
|
// Use MM_CIGAR_* from minimap.h if you need the full list.
|
||||||
@@ -23,6 +25,8 @@
|
|||||||
#define KSW_CIGAR_DEL 2
|
#define KSW_CIGAR_DEL 2
|
||||||
#define KSW_CIGAR_N_SKIP 3
|
#define KSW_CIGAR_N_SKIP 3
|
||||||
|
|
||||||
|
#define KSW_SPSC_OFFSET 64
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
extern "C" {
|
extern "C" {
|
||||||
#endif
|
#endif
|
||||||
@@ -68,7 +72,7 @@ void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t gape2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
int8_t gapo, int8_t gape, int8_t gapo2, int8_t gape2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||||
|
|
||||||
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
int8_t gapo, int8_t gape, int8_t gapo2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
||||||
|
|
||||||
void ksw_extf2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t mch, int8_t mis, int8_t e, int w, int xdrop, ksw_extz_t *ez);
|
void ksw_extf2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t mch, int8_t mis, int8_t e, int w, int xdrop, ksw_extz_t *ez);
|
||||||
|
|
||||||
|
|||||||
+5
-5
@@ -80,17 +80,17 @@ void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
}
|
}
|
||||||
|
|
||||||
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||||
{
|
{
|
||||||
extern void ksw_exts2_sse2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
extern void ksw_exts2_sse2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
||||||
extern void ksw_exts2_sse41(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
extern void ksw_exts2_sse41(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
||||||
if (ksw_simd < 0) ksw_simd = x86_simd();
|
if (ksw_simd < 0) ksw_simd = x86_simd();
|
||||||
if (ksw_simd & SIMD_SSE4_1)
|
if (ksw_simd & SIMD_SSE4_1)
|
||||||
ksw_exts2_sse41(km, qlen, query, tlen, target, m, mat, q, e, q2, noncan, zdrop, junc_bonus, flag, junc, ez);
|
ksw_exts2_sse41(km, qlen, query, tlen, target, m, mat, q, e, q2, noncan, zdrop, end_bonus, junc_bonus, junc_pen, flag, junc, ez);
|
||||||
else if (ksw_simd & SIMD_SSE2)
|
else if (ksw_simd & SIMD_SSE2)
|
||||||
ksw_exts2_sse2(km, qlen, query, tlen, target, m, mat, q, e, q2, noncan, zdrop, junc_bonus, flag, junc, ez);
|
ksw_exts2_sse2(km, qlen, query, tlen, target, m, mat, q, e, q2, noncan, zdrop, end_bonus, junc_bonus, junc_pen, flag, junc, ez);
|
||||||
else abort();
|
else abort();
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
+1
-1
@@ -358,7 +358,7 @@ void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||||
// update ez
|
// update ez
|
||||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||||
ez->mte = H[en0], ez->mte_q = r - en;
|
ez->mte = H[en0], ez->mte_q = r - en0;
|
||||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e2)) break;
|
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e2)) break;
|
||||||
|
|||||||
+95
-45
@@ -24,14 +24,14 @@
|
|||||||
#ifdef KSW_CPU_DISPATCH
|
#ifdef KSW_CPU_DISPATCH
|
||||||
#ifdef __SSE4_1__
|
#ifdef __SSE4_1__
|
||||||
void ksw_exts2_sse41(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_exts2_sse41(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||||
#else
|
#else
|
||||||
void ksw_exts2_sse2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_exts2_sse2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||||
#endif
|
#endif
|
||||||
#else
|
#else
|
||||||
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||||
#endif // ~KSW_CPU_DISPATCH
|
#endif // ~KSW_CPU_DISPATCH
|
||||||
{
|
{
|
||||||
#define __dp_code_block1 \
|
#define __dp_code_block1 \
|
||||||
@@ -71,6 +71,7 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
|
|
||||||
ksw_reset_extz(ez);
|
ksw_reset_extz(ez);
|
||||||
if (m <= 1 || qlen <= 0 || tlen <= 0 || q2 <= q + e) return;
|
if (m <= 1 || qlen <= 0 || tlen <= 0 || q2 <= q + e) return;
|
||||||
|
assert((flag & KSW_EZ_SPLICE_FOR) == 0 || (flag & KSW_EZ_SPLICE_REV) == 0); // can't be both set
|
||||||
|
|
||||||
zero_ = _mm_set1_epi8(0);
|
zero_ = _mm_set1_epi8(0);
|
||||||
q_ = _mm_set1_epi8(q);
|
q_ = _mm_set1_epi8(q);
|
||||||
@@ -118,55 +119,100 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
|
|
||||||
// set the donor and acceptor arrays. TODO: this assumes 0/1/2/3 encoding!
|
// set the donor and acceptor arrays. TODO: this assumes 0/1/2/3 encoding!
|
||||||
if (flag & (KSW_EZ_SPLICE_FOR|KSW_EZ_SPLICE_REV)) {
|
if (flag & (KSW_EZ_SPLICE_FOR|KSW_EZ_SPLICE_REV)) {
|
||||||
int semi_cost = flag&KSW_EZ_SPLICE_FLANK? -noncan/2 : 0; // GTr or yAG is worth 0.5 bit; see PMID:18688272
|
const int sp0[4] = { 8, 15, 21, 30 };
|
||||||
memset(donor, -noncan, tlen_ * 16);
|
int sp[4];
|
||||||
memset(acceptor, -noncan, tlen_ * 16);
|
if (flag & KSW_EZ_SPLICE_CMPLX) {
|
||||||
|
for (t = 0; t < 4; ++t)
|
||||||
|
sp[t] = (int)((double)sp0[t] / 3. + .499);
|
||||||
|
} else {
|
||||||
|
sp[0] = flag&KSW_EZ_SPLICE_FLANK? noncan / 2 : 0;
|
||||||
|
sp[1] = sp[2] = sp[3] = noncan;
|
||||||
|
}
|
||||||
|
memset(donor, -sp[3], tlen_ * 16);
|
||||||
|
memset(acceptor, -sp[3], tlen_ * 16);
|
||||||
if (!(flag & KSW_EZ_REV_CIGAR)) {
|
if (!(flag & KSW_EZ_REV_CIGAR)) {
|
||||||
for (t = 0; t < tlen - 4; ++t) {
|
for (t = 0; t < tlen - 4; ++t) {
|
||||||
int can_type = 0; // type of canonical site: 0=none, 1=GT/AG only, 2=GTr/yAG
|
int z = 3;
|
||||||
if ((flag & KSW_EZ_SPLICE_FOR) && target[t+1] == 2 && target[t+2] == 3) can_type = 1; // GTr...
|
if (flag & KSW_EZ_SPLICE_FOR) {
|
||||||
if ((flag & KSW_EZ_SPLICE_REV) && target[t+1] == 1 && target[t+2] == 3) can_type = 1; // CTr...
|
if (target[t+1] == 2 && target[t+2] == 3) // |GT.
|
||||||
if (can_type && (target[t+3] == 0 || target[t+3] == 2)) can_type = 2;
|
z = target[t+3] == 0 || target[t+3] == 2? -1 : 0; // |GTr or not
|
||||||
if (can_type) ((int8_t*)donor)[t] = can_type == 2? 0 : semi_cost;
|
else if (target[t+1] == 2 && target[t+2] == 1) z = 1; // |GC.
|
||||||
|
else if (target[t+1] == 0 && target[t+2] == 3) z = 2; // |AT.
|
||||||
|
} else if (flag & KSW_EZ_SPLICE_REV) {
|
||||||
|
if (target[t+1] == 1 && target[t+2] == 3) // |CT. (revcomp of .AG|)
|
||||||
|
z = target[t+3] == 0 || target[t+3] == 2? -1 : 0;
|
||||||
|
else if (target[t+1] == 2 && target[t+2] == 3) z = 2; // |GT. (revcomp of .AC|)
|
||||||
|
}
|
||||||
|
((int8_t*)donor)[t] = z < 0? 0 : -sp[z];
|
||||||
}
|
}
|
||||||
if (junc)
|
|
||||||
for (t = 0; t < tlen - 1; ++t)
|
|
||||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&8)))
|
|
||||||
((int8_t*)donor)[t] += junc_bonus;
|
|
||||||
for (t = 2; t < tlen; ++t) {
|
for (t = 2; t < tlen; ++t) {
|
||||||
int can_type = 0;
|
int z = 3;
|
||||||
if ((flag & KSW_EZ_SPLICE_FOR) && target[t-1] == 0 && target[t] == 2) can_type = 1; // ...yAG
|
if (flag & KSW_EZ_SPLICE_FOR) {
|
||||||
if ((flag & KSW_EZ_SPLICE_REV) && target[t-1] == 0 && target[t] == 1) can_type = 1; // ...yAC
|
if (target[t-1] == 0 && target[t] == 2) // .AG|
|
||||||
if (can_type && (target[t-2] == 1 || target[t-2] == 3)) can_type = 2;
|
z = target[t-2] == 1 || target[t-2] == 3? -1 : 0; // yAG| or not
|
||||||
if (can_type) ((int8_t*)acceptor)[t] = can_type == 2? 0 : semi_cost;
|
else if (target[t-1] == 0 && target[t] == 1) z = 2; // .AC|
|
||||||
|
} else if (flag & KSW_EZ_SPLICE_REV) {
|
||||||
|
if (target[t-1] == 0 && target[t] == 1) // .AC| (revcomp of |GT.)
|
||||||
|
z = target[t-2] == 1 || target[t-2] == 3? -1 : 0; // yAC| or not
|
||||||
|
else if (target[t-1] == 2 && target[t] == 1) z = 1; // .GC| (revcomp of |GC.)
|
||||||
|
else if (target[t-1] == 0 && target[t] == 3) z = 2; // .AT| (revcomp of |AT.)
|
||||||
|
}
|
||||||
|
((int8_t*)acceptor)[t] = z < 0? 0 : -sp[z];
|
||||||
}
|
}
|
||||||
if (junc)
|
|
||||||
for (t = 0; t < tlen; ++t)
|
|
||||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&4)))
|
|
||||||
((int8_t*)acceptor)[t] += junc_bonus;
|
|
||||||
} else {
|
} else {
|
||||||
for (t = 0; t < tlen - 4; ++t) {
|
for (t = 0; t < tlen - 4; ++t) {
|
||||||
int can_type = 0; // type of canonical site: 0=none, 1=GT/AG only, 2=GTr/yAG
|
int z = 3;
|
||||||
if ((flag & KSW_EZ_SPLICE_FOR) && target[t+1] == 2 && target[t+2] == 0) can_type = 1; // GAy...
|
if (flag & KSW_EZ_SPLICE_FOR) {
|
||||||
if ((flag & KSW_EZ_SPLICE_REV) && target[t+1] == 1 && target[t+2] == 0) can_type = 1; // CAy...
|
if (target[t+1] == 2 && target[t+2] == 0) // |GA. (rev of .AG|)
|
||||||
if (can_type && (target[t+3] == 1 || target[t+3] == 3)) can_type = 2;
|
z = target[t+3] == 1 || target[t+3] == 3? -1 : 0;
|
||||||
if (can_type) ((int8_t*)donor)[t] = can_type == 2? 0 : semi_cost;
|
else if (target[t+1] == 1 && target[t+2] == 0) z = 2; // |CA. (rev of .AC|)
|
||||||
|
} else if (flag & KSW_EZ_SPLICE_REV) {
|
||||||
|
if (target[t+1] == 1 && target[t+2] == 0) // |CA. (comp of |GT.)
|
||||||
|
z = target[t+3] == 1 || target[t+3] == 3? -1 : 0;
|
||||||
|
else if (target[t+1] == 1 && target[t+2] == 2) z = 1; // |CG. (comp of |GC.)
|
||||||
|
else if (target[t+1] == 3 && target[t+2] == 0) z = 2; // |TA. (comp of |AT.)
|
||||||
|
}
|
||||||
|
((int8_t*)donor)[t] = z < 0? 0 : -sp[z];
|
||||||
}
|
}
|
||||||
if (junc)
|
|
||||||
for (t = 0; t < tlen - 1; ++t)
|
|
||||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&4)))
|
|
||||||
((int8_t*)donor)[t] += junc_bonus;
|
|
||||||
for (t = 2; t < tlen; ++t) {
|
for (t = 2; t < tlen; ++t) {
|
||||||
int can_type = 0;
|
int z = 3;
|
||||||
if ((flag & KSW_EZ_SPLICE_FOR) && target[t-1] == 3 && target[t] == 2) can_type = 1; // ...rTG
|
if (flag & KSW_EZ_SPLICE_FOR) {
|
||||||
if ((flag & KSW_EZ_SPLICE_REV) && target[t-1] == 3 && target[t] == 1) can_type = 1; // ...rTC
|
if (target[t-1] == 3 && target[t] == 2) // .TG| (rev of |GT.)
|
||||||
if (can_type && (target[t-2] == 0 || target[t-2] == 2)) can_type = 2;
|
z = target[t-2] == 0 || target[t-2] == 2? -1 : 0;
|
||||||
if (can_type) ((int8_t*)acceptor)[t] = can_type == 2? 0 : semi_cost;
|
else if (target[t-1] == 1 && target[t] == 2) z = 1; // .CG| (rev of |GC.)
|
||||||
|
else if (target[t-1] == 3 && target[t] == 0) z = 2; // .TA| (rev of |AT.)
|
||||||
|
} else if (flag & KSW_EZ_SPLICE_REV) {
|
||||||
|
if (target[t-1] == 3 && target[t] == 1) // .TC| (comp of .AG|)
|
||||||
|
z = target[t-2] == 0 || target[t-2] == 2? -1 : 0;
|
||||||
|
else if (target[t-1] == 3 && target[t] == 2) z = 2; // .TG| (comp of .AC|)
|
||||||
|
}
|
||||||
|
((int8_t*)acceptor)[t] = z < 0? 0 : -sp[z];
|
||||||
}
|
}
|
||||||
if (junc)
|
}
|
||||||
for (t = 0; t < tlen; ++t)
|
}
|
||||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&8)))
|
|
||||||
((int8_t*)acceptor)[t] += junc_bonus;
|
if (junc && (flag & KSW_EZ_SPLICE_SCORE)) { // junc[] keeps the donor score
|
||||||
|
uint8_t donor_val = !!(flag & KSW_EZ_SPLICE_FOR) == !(flag & KSW_EZ_REV_CIGAR)? 0 : 1;
|
||||||
|
for (t = 0; t < tlen - 1; ++t)
|
||||||
|
((int8_t*)donor)[t] += junc[t+1] == 0xff || (junc[t+1]&1) != donor_val? -junc_pen : (int8_t)(junc[t+1]>>1) - (int8_t)KSW_SPSC_OFFSET;
|
||||||
|
for (t = 0; t < tlen - 1; ++t)
|
||||||
|
((int8_t*)acceptor)[t] += junc[t+1] == 0xff || (junc[t+1]&1) != !donor_val? -junc_pen : (int8_t)(junc[t+1]>>1) - (int8_t)KSW_SPSC_OFFSET;
|
||||||
|
//for (t = 0; t < tlen - 1; ++t) if (junc[t+1] != 0xff) fprintf(stderr, "Y2\t%d\t%d\t%c\t%d\n", ((int8_t*)donor)[t], ((int8_t*)acceptor)[t], "DA"[junc[t+1]&1], (int8_t)(junc[t+1]>>1) - (int8_t)KSW_SPSC_OFFSET);
|
||||||
|
} else if (junc) { // junc[] keeps the splice sites
|
||||||
|
if (!(flag & KSW_EZ_REV_CIGAR)) {
|
||||||
|
for (t = 0; t < tlen - 1; ++t)
|
||||||
|
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&8)))
|
||||||
|
((int8_t*)donor)[t] += junc_bonus;
|
||||||
|
for (t = 0; t < tlen; ++t)
|
||||||
|
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&4)))
|
||||||
|
((int8_t*)acceptor)[t] += junc_bonus;
|
||||||
|
} else {
|
||||||
|
for (t = 0; t < tlen - 1; ++t)
|
||||||
|
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&4)))
|
||||||
|
((int8_t*)donor)[t] += junc_bonus;
|
||||||
|
for (t = 0; t < tlen; ++t)
|
||||||
|
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&8)))
|
||||||
|
((int8_t*)acceptor)[t] += junc_bonus;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -376,7 +422,7 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||||
// update ez
|
// update ez
|
||||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||||
ez->mte = H[en0], ez->mte_q = r - en;
|
ez->mte = H[en0], ez->mte_q = r - en0;
|
||||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, 0)) break;
|
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, 0)) break;
|
||||||
@@ -406,10 +452,14 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
if (!approx_max) kfree(km, H);
|
if (!approx_max) kfree(km, H);
|
||||||
if (with_cigar) { // backtrack
|
if (with_cigar) { // backtrack
|
||||||
int rev_cigar = !!(flag & KSW_EZ_REV_CIGAR);
|
int rev_cigar = !!(flag & KSW_EZ_REV_CIGAR);
|
||||||
if (!ez->zdropped && !(flag&KSW_EZ_EXTZ_ONLY))
|
if (!ez->zdropped && !(flag&KSW_EZ_EXTZ_ONLY)) {
|
||||||
ksw_backtrack(km, 1, rev_cigar, long_thres, (uint8_t*)p, off, off_end, n_col_*16, tlen-1, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
ksw_backtrack(km, 1, rev_cigar, long_thres, (uint8_t*)p, off, off_end, n_col_*16, tlen-1, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||||
else if (ez->max_t >= 0 && ez->max_q >= 0)
|
} else if (!ez->zdropped && (flag&KSW_EZ_EXTZ_ONLY) && ez->mqe + end_bonus > (int)ez->max) {
|
||||||
|
ez->reach_end = 1;
|
||||||
|
ksw_backtrack(km, 1, rev_cigar, long_thres, (uint8_t*)p, off, off_end, n_col_*16, ez->mqe_t, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||||
|
} else if (ez->max_t >= 0 && ez->max_q >= 0) {
|
||||||
ksw_backtrack(km, 1, rev_cigar, long_thres, (uint8_t*)p, off, off_end, n_col_*16, ez->max_t, ez->max_q, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
ksw_backtrack(km, 1, rev_cigar, long_thres, (uint8_t*)p, off, off_end, n_col_*16, ez->max_t, ez->max_q, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||||
|
}
|
||||||
kfree(km, mem2); kfree(km, off);
|
kfree(km, mem2); kfree(km, off);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+1
-1
@@ -269,7 +269,7 @@ void ksw_extz2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
|||||||
} else H[0] = v8[0] - qe - qe, max_H = H[0], max_t = 0; // special casing r==0
|
} else H[0] = v8[0] - qe - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||||
// update ez
|
// update ez
|
||||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||||
ez->mte = H[en0], ez->mte_q = r - en;
|
ez->mte = H[en0], ez->mte_q = r - en0;
|
||||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e)) break;
|
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e)) break;
|
||||||
|
|||||||
+2
-2
@@ -67,7 +67,7 @@ void *ksw_ll_qinit(void *km, int size, int qlen, const uint8_t *query, int m, co
|
|||||||
const int8_t *ma = mat + a * m;
|
const int8_t *ma = mat + a * m;
|
||||||
for (i = 0; i < slen; ++i)
|
for (i = 0; i < slen; ++i)
|
||||||
for (k = i; k < nlen; k += slen) // p iterations
|
for (k = i; k < nlen; k += slen) // p iterations
|
||||||
*t++ = (k >= qlen? 0 : ma[query[k]]) + q->shift;
|
*t++ = (k >= qlen? -1 : ma[query[k]]) + q->shift;
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
int16_t *t = (int16_t*)q->qp;
|
int16_t *t = (int16_t*)q->qp;
|
||||||
@@ -76,7 +76,7 @@ void *ksw_ll_qinit(void *km, int size, int qlen, const uint8_t *query, int m, co
|
|||||||
const int8_t *ma = mat + a * m;
|
const int8_t *ma = mat + a * m;
|
||||||
for (i = 0; i < slen; ++i)
|
for (i = 0; i < slen; ++i)
|
||||||
for (k = i; k < nlen; k += slen) // p iterations
|
for (k = i; k < nlen; k += slen) // p iterations
|
||||||
*t++ = (k >= qlen? 0 : ma[query[k]]);
|
*t++ = (k >= qlen? -1 : ma[query[k]]);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return q;
|
return q;
|
||||||
|
|||||||
@@ -6,7 +6,25 @@
|
|||||||
#include "kalloc.h"
|
#include "kalloc.h"
|
||||||
#include "krmq.h"
|
#include "krmq.h"
|
||||||
|
|
||||||
uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_t *p, int32_t *v, int32_t *t, int32_t min_cnt, int32_t min_sc, int32_t *n_u_, int32_t *n_v_)
|
static int64_t mg_chain_bk_end(int32_t max_drop, const mm128_t *z, const int32_t *f, const int64_t *p, int32_t *t, int64_t k)
|
||||||
|
{
|
||||||
|
int64_t i = z[k].y, end_i = -1, max_i = i;
|
||||||
|
int32_t max_s = 0;
|
||||||
|
if (i < 0 || t[i] != 0) return i;
|
||||||
|
do {
|
||||||
|
int32_t s;
|
||||||
|
t[i] = 2;
|
||||||
|
end_i = i = p[i];
|
||||||
|
s = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||||
|
if (s > max_s) max_s = s, max_i = i;
|
||||||
|
else if (max_s - s > max_drop) break;
|
||||||
|
} while (i >= 0 && t[i] == 0);
|
||||||
|
for (i = z[k].y; i >= 0 && i != end_i; i = p[i]) // reset modified t[]
|
||||||
|
t[i] = 0;
|
||||||
|
return max_i;
|
||||||
|
}
|
||||||
|
|
||||||
|
uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_t *p, int32_t *v, int32_t *t, int32_t min_cnt, int32_t min_sc, int32_t max_drop, int32_t *n_u_, int32_t *n_v_)
|
||||||
{
|
{
|
||||||
mm128_t *z;
|
mm128_t *z;
|
||||||
uint64_t *u;
|
uint64_t *u;
|
||||||
@@ -17,33 +35,39 @@ uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_
|
|||||||
for (i = 0, n_z = 0; i < n; ++i) // precompute n_z
|
for (i = 0, n_z = 0; i < n; ++i) // precompute n_z
|
||||||
if (f[i] >= min_sc) ++n_z;
|
if (f[i] >= min_sc) ++n_z;
|
||||||
if (n_z == 0) return 0;
|
if (n_z == 0) return 0;
|
||||||
KMALLOC(km, z, n_z);
|
z = Kmalloc(km, mm128_t, n_z);
|
||||||
for (i = 0, k = 0; i < n; ++i) // populate z[]
|
for (i = 0, k = 0; i < n; ++i) // populate z[]
|
||||||
if (f[i] >= min_sc) z[k].x = f[i], z[k++].y = i;
|
if (f[i] >= min_sc) z[k].x = f[i], z[k++].y = i;
|
||||||
radix_sort_128x(z, z + n_z);
|
radix_sort_128x(z, z + n_z);
|
||||||
|
|
||||||
memset(t, 0, n * 4);
|
memset(t, 0, n * 4);
|
||||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // precompute n_u
|
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // precompute n_u
|
||||||
int64_t n_v0 = n_v;
|
if (t[z[k].y] == 0) {
|
||||||
int32_t sc;
|
int64_t n_v0 = n_v, end_i;
|
||||||
for (i = z[k].y; i >= 0 && t[i] == 0; i = p[i])
|
int32_t sc;
|
||||||
++n_v, t[i] = 1;
|
end_i = mg_chain_bk_end(max_drop, z, f, p, t, k);
|
||||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
for (i = z[k].y; i != end_i; i = p[i])
|
||||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
++n_v, t[i] = 1;
|
||||||
++n_u;
|
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||||
else n_v = n_v0;
|
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||||
|
++n_u;
|
||||||
|
else n_v = n_v0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
KMALLOC(km, u, n_u);
|
u = Kmalloc(km, uint64_t, n_u);
|
||||||
memset(t, 0, n * 4);
|
memset(t, 0, n * 4);
|
||||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // populate u[]
|
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // populate u[]
|
||||||
int64_t n_v0 = n_v;
|
if (t[z[k].y] == 0) {
|
||||||
int32_t sc;
|
int64_t n_v0 = n_v, end_i;
|
||||||
for (i = z[k].y; i >= 0 && t[i] == 0; i = p[i])
|
int32_t sc;
|
||||||
v[n_v++] = i, t[i] = 1;
|
end_i = mg_chain_bk_end(max_drop, z, f, p, t, k);
|
||||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
for (i = z[k].y; i != end_i; i = p[i])
|
||||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
v[n_v++] = i, t[i] = 1;
|
||||||
u[n_u++] = (uint64_t)sc << 32 | (n_v - n_v0);
|
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||||
else n_v = n_v0;
|
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||||
|
u[n_u++] = (uint64_t)sc << 32 | (n_v - n_v0);
|
||||||
|
else n_v = n_v0;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
kfree(km, z);
|
kfree(km, z);
|
||||||
assert(n_v < INT32_MAX);
|
assert(n_v < INT32_MAX);
|
||||||
@@ -58,7 +82,7 @@ static mm128_t *compact_a(void *km, int32_t n_u, uint64_t *u, int32_t n_v, int32
|
|||||||
int64_t i, j, k;
|
int64_t i, j, k;
|
||||||
|
|
||||||
// write the result to b[]
|
// write the result to b[]
|
||||||
KMALLOC(km, b, n_v);
|
b = Kmalloc(km, mm128_t, n_v);
|
||||||
for (i = 0, k = 0; i < n_u; ++i) {
|
for (i = 0, k = 0; i < n_u; ++i) {
|
||||||
int32_t k0 = k, ni = (int32_t)u[i];
|
int32_t k0 = k, ni = (int32_t)u[i];
|
||||||
for (j = 0; j < ni; ++j)
|
for (j = 0; j < ni; ++j)
|
||||||
@@ -67,13 +91,13 @@ static mm128_t *compact_a(void *km, int32_t n_u, uint64_t *u, int32_t n_v, int32
|
|||||||
kfree(km, v);
|
kfree(km, v);
|
||||||
|
|
||||||
// sort u[] and a[] by the target position, such that adjacent chains may be joined
|
// sort u[] and a[] by the target position, such that adjacent chains may be joined
|
||||||
KMALLOC(km, w, n_u);
|
w = Kmalloc(km, mm128_t, n_u);
|
||||||
for (i = k = 0; i < n_u; ++i) {
|
for (i = k = 0; i < n_u; ++i) {
|
||||||
w[i].x = b[k].x, w[i].y = (uint64_t)k<<32|i;
|
w[i].x = b[k].x, w[i].y = (uint64_t)k<<32|i;
|
||||||
k += (int32_t)u[i];
|
k += (int32_t)u[i];
|
||||||
}
|
}
|
||||||
radix_sort_128x(w, w + n_u);
|
radix_sort_128x(w, w + n_u);
|
||||||
KMALLOC(km, u2, n_u);
|
u2 = Kmalloc(km, uint64_t, n_u);
|
||||||
for (i = k = 0; i < n_u; ++i) {
|
for (i = k = 0; i < n_u; ++i) {
|
||||||
int32_t j = (int32_t)w[i].y, n = (int32_t)u[j];
|
int32_t j = (int32_t)w[i].y, n = (int32_t)u[j];
|
||||||
u2[i] = u[j];
|
u2[i] = u[j];
|
||||||
@@ -114,7 +138,7 @@ static inline int32_t comput_sc(const mm128_t *ai, const mm128_t *aj, int32_t ma
|
|||||||
}
|
}
|
||||||
|
|
||||||
/* Input:
|
/* Input:
|
||||||
* a[].x: tid<<33 | rev<<32 | tpos
|
* a[].x: rev<<63 | tid<<32 | tpos
|
||||||
* a[].y: flags<<40 | q_span<<32 | q_pos
|
* a[].y: flags<<40 | q_span<<32 | q_pos
|
||||||
* Output:
|
* Output:
|
||||||
* n_u: #chains
|
* n_u: #chains
|
||||||
@@ -124,8 +148,8 @@ static inline int32_t comput_sc(const mm128_t *ai, const mm128_t *aj, int32_t ma
|
|||||||
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||||
int is_cdna, int n_seg, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
int is_cdna, int n_seg, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||||
{ // TODO: make sure this works when n has more than 32 bits
|
{ // TODO: make sure this works when n has more than 32 bits
|
||||||
int32_t *f, *t, *v, n_u, n_v, mmax_f = 0;
|
int32_t *f, *t, *v, n_u, n_v, mmax_f = 0, max_drop = bw;
|
||||||
int64_t *p, i, j, max_ii, st = 0, n_iter = 0;
|
int64_t *p, i, j, max_ii, st = 0;
|
||||||
uint64_t *u;
|
uint64_t *u;
|
||||||
|
|
||||||
if (_u) *_u = 0, *n_u_ = 0;
|
if (_u) *_u = 0, *n_u_ = 0;
|
||||||
@@ -135,10 +159,11 @@ mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int
|
|||||||
}
|
}
|
||||||
if (max_dist_x < bw) max_dist_x = bw;
|
if (max_dist_x < bw) max_dist_x = bw;
|
||||||
if (max_dist_y < bw && !is_cdna) max_dist_y = bw;
|
if (max_dist_y < bw && !is_cdna) max_dist_y = bw;
|
||||||
KMALLOC(km, p, n);
|
if (is_cdna) max_drop = INT32_MAX;
|
||||||
KMALLOC(km, f, n);
|
p = Kmalloc(km, int64_t, n);
|
||||||
KMALLOC(km, v, n);
|
f = Kmalloc(km, int32_t, n);
|
||||||
KCALLOC(km, t, n);
|
v = Kmalloc(km, int32_t, n);
|
||||||
|
t = Kcalloc(km, int32_t, n);
|
||||||
|
|
||||||
// fill the score and backtrack arrays
|
// fill the score and backtrack arrays
|
||||||
for (i = 0, max_ii = -1; i < n; ++i) {
|
for (i = 0, max_ii = -1; i < n; ++i) {
|
||||||
@@ -149,7 +174,6 @@ mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int
|
|||||||
for (j = i - 1; j >= st; --j) {
|
for (j = i - 1; j >= st; --j) {
|
||||||
int32_t sc;
|
int32_t sc;
|
||||||
sc = comput_sc(&a[i], &a[j], max_dist_x, max_dist_y, bw, chn_pen_gap, chn_pen_skip, is_cdna, n_seg);
|
sc = comput_sc(&a[i], &a[j], max_dist_x, max_dist_y, bw, chn_pen_gap, chn_pen_skip, is_cdna, n_seg);
|
||||||
++n_iter;
|
|
||||||
if (sc == INT32_MIN) continue;
|
if (sc == INT32_MIN) continue;
|
||||||
sc += f[j];
|
sc += f[j];
|
||||||
if (sc > max_f) {
|
if (sc > max_f) {
|
||||||
@@ -179,9 +203,10 @@ mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int
|
|||||||
if (max_ii < 0 || (a[i].x - a[max_ii].x <= (int64_t)max_dist_x && f[max_ii] < f[i]))
|
if (max_ii < 0 || (a[i].x - a[max_ii].x <= (int64_t)max_dist_x && f[max_ii] < f[i]))
|
||||||
max_ii = i;
|
max_ii = i;
|
||||||
if (mmax_f < max_f) mmax_f = max_f;
|
if (mmax_f < max_f) mmax_f = max_f;
|
||||||
|
//fprintf(stderr, "X1\t%ld\t%ld:%d\t%ld\t%ld:%d\t%ld\t%ld\n", (long)i, (long)(a[i].x>>32), (int32_t)a[i].x, (long)max_j, max_j<0?-1L:(long)(a[max_j].x>>32), max_j<0?-1:(int32_t)a[max_j].x, (long)max_f, (long)v[i]);
|
||||||
}
|
}
|
||||||
|
|
||||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, &n_u, &n_v);
|
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, max_drop, &n_u, &n_v);
|
||||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||||
kfree(km, p); kfree(km, f); kfree(km, t);
|
kfree(km, p); kfree(km, f); kfree(km, t);
|
||||||
if (n_u == 0) {
|
if (n_u == 0) {
|
||||||
@@ -225,8 +250,8 @@ static inline int32_t comput_sc_simple(const mm128_t *ai, const mm128_t *aj, flo
|
|||||||
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||||
int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||||
{
|
{
|
||||||
int32_t *f,*t, *v, n_u, n_v, mmax_f = 0, max_rmq_size = 0;
|
int32_t *f,*t, *v, n_u, n_v, mmax_f = 0, max_rmq_size = 0, max_drop = bw;
|
||||||
int64_t *p, i, i0, st = 0, st_inner = 0, n_iter = 0;
|
int64_t *p, i, i0, st = 0, st_inner = 0;
|
||||||
uint64_t *u;
|
uint64_t *u;
|
||||||
lc_elem_t *root = 0, *root_inner = 0;
|
lc_elem_t *root = 0, *root_inner = 0;
|
||||||
void *mem_mp = 0;
|
void *mem_mp = 0;
|
||||||
@@ -238,11 +263,12 @@ mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_ski
|
|||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
if (max_dist < bw) max_dist = bw;
|
if (max_dist < bw) max_dist = bw;
|
||||||
if (max_dist_inner <= 0 || max_dist_inner >= max_dist) max_dist_inner = 0;
|
if (max_dist_inner < 0) max_dist_inner = 0;
|
||||||
KMALLOC(km, p, n);
|
if (max_dist_inner > max_dist) max_dist_inner = max_dist;
|
||||||
KMALLOC(km, f, n);
|
p = Kmalloc(km, int64_t, n);
|
||||||
KCALLOC(km, t, n);
|
f = Kmalloc(km, int32_t, n);
|
||||||
KMALLOC(km, v, n);
|
t = Kcalloc(km, int32_t, n);
|
||||||
|
v = Kmalloc(km, int32_t, n);
|
||||||
mem_mp = km_init2(km, 0x10000);
|
mem_mp = km_init2(km, 0x10000);
|
||||||
mp = kmp_init_rmq(mem_mp);
|
mp = kmp_init_rmq(mem_mp);
|
||||||
|
|
||||||
@@ -300,12 +326,11 @@ mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_ski
|
|||||||
krmq_interval(lc_elem, root_inner, &s, &lo, &hi);
|
krmq_interval(lc_elem, root_inner, &s, &lo, &hi);
|
||||||
if (lo) {
|
if (lo) {
|
||||||
const lc_elem_t *q;
|
const lc_elem_t *q;
|
||||||
int32_t width, n_rmq_iter = 0;
|
int32_t width;
|
||||||
krmq_itr_t(lc_elem) itr;
|
krmq_itr_t(lc_elem) itr;
|
||||||
krmq_itr_find(lc_elem, root_inner, lo, &itr);
|
krmq_itr_find(lc_elem, root_inner, lo, &itr);
|
||||||
while ((q = krmq_at(&itr)) != 0) {
|
while ((q = krmq_at(&itr)) != 0) {
|
||||||
if (q->y < (int32_t)a[i].y - max_dist_inner) break;
|
if (q->y < (int32_t)a[i].y - max_dist_inner) break;
|
||||||
++n_rmq_iter;
|
|
||||||
j = q->i;
|
j = q->i;
|
||||||
sc = f[j] + comput_sc_simple(&a[i], &a[j], chn_pen_gap, chn_pen_skip, 0, &width);
|
sc = f[j] + comput_sc_simple(&a[i], &a[j], chn_pen_gap, chn_pen_skip, 0, &width);
|
||||||
if (width <= bw) {
|
if (width <= bw) {
|
||||||
@@ -320,7 +345,6 @@ mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_ski
|
|||||||
}
|
}
|
||||||
if (!krmq_itr_prev(lc_elem, &itr)) break;
|
if (!krmq_itr_prev(lc_elem, &itr)) break;
|
||||||
}
|
}
|
||||||
n_iter += n_rmq_iter;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -333,7 +357,7 @@ mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_ski
|
|||||||
}
|
}
|
||||||
km_destroy(mem_mp);
|
km_destroy(mem_mp);
|
||||||
|
|
||||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, &n_u, &n_v);
|
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, max_drop, &n_u, &n_v);
|
||||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||||
kfree(km, p); kfree(km, f); kfree(km, t);
|
kfree(km, p); kfree(km, f); kfree(km, t);
|
||||||
if (n_u == 0) {
|
if (n_u == 0) {
|
||||||
|
|||||||
-1
Submodule lib/simde deleted from b30129b3b4
@@ -7,8 +7,6 @@
|
|||||||
#include "mmpriv.h"
|
#include "mmpriv.h"
|
||||||
#include "ketopt.h"
|
#include "ketopt.h"
|
||||||
|
|
||||||
#define MM_VERSION "2.22-r1101"
|
|
||||||
|
|
||||||
#ifdef __linux__
|
#ifdef __linux__
|
||||||
#include <sys/resource.h>
|
#include <sys/resource.h>
|
||||||
#include <sys/time.h>
|
#include <sys/time.h>
|
||||||
@@ -37,12 +35,12 @@ static ko_longopt_t long_options[] = {
|
|||||||
{ "splice", ko_no_argument, 310 },
|
{ "splice", ko_no_argument, 310 },
|
||||||
{ "cost-non-gt-ag", ko_required_argument, 'C' },
|
{ "cost-non-gt-ag", ko_required_argument, 'C' },
|
||||||
{ "no-long-join", ko_no_argument, 312 },
|
{ "no-long-join", ko_no_argument, 312 },
|
||||||
{ "sr", ko_no_argument, 313 },
|
{ "sr", ko_optional_argument, 313 },
|
||||||
{ "frag", ko_required_argument, 314 },
|
{ "frag", ko_required_argument, 314 },
|
||||||
{ "secondary", ko_required_argument, 315 },
|
{ "secondary", ko_required_argument, 315 },
|
||||||
{ "cs", ko_optional_argument, 316 },
|
{ "cs", ko_optional_argument, 316 },
|
||||||
{ "end-bonus", ko_required_argument, 317 },
|
{ "end-bonus", ko_required_argument, 317 },
|
||||||
{ "no-pairing", ko_no_argument, 318 },
|
{ "no-pairing", ko_no_argument, 318 }, // deprecated but reserved for backward compatibility
|
||||||
{ "splice-flank", ko_required_argument, 319 },
|
{ "splice-flank", ko_required_argument, 319 },
|
||||||
{ "idx-no-seq", ko_no_argument, 320 },
|
{ "idx-no-seq", ko_no_argument, 320 },
|
||||||
{ "end-seed-pen", ko_required_argument, 321 },
|
{ "end-seed-pen", ko_required_argument, 321 },
|
||||||
@@ -74,6 +72,22 @@ static ko_longopt_t long_options[] = {
|
|||||||
{ "rmq", ko_optional_argument, 347 },
|
{ "rmq", ko_optional_argument, 347 },
|
||||||
{ "qstrand", ko_no_argument, 348 },
|
{ "qstrand", ko_no_argument, 348 },
|
||||||
{ "cap-kalloc", ko_required_argument, 349 },
|
{ "cap-kalloc", ko_required_argument, 349 },
|
||||||
|
{ "q-occ-frac", ko_required_argument, 350 },
|
||||||
|
{ "chain-skip-scale",ko_required_argument,351 },
|
||||||
|
{ "print-chains", ko_no_argument, 352 },
|
||||||
|
{ "no-hash-name", ko_no_argument, 353 },
|
||||||
|
{ "secondary-seq", ko_no_argument, 354 },
|
||||||
|
{ "ds", ko_no_argument, 355 },
|
||||||
|
{ "rmq-inner", ko_required_argument, 356 },
|
||||||
|
{ "spsc", ko_required_argument, 357 },
|
||||||
|
{ "junc-pen", ko_required_argument, 358 },
|
||||||
|
{ "pairing", ko_required_argument, 359 },
|
||||||
|
{ "jump-min-match", ko_required_argument, 360 },
|
||||||
|
{ "write-junc", ko_no_argument, 361 },
|
||||||
|
{ "pass1", ko_required_argument, 362 },
|
||||||
|
{ "spsc-scale", ko_required_argument, 363 },
|
||||||
|
{ "spsc0", ko_required_argument, 364 },
|
||||||
|
{ "dbg-seed-occ", ko_no_argument, 501 },
|
||||||
{ "help", ko_no_argument, 'h' },
|
{ "help", ko_no_argument, 'h' },
|
||||||
{ "max-intron-len", ko_required_argument, 'G' },
|
{ "max-intron-len", ko_required_argument, 'G' },
|
||||||
{ "version", ko_no_argument, 'V' },
|
{ "version", ko_no_argument, 'V' },
|
||||||
@@ -117,12 +131,13 @@ static inline void yes_or_no(mm_mapopt_t *opt, int64_t flag, int long_idx, const
|
|||||||
|
|
||||||
int main(int argc, char *argv[])
|
int main(int argc, char *argv[])
|
||||||
{
|
{
|
||||||
const char *opt_str = "2aSDw:k:K:t:r:f:Vv:g:G:I:d:XT:s:x:Hcp:M:n:z:A:B:O:E:m:N:Qu:R:hF:LC:yYPo:e:U:";
|
const char *opt_str = "2aSDw:k:K:t:r:f:Vv:g:G:I:d:XT:s:x:Hcp:M:n:z:A:B:b:O:E:m:N:Qu:R:hF:LC:yYPo:e:U:J:j:";
|
||||||
ketopt_t o = KETOPT_INIT;
|
ketopt_t o = KETOPT_INIT;
|
||||||
mm_mapopt_t opt;
|
mm_mapopt_t opt;
|
||||||
mm_idxopt_t ipt;
|
mm_idxopt_t ipt;
|
||||||
int i, c, n_threads = 3, n_parts, old_best_n = -1;
|
int i, c, n_threads = 3, n_parts, old_best_n = -1;
|
||||||
char *fnw = 0, *rg = 0, *junc_bed = 0, *s, *alt_list = 0;
|
float spsc_scale = 0.7f;
|
||||||
|
char *fnw = 0, *rg = 0, *fn_bed_junc = 0, *fn_bed_jump = 0, *fn_bed_pass1 = 0, *fn_spsc = 0, *s, *alt_list = 0;
|
||||||
FILE *fp_help = stderr;
|
FILE *fp_help = stderr;
|
||||||
mm_idx_reader_t *idx_rdr;
|
mm_idx_reader_t *idx_rdr;
|
||||||
mm_idx_t *mi;
|
mm_idx_t *mi;
|
||||||
@@ -175,6 +190,7 @@ int main(int argc, char *argv[])
|
|||||||
else if (c == 'm') opt.min_chain_score = atoi(o.arg);
|
else if (c == 'm') opt.min_chain_score = atoi(o.arg);
|
||||||
else if (c == 'A') opt.a = atoi(o.arg);
|
else if (c == 'A') opt.a = atoi(o.arg);
|
||||||
else if (c == 'B') opt.b = atoi(o.arg);
|
else if (c == 'B') opt.b = atoi(o.arg);
|
||||||
|
else if (c == 'b') opt.transition = atoi(o.arg);
|
||||||
else if (c == 's') opt.min_dp_max = atoi(o.arg);
|
else if (c == 's') opt.min_dp_max = atoi(o.arg);
|
||||||
else if (c == 'C') opt.noncan = atoi(o.arg);
|
else if (c == 'C') opt.noncan = atoi(o.arg);
|
||||||
else if (c == 'I') ipt.batch_size = mm_parse_num(o.arg);
|
else if (c == 'I') ipt.batch_size = mm_parse_num(o.arg);
|
||||||
@@ -183,7 +199,13 @@ int main(int argc, char *argv[])
|
|||||||
else if (c == 'R') rg = o.arg;
|
else if (c == 'R') rg = o.arg;
|
||||||
else if (c == 'h') fp_help = stdout;
|
else if (c == 'h') fp_help = stdout;
|
||||||
else if (c == '2') opt.flag |= MM_F_2_IO_THREADS;
|
else if (c == '2') opt.flag |= MM_F_2_IO_THREADS;
|
||||||
else if (c == 'o') {
|
else if (c == 'j') fn_bed_jump = o.arg;
|
||||||
|
else if (c == 'J') {
|
||||||
|
int t;
|
||||||
|
t = atoi(o.arg);
|
||||||
|
if (t == 0) opt.flag |= MM_F_SPLICE_OLD;
|
||||||
|
else if (t == 1) opt.flag &= ~MM_F_SPLICE_OLD;
|
||||||
|
} else if (c == 'o') {
|
||||||
if (strcmp(o.arg, "-") != 0) {
|
if (strcmp(o.arg, "-") != 0) {
|
||||||
if (freopen(o.arg, "wb", stdout) == NULL) {
|
if (freopen(o.arg, "wb", stdout) == NULL) {
|
||||||
fprintf(stderr, "[ERROR]\033[1;31m failed to write the output to file '%s'\033[0m: %s\n", o.arg, strerror(errno));
|
fprintf(stderr, "[ERROR]\033[1;31m failed to write the output to file '%s'\033[0m: %s\n", o.arg, strerror(errno));
|
||||||
@@ -202,9 +224,8 @@ int main(int argc, char *argv[])
|
|||||||
else if (c == 309) mm_dbg_flag |= MM_DBG_PRINT_QNAME | MM_DBG_PRINT_ALN_SEQ, n_threads = 1; // --print-aln-seq
|
else if (c == 309) mm_dbg_flag |= MM_DBG_PRINT_QNAME | MM_DBG_PRINT_ALN_SEQ, n_threads = 1; // --print-aln-seq
|
||||||
else if (c == 310) opt.flag |= MM_F_SPLICE; // --splice
|
else if (c == 310) opt.flag |= MM_F_SPLICE; // --splice
|
||||||
else if (c == 312) opt.flag |= MM_F_NO_LJOIN; // --no-long-join
|
else if (c == 312) opt.flag |= MM_F_NO_LJOIN; // --no-long-join
|
||||||
else if (c == 313) opt.flag |= MM_F_SR; // --sr
|
|
||||||
else if (c == 317) opt.end_bonus = atoi(o.arg); // --end-bonus
|
else if (c == 317) opt.end_bonus = atoi(o.arg); // --end-bonus
|
||||||
else if (c == 318) opt.flag |= MM_F_INDEPEND_SEG; // --no-pairing
|
else if (c == 318) opt.flag |= MM_F_INDEPEND_SEG; // --no-pairing (deprecated)
|
||||||
else if (c == 320) ipt.flag |= MM_I_NO_SEQ; // --idx-no-seq
|
else if (c == 320) ipt.flag |= MM_I_NO_SEQ; // --idx-no-seq
|
||||||
else if (c == 321) opt.anchor_ext_shift = atoi(o.arg); // --end-seed-pen
|
else if (c == 321) opt.anchor_ext_shift = atoi(o.arg); // --end-seed-pen
|
||||||
else if (c == 322) opt.flag |= MM_F_FOR_ONLY; // --for-only
|
else if (c == 322) opt.flag |= MM_F_FOR_ONLY; // --for-only
|
||||||
@@ -220,17 +241,42 @@ int main(int argc, char *argv[])
|
|||||||
else if (c == 336) opt.flag |= MM_F_HARD_MLEVEL; // --hard-mask-level
|
else if (c == 336) opt.flag |= MM_F_HARD_MLEVEL; // --hard-mask-level
|
||||||
else if (c == 337) opt.max_sw_mat = mm_parse_num(o.arg); // --cap-sw-mat
|
else if (c == 337) opt.max_sw_mat = mm_parse_num(o.arg); // --cap-sw-mat
|
||||||
else if (c == 338) opt.max_qlen = mm_parse_num(o.arg); // --max-qlen
|
else if (c == 338) opt.max_qlen = mm_parse_num(o.arg); // --max-qlen
|
||||||
else if (c == 340) junc_bed = o.arg; // --junc-bed
|
else if (c == 340) fn_bed_junc = o.arg; // --junc-bed
|
||||||
else if (c == 341) opt.junc_bonus = atoi(o.arg); // --junc-bonus
|
else if (c == 341) opt.junc_bonus = atoi(o.arg); // --junc-bonus
|
||||||
else if (c == 342) opt.flag |= MM_F_SAM_HIT_ONLY; // --sam-hit-only
|
else if (c == 342) opt.flag |= MM_F_SAM_HIT_ONLY; // --sam-hit-only
|
||||||
else if (c == 343) opt.chain_gap_scale = atof(o.arg); // --chain-gap-scale
|
else if (c == 343) opt.chain_gap_scale = atof(o.arg); // --chain-gap-scale
|
||||||
|
else if (c == 351) opt.chain_skip_scale = atof(o.arg); // --chain-skip-scale
|
||||||
else if (c == 344) alt_list = o.arg; // --alt
|
else if (c == 344) alt_list = o.arg; // --alt
|
||||||
else if (c == 345) opt.alt_drop = atof(o.arg); // --alt-drop
|
else if (c == 345) opt.alt_drop = atof(o.arg); // --alt-drop
|
||||||
else if (c == 346) opt.mask_len = mm_parse_num(o.arg); // --mask-len
|
else if (c == 346) opt.mask_len = mm_parse_num(o.arg); // --mask-len
|
||||||
else if (c == 348) opt.flag |= MM_F_QSTRAND | MM_F_NO_INV; // --qstrand
|
else if (c == 348) opt.flag |= MM_F_QSTRAND | MM_F_NO_INV; // --qstrand
|
||||||
else if (c == 349) opt.cap_kalloc = mm_parse_num(o.arg); // --cap-kalloc
|
else if (c == 349) opt.cap_kalloc = mm_parse_num(o.arg); // --cap-kalloc
|
||||||
|
else if (c == 350) opt.q_occ_frac = atof(o.arg); // --q-occ-frac
|
||||||
|
else if (c == 352) mm_dbg_flag |= MM_DBG_PRINT_CHAIN; // --print-chains
|
||||||
|
else if (c == 353) opt.flag |= MM_F_NO_HASH_NAME; // --no-hash-name
|
||||||
|
else if (c == 354) opt.flag |= MM_F_SECONDARY_SEQ; // --secondary-seq
|
||||||
|
else if (c == 355) opt.flag |= MM_F_OUT_DS; // --ds
|
||||||
|
else if (c == 356) opt.rmq_inner_dist = mm_parse_num(o.arg); // --rmq-inner
|
||||||
|
else if (c == 357) fn_spsc = o.arg; // --spsc
|
||||||
|
else if (c == 360) opt.jump_min_match = mm_parse_num(o.arg); // --jump-min-match
|
||||||
|
else if (c == 361) opt.flag |= MM_F_OUT_JUNC | MM_F_CIGAR; // --write-junc
|
||||||
|
else if (c == 362) fn_bed_pass1 = o.arg; // --jump-pass1
|
||||||
|
else if (c == 501) mm_dbg_flag |= MM_DBG_SEED_FREQ; // --dbg-seed-occ
|
||||||
|
else if (c == 363) spsc_scale = atof(o.arg); // --spsc-scale
|
||||||
|
else if (c == 358 || c == 364) opt.junc_pen = atoi(o.arg); // --junc-pen or --spsc0
|
||||||
else if (c == 330) {
|
else if (c == 330) {
|
||||||
fprintf(stderr, "[WARNING] \033[1;31m --lj-min-ratio has been deprecated.\033[0m\n");
|
fprintf(stderr, "[WARNING] \033[1;31m --lj-min-ratio has been deprecated.\033[0m\n");
|
||||||
|
} else if (c == 313) { // --sr
|
||||||
|
if (o.arg == 0 || strcmp(o.arg, "dna") == 0) {
|
||||||
|
opt.flag |= MM_F_SR;
|
||||||
|
} else if (strcmp(o.arg, "rna") == 0) {
|
||||||
|
opt.flag |= MM_F_SR_RNA;
|
||||||
|
} else if (strcmp(o.arg, "no") == 0) {
|
||||||
|
opt.flag &= ~(uint64_t)(MM_F_SR|MM_F_SR_RNA);
|
||||||
|
} else if (mm_verbose >= 2) {
|
||||||
|
opt.flag |= MM_F_SR;
|
||||||
|
fprintf(stderr, "[WARNING]\033[1;31m --sr only takes 'dna' or 'rna'. Invalid values are assumed to be 'dna'.\033[0m\n");
|
||||||
|
}
|
||||||
} else if (c == 314) { // --frag
|
} else if (c == 314) { // --frag
|
||||||
yes_or_no(&opt, MM_F_FRAG_MODE, o.longidx, o.arg, 1);
|
yes_or_no(&opt, MM_F_FRAG_MODE, o.longidx, o.arg, 1);
|
||||||
} else if (c == 315) { // --secondary
|
} else if (c == 315) { // --secondary
|
||||||
@@ -253,7 +299,16 @@ int main(int argc, char *argv[])
|
|||||||
} else if (c == 326) { // --dual
|
} else if (c == 326) { // --dual
|
||||||
yes_or_no(&opt, MM_F_NO_DUAL, o.longidx, o.arg, 0);
|
yes_or_no(&opt, MM_F_NO_DUAL, o.longidx, o.arg, 0);
|
||||||
} else if (c == 347) { // --rmq
|
} else if (c == 347) { // --rmq
|
||||||
yes_or_no(&opt, MM_F_RMQ, o.longidx, o.arg, 1);
|
if (o.arg) yes_or_no(&opt, MM_F_RMQ, o.longidx, o.arg, 1);
|
||||||
|
else opt.flag |= MM_F_RMQ;
|
||||||
|
} else if (c == 359) { // --pairing
|
||||||
|
if (strcmp(o.arg, "no") == 0) opt.flag |= MM_F_INDEPEND_SEG;
|
||||||
|
else if (strcmp(o.arg, "weak") == 0) opt.flag |= MM_F_WEAK_PAIRING, opt.flag &= ~(uint64_t)MM_F_INDEPEND_SEG;
|
||||||
|
else {
|
||||||
|
if (strcmp(o.arg, "strong") != 0 && mm_verbose >= 2)
|
||||||
|
fprintf(stderr, "[WARNING]\033[1;31m unrecognized argument for --pairing; assuming 'strong'.\033[0m\n");
|
||||||
|
opt.flag &= ~(uint64_t)(MM_F_INDEPEND_SEG|MM_F_WEAK_PAIRING);
|
||||||
|
}
|
||||||
} else if (c == 'S') {
|
} else if (c == 'S') {
|
||||||
opt.flag |= MM_F_OUT_CS | MM_F_CIGAR | MM_F_OUT_CS_LONG;
|
opt.flag |= MM_F_OUT_CS | MM_F_CIGAR | MM_F_OUT_CS_LONG;
|
||||||
if (mm_verbose >= 2)
|
if (mm_verbose >= 2)
|
||||||
@@ -294,10 +349,6 @@ int main(int argc, char *argv[])
|
|||||||
if (*s == ',') opt.e2 = strtol(s + 1, &s, 10);
|
if (*s == ',') opt.e2 = strtol(s + 1, &s, 10);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if ((opt.flag & MM_F_SPLICE) && (opt.flag & MM_F_FRAG_MODE)) {
|
|
||||||
fprintf(stderr, "[ERROR]\033[1;31m --splice and --frag should not be specified at the same time.\033[0m\n");
|
|
||||||
return 1;
|
|
||||||
}
|
|
||||||
if (!fnw && !(opt.flag&MM_F_CIGAR))
|
if (!fnw && !(opt.flag&MM_F_CIGAR))
|
||||||
ipt.flag |= MM_I_NO_SEQ;
|
ipt.flag |= MM_I_NO_SEQ;
|
||||||
if (mm_check_opt(&ipt, &opt) < 0)
|
if (mm_check_opt(&ipt, &opt) < 0)
|
||||||
@@ -314,7 +365,7 @@ int main(int argc, char *argv[])
|
|||||||
fprintf(fp_help, " -H use homopolymer-compressed k-mer (preferrable for PacBio)\n");
|
fprintf(fp_help, " -H use homopolymer-compressed k-mer (preferrable for PacBio)\n");
|
||||||
fprintf(fp_help, " -k INT k-mer size (no larger than 28) [%d]\n", ipt.k);
|
fprintf(fp_help, " -k INT k-mer size (no larger than 28) [%d]\n", ipt.k);
|
||||||
fprintf(fp_help, " -w INT minimizer window size [%d]\n", ipt.w);
|
fprintf(fp_help, " -w INT minimizer window size [%d]\n", ipt.w);
|
||||||
fprintf(fp_help, " -I NUM split index for every ~NUM input bases [4G]\n");
|
fprintf(fp_help, " -I NUM split index for every ~NUM input bases [8G]\n");
|
||||||
fprintf(fp_help, " -d FILE dump index to FILE []\n");
|
fprintf(fp_help, " -d FILE dump index to FILE []\n");
|
||||||
fprintf(fp_help, " Mapping:\n");
|
fprintf(fp_help, " Mapping:\n");
|
||||||
fprintf(fp_help, " -f FLOAT filter out top FLOAT fraction of repetitive minimizers [%g]\n", opt.mid_occ_frac);
|
fprintf(fp_help, " -f FLOAT filter out top FLOAT fraction of repetitive minimizers [%g]\n", opt.mid_occ_frac);
|
||||||
@@ -336,6 +387,8 @@ int main(int argc, char *argv[])
|
|||||||
fprintf(fp_help, " -z INT[,INT] Z-drop score and inversion Z-drop score [%d,%d]\n", opt.zdrop, opt.zdrop_inv);
|
fprintf(fp_help, " -z INT[,INT] Z-drop score and inversion Z-drop score [%d,%d]\n", opt.zdrop, opt.zdrop_inv);
|
||||||
fprintf(fp_help, " -s INT minimal peak DP alignment score [%d]\n", opt.min_dp_max);
|
fprintf(fp_help, " -s INT minimal peak DP alignment score [%d]\n", opt.min_dp_max);
|
||||||
fprintf(fp_help, " -u CHAR how to find GT-AG. f:transcript strand, b:both strands, n:don't match GT-AG [n]\n");
|
fprintf(fp_help, " -u CHAR how to find GT-AG. f:transcript strand, b:both strands, n:don't match GT-AG [n]\n");
|
||||||
|
fprintf(fp_help, " -J INT splice mode. 0: original minimap2 model; 1: miniprot model [1]\n");
|
||||||
|
fprintf(fp_help, " -j FILE junctions in BED12 to extend *short* RNA-seq alignment []\n");
|
||||||
fprintf(fp_help, " Input/Output:\n");
|
fprintf(fp_help, " Input/Output:\n");
|
||||||
fprintf(fp_help, " -a output in the SAM format (PAF by default)\n");
|
fprintf(fp_help, " -a output in the SAM format (PAF by default)\n");
|
||||||
fprintf(fp_help, " -o FILE output alignments to FILE [stdout]\n");
|
fprintf(fp_help, " -o FILE output alignments to FILE [stdout]\n");
|
||||||
@@ -343,21 +396,24 @@ int main(int argc, char *argv[])
|
|||||||
fprintf(fp_help, " -R STR SAM read group line in a format like '@RG\\tID:foo\\tSM:bar' []\n");
|
fprintf(fp_help, " -R STR SAM read group line in a format like '@RG\\tID:foo\\tSM:bar' []\n");
|
||||||
fprintf(fp_help, " -c output CIGAR in PAF\n");
|
fprintf(fp_help, " -c output CIGAR in PAF\n");
|
||||||
fprintf(fp_help, " --cs[=STR] output the cs tag; STR is 'short' (if absent) or 'long' [none]\n");
|
fprintf(fp_help, " --cs[=STR] output the cs tag; STR is 'short' (if absent) or 'long' [none]\n");
|
||||||
|
fprintf(fp_help, " --ds output the ds tag, which is an extension to cs\n");
|
||||||
fprintf(fp_help, " --MD output the MD tag\n");
|
fprintf(fp_help, " --MD output the MD tag\n");
|
||||||
fprintf(fp_help, " --eqx write =/X CIGAR operators\n");
|
fprintf(fp_help, " --eqx write =/X CIGAR operators\n");
|
||||||
fprintf(fp_help, " -Y use soft clipping for supplementary alignments\n");
|
fprintf(fp_help, " -Y use soft clipping for supplementary alignments\n");
|
||||||
|
fprintf(fp_help, " -y copy FASTA/Q comments to output SAM\n");
|
||||||
fprintf(fp_help, " -t INT number of threads [%d]\n", n_threads);
|
fprintf(fp_help, " -t INT number of threads [%d]\n", n_threads);
|
||||||
fprintf(fp_help, " -K NUM minibatch size for mapping [500M]\n");
|
fprintf(fp_help, " -K NUM minibatch size for mapping [500M]\n");
|
||||||
// fprintf(fp_help, " -v INT verbose level [%d]\n", mm_verbose);
|
// fprintf(fp_help, " -v INT verbose level [%d]\n", mm_verbose);
|
||||||
fprintf(fp_help, " --version show version number\n");
|
fprintf(fp_help, " --version show version number\n");
|
||||||
fprintf(fp_help, " Preset:\n");
|
fprintf(fp_help, " Preset:\n");
|
||||||
fprintf(fp_help, " -x STR preset (always applied before other options; see minimap2.1 for details) []\n");
|
fprintf(fp_help, " -x STR preset (always applied before other options; see minimap2.1 for details) []\n");
|
||||||
fprintf(fp_help, " - map-pb/map-ont - PacBio CLR/Nanopore vs reference mapping\n");
|
fprintf(fp_help, " - lr:hq - accurate long reads (error rate <1%%) against a reference genome\n");
|
||||||
fprintf(fp_help, " - map-hifi - PacBio HiFi reads vs reference mapping\n");
|
fprintf(fp_help, " - splice/splice:hq - spliced alignment for long reads/accurate long reads\n");
|
||||||
fprintf(fp_help, " - ava-pb/ava-ont - PacBio/Nanopore read overlap\n");
|
fprintf(fp_help, " - splice:sr - spliced alignment for short RNA-seq reads\n");
|
||||||
fprintf(fp_help, " - asm5/asm10/asm20 - asm-to-ref mapping, for ~0.1/1/5%% sequence divergence\n");
|
fprintf(fp_help, " - asm5/asm10/asm20 - asm-to-ref mapping, for ~0.1/1/5%% sequence divergence\n");
|
||||||
fprintf(fp_help, " - splice/splice:hq - long-read/Pacbio-CCS spliced alignment\n");
|
fprintf(fp_help, " - sr - short reads against a reference\n");
|
||||||
fprintf(fp_help, " - sr - genomic short-read mapping\n");
|
fprintf(fp_help, " - map-pb/map-hifi/map-ont/map-iclr - CLR/HiFi/Nanopore/ICLR vs reference mapping\n");
|
||||||
|
fprintf(fp_help, " - ava-pb/ava-ont - PacBio CLR/Nanopore read overlap\n");
|
||||||
fprintf(fp_help, "\nSee `man ./minimap2.1' for detailed description of these and other advanced command-line options.\n");
|
fprintf(fp_help, "\nSee `man ./minimap2.1' for detailed description of these and other advanced command-line options.\n");
|
||||||
return fp_help == stdout? 0 : 1;
|
return fp_help == stdout? 0 : 1;
|
||||||
}
|
}
|
||||||
@@ -408,7 +464,26 @@ int main(int argc, char *argv[])
|
|||||||
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), mi->n_seq);
|
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), mi->n_seq);
|
||||||
if (argc != o.ind + 1) mm_mapopt_update(&opt, mi);
|
if (argc != o.ind + 1) mm_mapopt_update(&opt, mi);
|
||||||
if (mm_verbose >= 3) mm_idx_stat(mi);
|
if (mm_verbose >= 3) mm_idx_stat(mi);
|
||||||
if (junc_bed) mm_idx_bed_read(mi, junc_bed, 1);
|
if (fn_bed_junc) {
|
||||||
|
mm_idx_bed_read(mi, fn_bed_junc, 1);
|
||||||
|
if (mi->I == 0 && mm_verbose >= 2)
|
||||||
|
fprintf(stderr, "[WARNING] failed to load the junction BED file\n");
|
||||||
|
}
|
||||||
|
if (fn_bed_jump) {
|
||||||
|
mm_idx_jjump_read(mi, fn_bed_jump, MM_JUNC_ANNO, -1);
|
||||||
|
if (mi->J == 0 && mm_verbose >= 2)
|
||||||
|
fprintf(stderr, "[WARNING] failed to load the jump BED file\n");
|
||||||
|
}
|
||||||
|
if (fn_bed_pass1) {
|
||||||
|
mm_idx_jjump_read(mi, fn_bed_pass1, MM_JUNC_MISC, 5);
|
||||||
|
if (mi->J == 0 && mm_verbose >= 2)
|
||||||
|
fprintf(stderr, "[WARNING] failed to load the pass-1 jump BED file\n");
|
||||||
|
}
|
||||||
|
if (fn_spsc) {
|
||||||
|
mm_idx_spsc_read2(mi, fn_spsc, mm_max_spsc_bonus(&opt), spsc_scale);
|
||||||
|
if (mi->spsc == 0 && mm_verbose >= 2)
|
||||||
|
fprintf(stderr, "[WARNING] failed to load the splice score file\n");
|
||||||
|
}
|
||||||
if (alt_list) mm_idx_alt_read(mi, alt_list);
|
if (alt_list) mm_idx_alt_read(mi, alt_list);
|
||||||
if (argc - (o.ind + 1) == 0) {
|
if (argc - (o.ind + 1) == 0) {
|
||||||
mm_idx_destroy(mi);
|
mm_idx_destroy(mi);
|
||||||
|
|||||||
@@ -10,11 +10,6 @@
|
|||||||
#include "bseq.h"
|
#include "bseq.h"
|
||||||
#include "khash.h"
|
#include "khash.h"
|
||||||
|
|
||||||
struct mm_tbuf_s {
|
|
||||||
void *km;
|
|
||||||
int rep_len, frag_gap;
|
|
||||||
};
|
|
||||||
|
|
||||||
mm_tbuf_t *mm_tbuf_init(void)
|
mm_tbuf_t *mm_tbuf_init(void)
|
||||||
{
|
{
|
||||||
mm_tbuf_t *b;
|
mm_tbuf_t *b;
|
||||||
@@ -212,7 +207,7 @@ static void chain_post(const mm_mapopt_t *opt, int max_chain_gap_ref, const mm_i
|
|||||||
{
|
{
|
||||||
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
||||||
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
||||||
if (n_segs <= 1) mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, n_regs, regs);
|
if (n_segs <= 1) mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, 1, opt->max_gap * 0.8, n_regs, regs);
|
||||||
else mm_select_sub_multi(km, opt->pri_ratio, 0.2f, 0.7f, max_chain_gap_ref, mi->k*2, opt->best_n, n_segs, qlens, n_regs, regs);
|
else mm_select_sub_multi(km, opt->pri_ratio, 0.2f, 0.7f, max_chain_gap_ref, mi->k*2, opt->best_n, n_segs, qlens, n_regs, regs);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -223,16 +218,16 @@ static mm_reg1_t *align_regs(const mm_mapopt_t *opt, const mm_idx_t *mi, void *k
|
|||||||
regs = mm_align_skeleton(km, opt, mi, qlen, seq, n_regs, regs, a); // this calls mm_filter_regs()
|
regs = mm_align_skeleton(km, opt, mi, qlen, seq, n_regs, regs, a); // this calls mm_filter_regs()
|
||||||
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
if (!(opt->flag & MM_F_ALL_CHAINS)) { // don't choose primary mapping(s)
|
||||||
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
mm_set_parent(km, opt->mask_level, opt->mask_len, *n_regs, regs, opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
||||||
mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, n_regs, regs);
|
mm_select_sub(km, opt->pri_ratio, mi->k*2, opt->best_n, 0, opt->max_gap * 0.8, n_regs, regs);
|
||||||
mm_set_sam_pri(*n_regs, regs);
|
mm_set_sam_pri(*n_regs, regs);
|
||||||
}
|
}
|
||||||
return regs;
|
return regs;
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **seqs, int *n_regs, mm_reg1_t **regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
void mm_map_frag_core(const mm_idx_t *mi, int n_segs, const int *qlens, const char **seqs, int *n_regs, mm_reg1_t **regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
||||||
{
|
{
|
||||||
int i, j, rep_len, qlen_sum, n_regs0, n_mini_pos;
|
int i, j, rep_len, qlen_sum, n_regs0, n_mini_pos;
|
||||||
int max_chain_gap_qry, max_chain_gap_ref, is_splice = !!(opt->flag & MM_F_SPLICE), is_sr = !!(opt->flag & MM_F_SR);
|
int max_chain_gap_qry, max_chain_gap_ref, is_splice = !!(opt->flag & MM_F_SPLICE), is_sr = !!(opt->flag & MM_F_SR), is_sr_rna = !!(opt->flag & MM_F_SR_RNA);
|
||||||
uint32_t hash;
|
uint32_t hash;
|
||||||
int64_t n_a;
|
int64_t n_a;
|
||||||
uint64_t *u, *mini_pos;
|
uint64_t *u, *mini_pos;
|
||||||
@@ -240,6 +235,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
mm128_v mv = {0,0,0};
|
mm128_v mv = {0,0,0};
|
||||||
mm_reg1_t *regs0;
|
mm_reg1_t *regs0;
|
||||||
km_stat_t kmst;
|
km_stat_t kmst;
|
||||||
|
float chn_pen_gap, chn_pen_skip;
|
||||||
|
|
||||||
for (i = 0, qlen_sum = 0; i < n_segs; ++i)
|
for (i = 0, qlen_sum = 0; i < n_segs; ++i)
|
||||||
qlen_sum += qlens[i], n_regs[i] = 0, regs[i] = 0;
|
qlen_sum += qlens[i], n_regs[i] = 0, regs[i] = 0;
|
||||||
@@ -247,11 +243,12 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
if (qlen_sum == 0 || n_segs <= 0 || n_segs > MM_MAX_SEG) return;
|
if (qlen_sum == 0 || n_segs <= 0 || n_segs > MM_MAX_SEG) return;
|
||||||
if (opt->max_qlen > 0 && qlen_sum > opt->max_qlen) return;
|
if (opt->max_qlen > 0 && qlen_sum > opt->max_qlen) return;
|
||||||
|
|
||||||
hash = qname? __ac_X31_hash_string(qname) : 0;
|
hash = qname && !(opt->flag & MM_F_NO_HASH_NAME)? __ac_X31_hash_string(qname) : 0;
|
||||||
hash ^= __ac_Wang_hash(qlen_sum) + __ac_Wang_hash(opt->seed);
|
hash ^= __ac_Wang_hash(qlen_sum) + __ac_Wang_hash(opt->seed);
|
||||||
hash = __ac_Wang_hash(hash);
|
hash = __ac_Wang_hash(hash);
|
||||||
|
|
||||||
collect_minimizers(b->km, opt, mi, n_segs, qlens, seqs, &mv);
|
collect_minimizers(b->km, opt, mi, n_segs, qlens, seqs, &mv);
|
||||||
|
if (opt->q_occ_frac > 0.0f) mm_seed_mz_flt(b->km, &mv, opt->mid_occ, opt->q_occ_frac);
|
||||||
if (opt->flag & MM_F_HEAP_SORT) a = collect_seed_hits_heap(b->km, opt, opt->mid_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
if (opt->flag & MM_F_HEAP_SORT) a = collect_seed_hits_heap(b->km, opt, opt->mid_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||||
else a = collect_seed_hits(b->km, opt, opt->mid_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
else a = collect_seed_hits(b->km, opt, opt->mid_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||||
|
|
||||||
@@ -273,12 +270,14 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
if (max_chain_gap_ref < opt->max_gap) max_chain_gap_ref = opt->max_gap;
|
if (max_chain_gap_ref < opt->max_gap) max_chain_gap_ref = opt->max_gap;
|
||||||
} else max_chain_gap_ref = opt->max_gap;
|
} else max_chain_gap_ref = opt->max_gap;
|
||||||
|
|
||||||
|
chn_pen_gap = opt->chain_gap_scale * 0.01 * mi->k;
|
||||||
|
chn_pen_skip = opt->chain_skip_scale * 0.01 * mi->k;
|
||||||
if (opt->flag & MM_F_RMQ) {
|
if (opt->flag & MM_F_RMQ) {
|
||||||
a = mg_lchain_rmq(opt->max_gap, opt->rmq_inner_dist, opt->bw, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
a = mg_lchain_rmq(opt->max_gap, opt->rmq_inner_dist, opt->bw, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
||||||
opt->chain_gap_scale * 0.01 * mi->k, 0.0f, n_a, a, &n_regs0, &u, b->km);
|
chn_pen_gap, chn_pen_skip, n_a, a, &n_regs0, &u, b->km);
|
||||||
} else {
|
} else {
|
||||||
a = mg_lchain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score,
|
a = mg_lchain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score,
|
||||||
opt->chain_gap_scale * 0.01 * mi->k, 0.0f, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
chn_pen_gap, chn_pen_skip, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (opt->bw_long > opt->bw && (opt->flag & (MM_F_SPLICE|MM_F_SR|MM_F_NO_LJOIN)) == 0 && n_segs == 1 && n_regs0 > 1) { // re-chain/long-join for long sequences
|
if (opt->bw_long > opt->bw && (opt->flag & (MM_F_SPLICE|MM_F_SR|MM_F_NO_LJOIN)) == 0 && n_segs == 1 && n_regs0 > 1) { // re-chain/long-join for long sequences
|
||||||
@@ -289,7 +288,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
kfree(b->km, u);
|
kfree(b->km, u);
|
||||||
radix_sort_128x(a, a + n_a);
|
radix_sort_128x(a, a + n_a);
|
||||||
a = mg_lchain_rmq(opt->max_gap, opt->rmq_inner_dist, opt->bw_long, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
a = mg_lchain_rmq(opt->max_gap, opt->rmq_inner_dist, opt->bw_long, opt->max_chain_skip, opt->rmq_size_cap, opt->min_cnt, opt->min_chain_score,
|
||||||
opt->chain_gap_scale * 0.01 * mi->k, 0.0f, n_a, a, &n_regs0, &u, b->km);
|
chn_pen_gap, chn_pen_skip, n_a, a, &n_regs0, &u, b->km);
|
||||||
}
|
}
|
||||||
} else if (opt->max_occ > opt->mid_occ && rep_len > 0 && !(opt->flag & MM_F_RMQ)) { // re-chain, mostly for short reads
|
} else if (opt->max_occ > opt->mid_occ && rep_len > 0 && !(opt->flag & MM_F_RMQ)) { // re-chain, mostly for short reads
|
||||||
int rechain = 0;
|
int rechain = 0;
|
||||||
@@ -312,7 +311,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
if (opt->flag & MM_F_HEAP_SORT) a = collect_seed_hits_heap(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
if (opt->flag & MM_F_HEAP_SORT) a = collect_seed_hits_heap(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||||
else a = collect_seed_hits(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
else a = collect_seed_hits(b->km, opt, opt->max_occ, mi, qname, &mv, qlen_sum, &n_a, &rep_len, &n_mini_pos, &mini_pos);
|
||||||
a = mg_lchain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score,
|
a = mg_lchain_dp(max_chain_gap_ref, max_chain_gap_qry, opt->bw, opt->max_chain_skip, opt->max_chain_iter, opt->min_cnt, opt->min_chain_score,
|
||||||
opt->chain_gap_scale * 0.01 * mi->k, 0.0f, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
chn_pen_gap, chn_pen_skip, is_splice, n_segs, n_a, a, &n_regs0, &u, b->km);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
b->frag_gap = max_chain_gap_ref;
|
b->frag_gap = max_chain_gap_ref;
|
||||||
@@ -324,20 +323,22 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
mm_hit_sort(b->km, &n_regs0, regs0, opt->alt_drop); // this step can be merged into mm_gen_regs(); will do if this shows up in profile
|
mm_hit_sort(b->km, &n_regs0, regs0, opt->alt_drop); // this step can be merged into mm_gen_regs(); will do if this shows up in profile
|
||||||
}
|
}
|
||||||
|
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_SEED)
|
if (mm_dbg_flag & (MM_DBG_PRINT_SEED|MM_DBG_PRINT_CHAIN))
|
||||||
for (j = 0; j < n_regs0; ++j)
|
for (j = 0; j < n_regs0; ++j)
|
||||||
for (i = regs0[j].as; i < regs0[j].as + regs0[j].cnt; ++i)
|
for (i = regs0[j].as; i < regs0[j].as + regs0[j].cnt; ++i)
|
||||||
fprintf(stderr, "CN\t%d\t%s\t%d\t%c\t%d\t%d\t%d\n", j, mi->seq[a[i].x<<1>>33].name, (int32_t)a[i].x, "+-"[a[i].x>>63], (int32_t)a[i].y, (int32_t)(a[i].y>>32&0xff),
|
fprintf(stderr, "CN\t%d\t%s\t%d\t%c\t%d\t%d\t%d\n", j, mi->seq[a[i].x<<1>>33].name, (int32_t)a[i].x, "+-"[a[i].x>>63], (int32_t)a[i].y, (int32_t)(a[i].y>>32&0xff),
|
||||||
i == regs0[j].as? 0 : ((int32_t)a[i].y - (int32_t)a[i-1].y) - ((int32_t)a[i].x - (int32_t)a[i-1].x));
|
i == regs0[j].as? 0 : ((int32_t)a[i].y - (int32_t)a[i-1].y) - ((int32_t)a[i].x - (int32_t)a[i-1].x));
|
||||||
|
|
||||||
chain_post(opt, max_chain_gap_ref, mi, b->km, qlen_sum, n_segs, qlens, &n_regs0, regs0, a);
|
chain_post(opt, max_chain_gap_ref, mi, b->km, qlen_sum, n_segs, qlens, &n_regs0, regs0, a);
|
||||||
if (!is_sr && !(opt->flag&MM_F_QSTRAND))
|
if (!is_sr && !(opt->flag&MM_F_QSTRAND)) {
|
||||||
mm_est_err(mi, qlen_sum, n_regs0, regs0, a, n_mini_pos, mini_pos);
|
mm_est_err(mi, qlen_sum, n_regs0, regs0, a, n_mini_pos, mini_pos);
|
||||||
|
n_regs0 = mm_filter_strand_retained(n_regs0, regs0);
|
||||||
|
}
|
||||||
|
|
||||||
if (n_segs == 1) { // uni-segment
|
if (n_segs == 1) { // uni-segment
|
||||||
regs0 = align_regs(opt, mi, b->km, qlens[0], seqs[0], &n_regs0, regs0, a);
|
regs0 = align_regs(opt, mi, b->km, qlens[0], seqs[0], &n_regs0, regs0, a);
|
||||||
regs0 = (mm_reg1_t*)realloc(regs0, sizeof(*regs0) * n_regs0);
|
regs0 = (mm_reg1_t*)realloc(regs0, sizeof(*regs0) * n_regs0);
|
||||||
mm_set_mapq(b->km, n_regs0, regs0, opt->min_chain_score, opt->a, rep_len, is_sr);
|
mm_set_mapq2(b->km, n_regs0, regs0, opt->min_chain_score, opt->a, rep_len, is_sr || is_sr_rna, is_splice);
|
||||||
n_regs[0] = n_regs0, regs[0] = regs0;
|
n_regs[0] = n_regs0, regs[0] = regs0;
|
||||||
} else { // multi-segment
|
} else { // multi-segment
|
||||||
mm_seg_t *seg;
|
mm_seg_t *seg;
|
||||||
@@ -346,7 +347,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
for (i = 0; i < n_segs; ++i) {
|
for (i = 0; i < n_segs; ++i) {
|
||||||
mm_set_parent(b->km, opt->mask_level, opt->mask_len, n_regs[i], regs[i], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop); // update mm_reg1_t::parent
|
mm_set_parent(b->km, opt->mask_level, opt->mask_len, n_regs[i], regs[i], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop); // update mm_reg1_t::parent
|
||||||
regs[i] = align_regs(opt, mi, b->km, qlens[i], seqs[i], &n_regs[i], regs[i], seg[i].a);
|
regs[i] = align_regs(opt, mi, b->km, qlens[i], seqs[i], &n_regs[i], regs[i], seg[i].a);
|
||||||
mm_set_mapq(b->km, n_regs[i], regs[i], opt->min_chain_score, opt->a, rep_len, is_sr);
|
mm_set_mapq2(b->km, n_regs[i], regs[i], opt->min_chain_score, opt->a, rep_len, is_sr || is_sr_rna, is_splice);
|
||||||
}
|
}
|
||||||
mm_seg_free(b->km, n_segs, seg);
|
mm_seg_free(b->km, n_segs, seg);
|
||||||
if (n_segs == 2 && opt->pe_ori >= 0 && (opt->flag&MM_F_CIGAR))
|
if (n_segs == 2 && opt->pe_ori >= 0 && (opt->flag&MM_F_CIGAR))
|
||||||
@@ -358,6 +359,10 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
kfree(b->km, u);
|
kfree(b->km, u);
|
||||||
kfree(b->km, mini_pos);
|
kfree(b->km, mini_pos);
|
||||||
|
|
||||||
|
if (mi->J && n_segs == 1 && is_splice)
|
||||||
|
for (i = 0; i < n_regs0; ++i)
|
||||||
|
mm_jump_split(b->km, mi, opt, qlens[0], (const uint8_t*)seqs[0], ®s0[i], 0);
|
||||||
|
|
||||||
if (b->km) {
|
if (b->km) {
|
||||||
km_stat(b->km, &kmst);
|
km_stat(b->km, &kmst);
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||||
@@ -372,6 +377,18 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **seqs, int *n_regs, mm_reg1_t **regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
||||||
|
{
|
||||||
|
if ((opt->flag & MM_F_WEAK_PAIRING) && n_segs == 2 && opt->pe_ori >= 0 && (opt->flag&MM_F_CIGAR)) {
|
||||||
|
int i;
|
||||||
|
for (i = 0; i < n_segs; ++i)
|
||||||
|
mm_map_frag_core(mi, 1, &qlens[i], &seqs[i], &n_regs[i], ®s[i], b, opt, qname);
|
||||||
|
mm_pair(b->km, opt->max_gap_ref, opt->pe_bonus, opt->a * 2 + opt->b, opt->a, qlens, n_regs, regs);
|
||||||
|
} else {
|
||||||
|
mm_map_frag_core(mi, n_segs, qlens, seqs, n_regs, regs, b, opt, qname);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
mm_reg1_t *mm_map(const mm_idx_t *mi, int qlen, const char *seq, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
mm_reg1_t *mm_map(const mm_idx_t *mi, int qlen, const char *seq, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
||||||
{
|
{
|
||||||
mm_reg1_t *regs;
|
mm_reg1_t *regs;
|
||||||
@@ -446,6 +463,10 @@ static void worker_for(void *_data, long i, int tid) // kt_for() callback
|
|||||||
r->qs = qlens[j] - r->qe;
|
r->qs = qlens[j] - r->qe;
|
||||||
r->qe = qlens[j] - t;
|
r->qe = qlens[j] - t;
|
||||||
r->rev = !r->rev;
|
r->rev = !r->rev;
|
||||||
|
if (r->p) {
|
||||||
|
if (r->p->trans_strand == 1) r->p->trans_strand = 2;
|
||||||
|
else if (r->p->trans_strand == 2) r->p->trans_strand = 1;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||||
@@ -505,10 +526,10 @@ static void merge_hits(step_t *s)
|
|||||||
mm_hit_sort(km, &s->n_reg[k], s->reg[k], opt->alt_drop);
|
mm_hit_sort(km, &s->n_reg[k], s->reg[k], opt->alt_drop);
|
||||||
mm_set_parent(km, opt->mask_level, opt->mask_len, s->n_reg[k], s->reg[k], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
mm_set_parent(km, opt->mask_level, opt->mask_len, s->n_reg[k], s->reg[k], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop);
|
||||||
if (!(opt->flag & MM_F_ALL_CHAINS)) {
|
if (!(opt->flag & MM_F_ALL_CHAINS)) {
|
||||||
mm_select_sub(km, opt->pri_ratio, s->p->mi->k*2, opt->best_n, &s->n_reg[k], s->reg[k]);
|
mm_select_sub(km, opt->pri_ratio, s->p->mi->k*2, opt->best_n, 0, opt->max_gap * 0.8, &s->n_reg[k], s->reg[k]);
|
||||||
mm_set_sam_pri(s->n_reg[k], s->reg[k]);
|
mm_set_sam_pri(s->n_reg[k], s->reg[k]);
|
||||||
}
|
}
|
||||||
mm_set_mapq(km, s->n_reg[k], s->reg[k], opt->min_chain_score, opt->a, rep_len, !!(opt->flag & MM_F_SR));
|
mm_set_mapq2(km, s->n_reg[k], s->reg[k], opt->min_chain_score, opt->a, rep_len, !!(opt->flag & (MM_F_SR|MM_F_SR_RNA)), !!(opt->flag & MM_F_SPLICE));
|
||||||
}
|
}
|
||||||
if (s->n_seg[f] == 2 && opt->pe_ori >= 0 && (opt->flag&MM_F_CIGAR))
|
if (s->n_seg[f] == 2 && opt->pe_ori >= 0 && (opt->flag&MM_F_CIGAR))
|
||||||
mm_pair(km, frag_gap_part[0], opt->pe_bonus, opt->a * 2 + opt->b, opt->a, qlens, &s->n_reg[k0], &s->reg[k0]);
|
mm_pair(km, frag_gap_part[0], opt->pe_bonus, opt->a * 2 + opt->b, opt->a, qlens, &s->n_reg[k0], &s->reg[k0]);
|
||||||
@@ -577,23 +598,30 @@ static void *worker_pipeline(void *shared, int step, void *in)
|
|||||||
mm_err_fwrite(r->p, r->p->capacity, 4, p->fp_split);
|
mm_err_fwrite(r->p, r->p->capacity, 4, p->fp_split);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
} else if (p->opt->flag & MM_F_OUT_JUNC) { // extra logic for --write-junc
|
||||||
|
for (j = 0; j < s->n_reg[i]; ++j) {
|
||||||
|
const mm_reg1_t *r = &s->reg[i][j];
|
||||||
|
if (r->id != r->parent || r->mapq < 10) continue;
|
||||||
|
mm_write_junc(&p->str, mi, t, r);
|
||||||
|
if (p->str.l > 0) mm_err_puts(p->str.s);
|
||||||
|
}
|
||||||
} else if (s->n_reg[i] > 0) { // the query has at least one hit
|
} else if (s->n_reg[i] > 0) { // the query has at least one hit
|
||||||
for (j = 0; j < s->n_reg[i]; ++j) {
|
for (j = 0; j < s->n_reg[i]; ++j) {
|
||||||
mm_reg1_t *r = &s->reg[i][j];
|
const mm_reg1_t *r = &s->reg[i][j];
|
||||||
assert(!r->sam_pri || r->id == r->parent);
|
assert(!r->sam_pri || r->id == r->parent);
|
||||||
if ((p->opt->flag & MM_F_NO_PRINT_2ND) && r->id != r->parent)
|
if ((p->opt->flag & MM_F_NO_PRINT_2ND) && r->id != r->parent)
|
||||||
continue;
|
continue;
|
||||||
if (p->opt->flag & MM_F_OUT_SAM)
|
if (p->opt->flag & MM_F_OUT_SAM)
|
||||||
mm_write_sam3(&p->str, mi, t, i - seg_st, j, s->n_seg[k], &s->n_reg[seg_st], (const mm_reg1_t*const*)&s->reg[seg_st], km, p->opt->flag, s->rep_len[i]);
|
mm_write_sam3(&p->str, mi, t, i - seg_st, j, s->n_seg[k], &s->n_reg[seg_st], (const mm_reg1_t*const*)&s->reg[seg_st], km, p->opt->flag, s->rep_len[i]);
|
||||||
else
|
else
|
||||||
mm_write_paf3(&p->str, mi, t, r, km, p->opt->flag, s->rep_len[i]);
|
mm_write_paf4(&p->str, mi, t, r, km, p->opt->flag, s->rep_len[i], s->n_seg[k], i - seg_st);
|
||||||
mm_err_puts(p->str.s);
|
mm_err_puts(p->str.s);
|
||||||
}
|
}
|
||||||
} else if ((p->opt->flag & MM_F_PAF_NO_HIT) || ((p->opt->flag & MM_F_OUT_SAM) && !(p->opt->flag & MM_F_SAM_HIT_ONLY))) { // output an empty hit, if requested
|
} else if ((p->opt->flag & MM_F_PAF_NO_HIT) || ((p->opt->flag & MM_F_OUT_SAM) && !(p->opt->flag & MM_F_SAM_HIT_ONLY))) { // output an empty hit, if requested
|
||||||
if (p->opt->flag & MM_F_OUT_SAM)
|
if (p->opt->flag & MM_F_OUT_SAM)
|
||||||
mm_write_sam3(&p->str, mi, t, i - seg_st, -1, s->n_seg[k], &s->n_reg[seg_st], (const mm_reg1_t*const*)&s->reg[seg_st], km, p->opt->flag, s->rep_len[i]);
|
mm_write_sam3(&p->str, mi, t, i - seg_st, -1, s->n_seg[k], &s->n_reg[seg_st], (const mm_reg1_t*const*)&s->reg[seg_st], km, p->opt->flag, s->rep_len[i]);
|
||||||
else
|
else
|
||||||
mm_write_paf3(&p->str, mi, t, 0, 0, p->opt->flag, s->rep_len[i]);
|
mm_write_paf4(&p->str, mi, t, 0, 0, p->opt->flag, s->rep_len[i], s->n_seg[k], i - seg_st);
|
||||||
mm_err_puts(p->str.s);
|
mm_err_puts(p->str.s);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -5,40 +5,49 @@
|
|||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <sys/types.h>
|
#include <sys/types.h>
|
||||||
|
|
||||||
#define MM_F_NO_DIAG 0x001 // no exact diagonal hit
|
#define MM_VERSION "2.31-r1302"
|
||||||
#define MM_F_NO_DUAL 0x002 // skip pairs where query name is lexicographically larger than target name
|
|
||||||
#define MM_F_CIGAR 0x004
|
#define MM_F_NO_DIAG (0x001LL) // no exact diagonal hit
|
||||||
#define MM_F_OUT_SAM 0x008
|
#define MM_F_NO_DUAL (0x002LL) // skip pairs where query name is lexicographically larger than target name
|
||||||
#define MM_F_NO_QUAL 0x010
|
#define MM_F_CIGAR (0x004LL)
|
||||||
#define MM_F_OUT_CG 0x020
|
#define MM_F_OUT_SAM (0x008LL)
|
||||||
#define MM_F_OUT_CS 0x040
|
#define MM_F_NO_QUAL (0x010LL)
|
||||||
#define MM_F_SPLICE 0x080 // splice mode
|
#define MM_F_OUT_CG (0x020LL)
|
||||||
#define MM_F_SPLICE_FOR 0x100 // match GT-AG
|
#define MM_F_OUT_CS (0x040LL)
|
||||||
#define MM_F_SPLICE_REV 0x200 // match CT-AC, the reverse complement of GT-AG
|
#define MM_F_SPLICE (0x080LL) // splice mode
|
||||||
#define MM_F_NO_LJOIN 0x400
|
#define MM_F_SPLICE_FOR (0x100LL) // match GT-AG
|
||||||
#define MM_F_OUT_CS_LONG 0x800
|
#define MM_F_SPLICE_REV (0x200LL) // match CT-AC, the reverse complement of GT-AG
|
||||||
#define MM_F_SR 0x1000
|
#define MM_F_NO_LJOIN (0x400LL)
|
||||||
#define MM_F_FRAG_MODE 0x2000
|
#define MM_F_OUT_CS_LONG (0x800LL)
|
||||||
#define MM_F_NO_PRINT_2ND 0x4000
|
#define MM_F_SR (0x1000LL)
|
||||||
#define MM_F_2_IO_THREADS 0x8000
|
#define MM_F_FRAG_MODE (0x2000LL)
|
||||||
#define MM_F_LONG_CIGAR 0x10000
|
#define MM_F_NO_PRINT_2ND (0x4000LL)
|
||||||
#define MM_F_INDEPEND_SEG 0x20000
|
#define MM_F_2_IO_THREADS (0x8000LL)
|
||||||
#define MM_F_SPLICE_FLANK 0x40000
|
#define MM_F_LONG_CIGAR (0x10000LL)
|
||||||
#define MM_F_SOFTCLIP 0x80000
|
#define MM_F_INDEPEND_SEG (0x20000LL)
|
||||||
#define MM_F_FOR_ONLY 0x100000
|
#define MM_F_SPLICE_FLANK (0x40000LL)
|
||||||
#define MM_F_REV_ONLY 0x200000
|
#define MM_F_SOFTCLIP (0x80000LL)
|
||||||
#define MM_F_HEAP_SORT 0x400000
|
#define MM_F_FOR_ONLY (0x100000LL)
|
||||||
#define MM_F_ALL_CHAINS 0x800000
|
#define MM_F_REV_ONLY (0x200000LL)
|
||||||
#define MM_F_OUT_MD 0x1000000
|
#define MM_F_HEAP_SORT (0x400000LL)
|
||||||
#define MM_F_COPY_COMMENT 0x2000000
|
#define MM_F_ALL_CHAINS (0x800000LL)
|
||||||
#define MM_F_EQX 0x4000000 // use =/X instead of M
|
#define MM_F_OUT_MD (0x1000000LL)
|
||||||
#define MM_F_PAF_NO_HIT 0x8000000 // output unmapped reads to PAF
|
#define MM_F_COPY_COMMENT (0x2000000LL)
|
||||||
#define MM_F_NO_END_FLT 0x10000000
|
#define MM_F_EQX (0x4000000LL) // use =/X instead of M
|
||||||
#define MM_F_HARD_MLEVEL 0x20000000
|
#define MM_F_PAF_NO_HIT (0x8000000LL) // output unmapped reads to PAF
|
||||||
#define MM_F_SAM_HIT_ONLY 0x40000000
|
#define MM_F_NO_END_FLT (0x10000000LL)
|
||||||
|
#define MM_F_HARD_MLEVEL (0x20000000LL)
|
||||||
|
#define MM_F_SAM_HIT_ONLY (0x40000000LL)
|
||||||
#define MM_F_RMQ (0x80000000LL)
|
#define MM_F_RMQ (0x80000000LL)
|
||||||
#define MM_F_QSTRAND (0x100000000LL)
|
#define MM_F_QSTRAND (0x100000000LL)
|
||||||
#define MM_F_NO_INV (0x200000000LL)
|
#define MM_F_NO_INV (0x200000000LL)
|
||||||
|
#define MM_F_NO_HASH_NAME (0x400000000LL)
|
||||||
|
#define MM_F_SPLICE_OLD (0x800000000LL)
|
||||||
|
#define MM_F_SECONDARY_SEQ (0x1000000000LL) //output SEQ field for seqondary alignments using hard clipping
|
||||||
|
#define MM_F_OUT_DS (0x2000000000LL)
|
||||||
|
#define MM_F_WEAK_PAIRING (0x4000000000LL)
|
||||||
|
#define MM_F_SR_RNA (0x8000000000LL)
|
||||||
|
#define MM_F_OUT_JUNC (0x10000000000LL)
|
||||||
|
|
||||||
#define MM_I_HPC 0x1
|
#define MM_I_HPC 0x1
|
||||||
#define MM_I_NO_SEQ 0x2
|
#define MM_I_NO_SEQ 0x2
|
||||||
@@ -85,6 +94,8 @@ typedef struct {
|
|||||||
uint32_t *S; // 4-bit packed sequence
|
uint32_t *S; // 4-bit packed sequence
|
||||||
struct mm_idx_bucket_s *B; // index (hidden)
|
struct mm_idx_bucket_s *B; // index (hidden)
|
||||||
struct mm_idx_intv_s *I; // intervals (hidden)
|
struct mm_idx_intv_s *I; // intervals (hidden)
|
||||||
|
struct mm_idx_spsc_s *spsc;// splice score (hidden)
|
||||||
|
struct mm_idx_jjump_s *J; // junctions to create jumps (hidden)
|
||||||
void *km, *h;
|
void *km, *h;
|
||||||
} mm_idx_t;
|
} mm_idx_t;
|
||||||
|
|
||||||
@@ -92,6 +103,7 @@ typedef struct {
|
|||||||
typedef struct {
|
typedef struct {
|
||||||
uint32_t capacity; // the capacity of cigar[]
|
uint32_t capacity; // the capacity of cigar[]
|
||||||
int32_t dp_score, dp_max, dp_max2; // DP score; score of the max-scoring segment; score of the best alternate mappings
|
int32_t dp_score, dp_max, dp_max2; // DP score; score of the max-scoring segment; score of the best alternate mappings
|
||||||
|
int32_t dp_max0; // DP score before mm_update_dp_max() adjustment
|
||||||
uint32_t n_ambi:30, trans_strand:2; // number of ambiguous bases; transcript strand: 0 for unknown, 1 for +, 2 for -
|
uint32_t n_ambi:30, trans_strand:2; // number of ambiguous bases; transcript strand: 0 for unknown, 1 for +, 2 for -
|
||||||
uint32_t n_cigar; // number of cigar operations in cigar[]
|
uint32_t n_cigar; // number of cigar operations in cigar[]
|
||||||
uint32_t cigar[];
|
uint32_t cigar[];
|
||||||
@@ -108,7 +120,7 @@ typedef struct {
|
|||||||
int32_t mlen, blen; // seeded exact match length; seeded alignment block length
|
int32_t mlen, blen; // seeded exact match length; seeded alignment block length
|
||||||
int32_t n_sub; // number of suboptimal mappings
|
int32_t n_sub; // number of suboptimal mappings
|
||||||
int32_t score0; // initial chaining score (before chain merging/spliting)
|
int32_t score0; // initial chaining score (before chain merging/spliting)
|
||||||
uint32_t mapq:8, split:2, rev:1, inv:1, sam_pri:1, proper_frag:1, pe_thru:1, seg_split:1, seg_id:8, split_inv:1, is_alt:1, dummy:6;
|
uint32_t mapq:8, split:2, rev:1, inv:1, sam_pri:1, proper_frag:1, pe_thru:1, seg_split:1, seg_id:8, split_inv:1, is_alt:1, strand_retained:1, is_spliced:1, dummy:4;
|
||||||
uint32_t hash;
|
uint32_t hash;
|
||||||
float div;
|
float div;
|
||||||
mm_extra_t *p;
|
mm_extra_t *p;
|
||||||
@@ -135,6 +147,7 @@ typedef struct {
|
|||||||
int min_cnt; // min number of minimizers on each chain
|
int min_cnt; // min number of minimizers on each chain
|
||||||
int min_chain_score; // min chaining score
|
int min_chain_score; // min chaining score
|
||||||
float chain_gap_scale;
|
float chain_gap_scale;
|
||||||
|
float chain_skip_scale;
|
||||||
int rmq_size_cap, rmq_inner_dist;
|
int rmq_size_cap, rmq_inner_dist;
|
||||||
int rmq_rescue_size;
|
int rmq_rescue_size;
|
||||||
float rmq_rescue_ratio;
|
float rmq_rescue_ratio;
|
||||||
@@ -147,9 +160,11 @@ typedef struct {
|
|||||||
float alt_drop;
|
float alt_drop;
|
||||||
|
|
||||||
int a, b, q, e, q2, e2; // matching score, mismatch, gap-open and gap-ext penalties
|
int a, b, q, e, q2, e2; // matching score, mismatch, gap-open and gap-ext penalties
|
||||||
|
int transition; // transition mismatch score (A:G, C:T)
|
||||||
int sc_ambi; // score when one or both bases are "N"
|
int sc_ambi; // score when one or both bases are "N"
|
||||||
int noncan; // cost of non-canonical splicing sites
|
int noncan; // cost of non-canonical splicing sites
|
||||||
int junc_bonus;
|
int junc_bonus; // bonus for a splice site in annotation
|
||||||
|
int junc_pen; // penalty for GT- or -AG not scored in --spsc
|
||||||
int zdrop, zdrop_inv; // break alignment if alignment score drops too fast along the diagonal
|
int zdrop, zdrop_inv; // break alignment if alignment score drops too fast along the diagonal
|
||||||
int end_bonus;
|
int end_bonus;
|
||||||
int min_dp_max; // drop an alignment if the score of the max scoring segment is below this threshold
|
int min_dp_max; // drop an alignment if the score of the max scoring segment is below this threshold
|
||||||
@@ -162,7 +177,10 @@ typedef struct {
|
|||||||
|
|
||||||
int pe_ori, pe_bonus;
|
int pe_ori, pe_bonus;
|
||||||
|
|
||||||
|
int32_t jump_min_match;
|
||||||
|
|
||||||
float mid_occ_frac; // only used by mm_mapopt_update(); see below
|
float mid_occ_frac; // only used by mm_mapopt_update(); see below
|
||||||
|
float q_occ_frac;
|
||||||
int32_t min_mid_occ, max_mid_occ;
|
int32_t min_mid_occ, max_mid_occ;
|
||||||
int32_t mid_occ; // ignore seeds with occurrences above this threshold
|
int32_t mid_occ; // ignore seeds with occurrences above this threshold
|
||||||
int32_t max_occ, max_max_occ, occ_dist;
|
int32_t max_occ, max_max_occ, occ_dist;
|
||||||
@@ -186,6 +204,11 @@ typedef struct {
|
|||||||
} mm_idx_reader_t;
|
} mm_idx_reader_t;
|
||||||
|
|
||||||
// memory buffer for thread-local storage during mapping
|
// memory buffer for thread-local storage during mapping
|
||||||
|
struct mm_tbuf_s {
|
||||||
|
void *km;
|
||||||
|
int rep_len, frag_gap;
|
||||||
|
};
|
||||||
|
|
||||||
typedef struct mm_tbuf_s mm_tbuf_t;
|
typedef struct mm_tbuf_s mm_tbuf_t;
|
||||||
|
|
||||||
// global variables
|
// global variables
|
||||||
@@ -385,6 +408,7 @@ int mm_map_file_frag(const mm_idx_t *idx, int n_segs, const char **fn, const mm_
|
|||||||
* @return the length of cs
|
* @return the length of cs
|
||||||
*/
|
*/
|
||||||
int mm_gen_cs(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden);
|
int mm_gen_cs(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden);
|
||||||
|
int mm_gen_ds(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden);
|
||||||
int mm_gen_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq);
|
int mm_gen_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq);
|
||||||
|
|
||||||
// query sequence name and sequence in the minimap2 index
|
// query sequence name and sequence in the minimap2 index
|
||||||
@@ -396,6 +420,11 @@ int mm_idx_alt_read(mm_idx_t *mi, const char *fn);
|
|||||||
int mm_idx_bed_read(mm_idx_t *mi, const char *fn, int read_junc);
|
int mm_idx_bed_read(mm_idx_t *mi, const char *fn, int read_junc);
|
||||||
int mm_idx_bed_junc(const mm_idx_t *mi, int32_t ctg, int32_t st, int32_t en, uint8_t *s);
|
int mm_idx_bed_junc(const mm_idx_t *mi, int32_t ctg, int32_t st, int32_t en, uint8_t *s);
|
||||||
|
|
||||||
|
int mm_max_spsc_bonus(const mm_mapopt_t *mo);
|
||||||
|
int32_t mm_idx_spsc_read(mm_idx_t *idx, const char *fn, int32_t max_sc);
|
||||||
|
int32_t mm_idx_spsc_read2(mm_idx_t *idx, const char *fn, int32_t max_sc, float scale);
|
||||||
|
int64_t mm_idx_spsc_get(const mm_idx_t *db, int32_t cid, int64_t st0, int64_t en0, int32_t rev, uint8_t *sc);
|
||||||
|
|
||||||
// deprecated APIs for backward compatibility
|
// deprecated APIs for backward compatibility
|
||||||
void mm_mapopt_init(mm_mapopt_t *opt);
|
void mm_mapopt_init(mm_mapopt_t *opt);
|
||||||
mm_idx_t *mm_idx_build(const char *fn, int w, int k, int flag, int n_threads);
|
mm_idx_t *mm_idx_build(const char *fn, int w, int k, int flag, int n_threads);
|
||||||
|
|||||||
+177
-49
@@ -1,4 +1,4 @@
|
|||||||
.TH minimap2 1 "7 August 2021" "minimap2-2.22 (r1101)" "Bioinformatics tools"
|
.TH minimap2 1 "19 May 2026" "minimap2-2.31 (r1302)" "Bioinformatics tools"
|
||||||
.SH NAME
|
.SH NAME
|
||||||
.PP
|
.PP
|
||||||
minimap2 - mapping and alignment between collections of DNA sequences
|
minimap2 - mapping and alignment between collections of DNA sequences
|
||||||
@@ -77,7 +77,7 @@ SAM format.
|
|||||||
Minimizer k-mer length [15]
|
Minimizer k-mer length [15]
|
||||||
.TP
|
.TP
|
||||||
.BI -w \ INT
|
.BI -w \ INT
|
||||||
Minimizer window size [2/3 of k-mer length]. A minimizer is the smallest k-mer
|
Minimizer window size [10]. A minimizer is the smallest k-mer
|
||||||
in a window of w consecutive k-mers.
|
in a window of w consecutive k-mers.
|
||||||
.TP
|
.TP
|
||||||
.B -H
|
.B -H
|
||||||
@@ -88,16 +88,17 @@ on the HPC sequence.
|
|||||||
.BI -I \ NUM
|
.BI -I \ NUM
|
||||||
Load at most
|
Load at most
|
||||||
.I NUM
|
.I NUM
|
||||||
target bases into RAM for indexing [4G]. If there are more than
|
target bases into RAM for indexing [8G]. If there are more than
|
||||||
.I NUM
|
.I NUM
|
||||||
bases in
|
bases in
|
||||||
.IR target.fa ,
|
.IR target.fa ,
|
||||||
minimap2 needs to read
|
minimap2 needs to read
|
||||||
.I query.fa
|
.I query.fa
|
||||||
multiple times to map it against each batch of target sequences.
|
multiple times to map it against each batch of target sequences. This would create a multi-part index.
|
||||||
.I NUM
|
.I NUM
|
||||||
may be ending with k/K/m/M/g/G. NB: mapping quality is incorrect given a
|
may be ending with k/K/m/M/g/G. NB: mapping quality is incorrect given a
|
||||||
multi-part index.
|
multi-part index. See also option
|
||||||
|
.BR --split-prefix .
|
||||||
.TP
|
.TP
|
||||||
.B --idx-no-seq
|
.B --idx-no-seq
|
||||||
Don't store target sequences in the index. It saves disk space and memory but
|
Don't store target sequences in the index. It saves disk space and memory but
|
||||||
@@ -151,10 +152,16 @@ Lower and upper bounds of k-mer occurrences [10,1000000]. The final k-mer occurr
|
|||||||
.BR -f }}.
|
.BR -f }}.
|
||||||
This option prevents excessively small or large
|
This option prevents excessively small or large
|
||||||
.B -f
|
.B -f
|
||||||
estimated from the input reference. It deprecates
|
estimated from the input reference. Available since r1034 and deprecating
|
||||||
.B --min-occ-floor
|
.B --min-occ-floor
|
||||||
in earlier versions of minimap2.
|
in earlier versions of minimap2.
|
||||||
.TP
|
.TP
|
||||||
|
.BI --q-occ-frac \ FLOAT
|
||||||
|
Discard a query minimizer if its occurrence is higher than
|
||||||
|
.I FLOAT
|
||||||
|
fraction of query minimizers and than the reference occurrence threshold
|
||||||
|
[0.01]. Set 0 to disable. Available since r1105.
|
||||||
|
.TP
|
||||||
.BI -e \ INT
|
.BI -e \ INT
|
||||||
Sample a high-frequency minimizer every
|
Sample a high-frequency minimizer every
|
||||||
.I INT
|
.I INT
|
||||||
@@ -248,6 +255,11 @@ or more of the shorter chain [0.5]
|
|||||||
Use the minigraph chaining algorithm [no]. The minigraph algorithm is better
|
Use the minigraph chaining algorithm [no]. The minigraph algorithm is better
|
||||||
for aligning contigs through long INDELs.
|
for aligning contigs through long INDELs.
|
||||||
.TP
|
.TP
|
||||||
|
.BI --rmq-inner \ NUM
|
||||||
|
Apply full dynamic programming for anchors within distance
|
||||||
|
.I NUM
|
||||||
|
[1000].
|
||||||
|
.TP
|
||||||
.B --hard-mask-level
|
.B --hard-mask-level
|
||||||
Honor option
|
Honor option
|
||||||
.B -M
|
.B -M
|
||||||
@@ -285,11 +297,13 @@ maximum alignment gap is mostly controlled by
|
|||||||
.B --splice
|
.B --splice
|
||||||
Enable the splice alignment mode.
|
Enable the splice alignment mode.
|
||||||
.TP
|
.TP
|
||||||
.B --sr
|
.BR --sr [= no | dna | rna ]
|
||||||
Enable short-read alignment heuristics. In the short-read mode, minimap2
|
Enable short-read alignment heuristics [no]. If this option is used with no argument,
|
||||||
applies a second round of chaining with a higher minimizer occurrence threshold
|
.RB ` dna '
|
||||||
if no good chain is found. In addition, minimap2 attempts to patch gaps between
|
is set. In the DNA short-read mode, minimap2 applies a second round of chaining
|
||||||
seeds with ungapped alignment.
|
with a higher minimizer occurrence threshold if no good chain is found. In
|
||||||
|
addition, minimap2 attempts to patch gaps between seeds with ungapped
|
||||||
|
alignment.
|
||||||
.TP
|
.TP
|
||||||
.BI --split-prefix \ STR
|
.BI --split-prefix \ STR
|
||||||
Prefix to create temporary files. Typically used for a multi-part index.
|
Prefix to create temporary files. Typically used for a multi-part index.
|
||||||
@@ -309,9 +323,8 @@ Only map to the reverse complement strand of the reference sequences.
|
|||||||
If yes, sort anchors with heap merge, instead of radix sort. Heap merge is
|
If yes, sort anchors with heap merge, instead of radix sort. Heap merge is
|
||||||
faster for short reads, but slower for long reads. [no]
|
faster for short reads, but slower for long reads. [no]
|
||||||
.TP
|
.TP
|
||||||
.B --no-pairing
|
.B --no-hash-name
|
||||||
Treat two reads in a pair as independent reads. The mate related fields in SAM
|
Produce the same alignment for identical sequences regardless of their sequence names.
|
||||||
are still properly populated.
|
|
||||||
.SS Alignment options
|
.SS Alignment options
|
||||||
.TP 10
|
.TP 10
|
||||||
.BI -A \ INT
|
.BI -A \ INT
|
||||||
@@ -320,6 +333,10 @@ Matching score [2]
|
|||||||
.BI -B \ INT
|
.BI -B \ INT
|
||||||
Mismatching penalty [4]
|
Mismatching penalty [4]
|
||||||
.TP
|
.TP
|
||||||
|
.BI -b \ INT
|
||||||
|
Mismatching penalty for transitions [same as
|
||||||
|
.BR -B ].
|
||||||
|
.TP
|
||||||
.BI -O \ INT1[,INT2]
|
.BI -O \ INT1[,INT2]
|
||||||
Gap open penalty [4,24]. If
|
Gap open penalty [4,24]. If
|
||||||
.I INT2
|
.I INT2
|
||||||
@@ -333,10 +350,28 @@ costs
|
|||||||
.RI min{ O1 + k * E1 , O2 + k * E2 }.
|
.RI min{ O1 + k * E1 , O2 + k * E2 }.
|
||||||
In the splice mode, the second gap penalties are not used.
|
In the splice mode, the second gap penalties are not used.
|
||||||
.TP
|
.TP
|
||||||
|
.BI -J \ INT
|
||||||
|
Splice model [1]. 0 for the original minimap2 splice model that always penalizes non-GT-AG splicing;
|
||||||
|
1 for the miniprot model that considers non-GT-AG. Option
|
||||||
|
.B -C
|
||||||
|
has no effect with the default
|
||||||
|
.BR -J1 .
|
||||||
|
.TP
|
||||||
|
.BR -j \ FILE
|
||||||
|
Junctions used to extend alignment towards ends of reads [].
|
||||||
|
.I FILE
|
||||||
|
can be gene annotations in the BED12 format (aka 12-column BED), or intron
|
||||||
|
positions in 5-column BED with the strand column required. BED12 file can be
|
||||||
|
converted from GTF/GFF3 with `paftools.js gff2bed anno.gtf'. This option is
|
||||||
|
intended for short RNA-seq reads, while
|
||||||
|
.B --junc-bed
|
||||||
|
for long noisy RNA-seq reads.
|
||||||
|
.TP
|
||||||
.BI -C \ INT
|
.BI -C \ INT
|
||||||
Cost for a non-canonical GT-AG splicing (effective with
|
Cost for a non-canonical GT-AG splicing (effective with
|
||||||
.BR --splice )
|
.B --splice
|
||||||
[0]
|
.BR -J0 )
|
||||||
|
[0].
|
||||||
.TP
|
.TP
|
||||||
.BI -z \ INT1[,INT2]
|
.BI -z \ INT1[,INT2]
|
||||||
Truncate an alignment if the running alignment score drops too quickly along
|
Truncate an alignment if the running alignment score drops too quickly along
|
||||||
@@ -373,7 +408,16 @@ no attempt to match GT-AG [n]
|
|||||||
Score bonus when alignment extends to the end of the query sequence [0].
|
Score bonus when alignment extends to the end of the query sequence [0].
|
||||||
.TP
|
.TP
|
||||||
.BI --score-N \ INT
|
.BI --score-N \ INT
|
||||||
Score of a mismatch involving ambiguous bases [1].
|
Penalty of a mismatch involving ambiguous bases [1].
|
||||||
|
.TP
|
||||||
|
.BR --pairing = strong | weak | no
|
||||||
|
How to pair paired-end reads [strong].
|
||||||
|
.RB ` no '
|
||||||
|
for aligning the two ends in a pair independently with no `properly paired' set.
|
||||||
|
.RB ` weak '
|
||||||
|
for aligning the two ends independently and then pairing the hits.
|
||||||
|
.RB ` strong '
|
||||||
|
for jointly aligning and pairing the two ends.
|
||||||
.TP
|
.TP
|
||||||
.BR --splice-flank = yes | no
|
.BR --splice-flank = yes | no
|
||||||
Assume the next base to a
|
Assume the next base to a
|
||||||
@@ -392,16 +436,49 @@ on SIRV data, please add
|
|||||||
.B --splice-flank=no
|
.B --splice-flank=no
|
||||||
to the command line.
|
to the command line.
|
||||||
.TP
|
.TP
|
||||||
|
.BR --spsc \ FILE
|
||||||
|
Splice scores []. Each line consists of five fields: 1) contig, 2) offset, 3) `+' or `-', 4) `D' or `A', and 5) score,
|
||||||
|
where offset is the number of bases before a splice junction, `D' indicates the
|
||||||
|
line corresponds to a donor site and `A' for an acceptor site.
|
||||||
|
A positive score suggests the junction is preferred and a negative score
|
||||||
|
suggests the junction is not preferred.
|
||||||
|
.TP
|
||||||
|
.BR --spsc0 \ INT
|
||||||
|
Penalty for positions not in
|
||||||
|
.I FILE
|
||||||
|
specified by
|
||||||
|
.B --spsc
|
||||||
|
[5]. Effective with
|
||||||
|
.B --spsc
|
||||||
|
but not
|
||||||
|
.BR --junc-bed .
|
||||||
|
.TP
|
||||||
|
.BR --spsc-scale \ FLOAT
|
||||||
|
Scale splice scores in
|
||||||
|
.B --spsc
|
||||||
|
by
|
||||||
|
.IR FLOAT
|
||||||
|
rounded to the nearest integer [0.7].
|
||||||
|
.TP
|
||||||
.BR --junc-bed \ FILE
|
.BR --junc-bed \ FILE
|
||||||
Gene annotations in the BED12 format (aka 12-column BED), or intron positions
|
Junctions to prefer during base alignment [].
|
||||||
in 5-column BED. With this option, minimap2 prefers splicing in annotations.
|
Same format as
|
||||||
BED12 file can be converted from GTF/GFF3 with `paftools.js gff2bed anno.gtf'
|
.BR -j .
|
||||||
[].
|
It is
|
||||||
|
.I NOT
|
||||||
|
recommended to apply this option to short RNA-seq reads. This would increase
|
||||||
|
run time with little improvement to junction accuracy.
|
||||||
.TP
|
.TP
|
||||||
.BR --junc-bonus \ INT
|
.BR --junc-bonus \ INT
|
||||||
Score bonus for a splice donor or acceptor found in annotation (effective with
|
Score bonus for a splice donor or acceptor found in annotation [9]. Effective with
|
||||||
.BR --junc-bed )
|
.B --junc-bed
|
||||||
[9].
|
but not
|
||||||
|
.BR --spsc .
|
||||||
|
.TP
|
||||||
|
.BR --jump-min-match \ INT
|
||||||
|
Minimum matching length to create a jump [3]. Equivalent to
|
||||||
|
.B STAR
|
||||||
|
.BR --alignSJDBoverhangMin .
|
||||||
.TP
|
.TP
|
||||||
.BI --end-seed-pen \ INT
|
.BI --end-seed-pen \ INT
|
||||||
Drop a terminal anchor if
|
Drop a terminal anchor if
|
||||||
@@ -427,7 +504,7 @@ Set 0 to disable [100m].
|
|||||||
.BI --cap-kalloc \ NUM
|
.BI --cap-kalloc \ NUM
|
||||||
Free thread-local kalloc memory reservoir if after the alignment the size of the reservoir above
|
Free thread-local kalloc memory reservoir if after the alignment the size of the reservoir above
|
||||||
.IR NUM .
|
.IR NUM .
|
||||||
Set 0 to disable [0].
|
Set 0 to disable [500m].
|
||||||
.SS Input/output options
|
.SS Input/output options
|
||||||
.TP 10
|
.TP 10
|
||||||
.B -a
|
.B -a
|
||||||
@@ -459,20 +536,13 @@ Copy input FASTA/Q comments to output.
|
|||||||
.B -c
|
.B -c
|
||||||
Generate CIGAR. In PAF, the CIGAR is written to the `cg' custom tag.
|
Generate CIGAR. In PAF, the CIGAR is written to the `cg' custom tag.
|
||||||
.TP
|
.TP
|
||||||
.BI --cs[= STR ]
|
.BR --cs [= short | long ]
|
||||||
Output the
|
Output the
|
||||||
.B cs
|
.B cs
|
||||||
tag.
|
tag.
|
||||||
.I STR
|
If no argument is given,
|
||||||
can be either
|
.RB ` short '
|
||||||
.I short
|
is set. [none]
|
||||||
or
|
|
||||||
.IR long .
|
|
||||||
If no
|
|
||||||
.I STR
|
|
||||||
is given,
|
|
||||||
.I short
|
|
||||||
is assumed. [none]
|
|
||||||
.TP
|
.TP
|
||||||
.B --MD
|
.B --MD
|
||||||
Output the MD tag (see the SAM spec).
|
Output the MD tag (see the SAM spec).
|
||||||
@@ -483,6 +553,29 @@ Output =/X CIGAR operators for sequence match/mismatch.
|
|||||||
.B -Y
|
.B -Y
|
||||||
In SAM output, use soft clipping for supplementary alignments.
|
In SAM output, use soft clipping for supplementary alignments.
|
||||||
.TP
|
.TP
|
||||||
|
.B --secondary-seq
|
||||||
|
In SAM output, show query sequences for secondary alignments.
|
||||||
|
.TP
|
||||||
|
.B --write-junc
|
||||||
|
Output splice junctions in 6-column BED: contig name, start, end,
|
||||||
|
read name, score and strand. Score is the sum of donor and acceptor scores,
|
||||||
|
where GT gets 3, GC gets 2 and AT gets 1 at donor sites,
|
||||||
|
while AG gets 3 and AC gets 1 at acceptor sites.
|
||||||
|
Alignments with mapping quality below 10 are ignored.
|
||||||
|
.TP
|
||||||
|
.BI --pass1 \ FILE
|
||||||
|
Junctions BED file outputted by
|
||||||
|
.B --write-junc
|
||||||
|
[]. Rows with scores lower than 5 are ignored. When both
|
||||||
|
.B -j
|
||||||
|
and
|
||||||
|
.B --pass1
|
||||||
|
are present, junctions in
|
||||||
|
.B -j
|
||||||
|
are preferred over in
|
||||||
|
.BR --pass1
|
||||||
|
when there is ambiguity.
|
||||||
|
.TP
|
||||||
.BI --seed \ INT
|
.BI --seed \ INT
|
||||||
Integer seed for randomizing equally best hits. Minimap2 hashes
|
Integer seed for randomizing equally best hits. Minimap2 hashes
|
||||||
.I INT
|
.I INT
|
||||||
@@ -543,42 +636,70 @@ are:
|
|||||||
Align noisy long reads of ~10% error rate to a reference genome. This is the
|
Align noisy long reads of ~10% error rate to a reference genome. This is the
|
||||||
default mode.
|
default mode.
|
||||||
.TP
|
.TP
|
||||||
|
.B lr:hq
|
||||||
|
Align accurate long reads (error rate <1%) to a reference genome
|
||||||
|
.RB ( -k19
|
||||||
|
.B -w19 -U50,500
|
||||||
|
.BR -g10k ).
|
||||||
|
This was recommended by ONT developers for recent Nanopore reads
|
||||||
|
produced with chemistry v14 that can reach ~99% in accuracy.
|
||||||
|
It was shown to work better for accurate Nanopore reads
|
||||||
|
than
|
||||||
|
.BR map-hifi .
|
||||||
|
.TP
|
||||||
.B map-hifi
|
.B map-hifi
|
||||||
Align PacBio high-fidelity (HiFi) reads to a reference genome
|
Align PacBio high-fidelity (HiFi) reads to a reference genome
|
||||||
.RB ( -k19
|
.RB ( -xlr:hq
|
||||||
.B -w19 -U50,500 -g10k -A1 -B4 -O6,26 -E2,1
|
.B -A1 -B4 -O6,26 -E2,1
|
||||||
.BR -s200 ).
|
.BR -s200 ).
|
||||||
|
It differs from
|
||||||
|
.B lr:hq
|
||||||
|
only in scoring. It has not been tested whether
|
||||||
|
.B lr:hq
|
||||||
|
would work better for PacBio HiFi reads.
|
||||||
.TP
|
.TP
|
||||||
.B map-pb
|
.B map-pb
|
||||||
Align older PacBio continuous long (CLR) reads to a reference genome
|
Align older PacBio continuous long (CLR) reads to a reference genome
|
||||||
.RB ( -Hk19 ).
|
.RB ( -Hk19 ).
|
||||||
|
Note that this data type is effectively deprecated by HiFi.
|
||||||
|
Unless you work on very old data, you probably want to use
|
||||||
|
.B map-hifi
|
||||||
|
or
|
||||||
|
.BR lr:hq .
|
||||||
|
.TP
|
||||||
|
.B map-iclr
|
||||||
|
Align Illumina Complete Long Reads (ICLR) to a reference genome
|
||||||
|
.RB ( -k19
|
||||||
|
.B -B6 -b4
|
||||||
|
.BR -O10,50 ).
|
||||||
|
This was recommended by Illumina developers.
|
||||||
.TP
|
.TP
|
||||||
.B asm5
|
.B asm5
|
||||||
Long assembly to reference mapping
|
Long assembly to reference mapping
|
||||||
.RB ( -k19
|
.RB ( -k19
|
||||||
.B -w19 -U50,500 --rmq -r100k -g10k -A1 -B19 -O39,81 -E3,1 -s200 -z200
|
.B -w19 -U50,500 --rmq -r1k,100k -g10k -A1 -B19 -O39,81 -E3,1 -s200 -z200
|
||||||
.BR -N50 ).
|
.BR -N50 ).
|
||||||
Typically, the alignment will not extend to regions with 5% or higher sequence
|
Typically, the alignment will not extend to regions with 5% or higher sequence
|
||||||
divergence. Only use this preset if the average divergence is far below 5%.
|
divergence. Use this preset if the average divergence is not much higher than 0.1%.
|
||||||
.TP
|
.TP
|
||||||
.B asm10
|
.B asm10
|
||||||
Long assembly to reference mapping
|
Long assembly to reference mapping
|
||||||
.RB ( -k19
|
.RB ( -k19
|
||||||
.B -w19 -U50,500 --rmq -r100k -g10k -A1 -B9 -O16,41 -E2,1 -s200 -z200
|
.B -w19 -U50,500 --rmq -r1k,100k -g10k -A1 -B9 -O16,41 -E2,1 -s200 -z200
|
||||||
.BR -N50 ).
|
.BR -N50 ).
|
||||||
Up to 10% sequence divergence.
|
Use this if the average divergence is around 1%.
|
||||||
.TP
|
.TP
|
||||||
.B asm20
|
.B asm20
|
||||||
Long assembly to reference mapping
|
Long assembly to reference mapping
|
||||||
.RB ( -k19
|
.RB ( -k19
|
||||||
.B -w10 -U50,500 --rmq -r100k -g10k -A1 -B4 -O6,26 -E2,1 -s200 -z200
|
.B -w10 -U50,500 --rmq -r1k,100k -g10k -A1 -B4 -O6,26 -E2,1 -s200 -z200
|
||||||
.BR -N50 ).
|
.BR -N50 ).
|
||||||
Up to 20% sequence divergence.
|
Use this if the average divergence is around several percent.
|
||||||
.TP
|
.TP
|
||||||
.B splice
|
.B splice
|
||||||
Long-read spliced alignment
|
Long-read spliced alignment
|
||||||
.RB ( -k15
|
.RB ( -k15
|
||||||
.B -w5 --splice -g2k -G200k -A1 -B2 -O2,32 -E1,0 -b0 -C9 -z200 -ub --junc-bonus=9 --cap-sw-mem=0
|
.B -w5 --splice -g2k -G200k -A1 -B2 -O2,32 -E1,0 -C9 -z200 -ub --junc-bonus=9 --cap-sw-mem=0
|
||||||
.BR --splice-flank=yes ).
|
.BR --splice-flank=yes ).
|
||||||
In the splice mode, 1) long deletions are taken as introns and represented as
|
In the splice mode, 1) long deletions are taken as introns and represented as
|
||||||
the
|
the
|
||||||
@@ -589,15 +710,21 @@ costs are different during chaining; 4) the computation of the
|
|||||||
tag ignores introns to demote hits to pseudogenes.
|
tag ignores introns to demote hits to pseudogenes.
|
||||||
.TP
|
.TP
|
||||||
.B splice:hq
|
.B splice:hq
|
||||||
Long-read splice alignment for PacBio CCS reads
|
Spliced alignment for accurate long RNA-seq reads such as PacBio iso-seq
|
||||||
.RB ( -xsplice
|
.RB ( -xsplice
|
||||||
.B -C5 -O6,24
|
.B -C5 -O6,24
|
||||||
.BR -B4 ).
|
.BR -B4 ).
|
||||||
.TP
|
.TP
|
||||||
|
.B splice:sr
|
||||||
|
Spliced alignment for short RNA-seq reads
|
||||||
|
.RB ( -xsplice:hq
|
||||||
|
.B --frag=yes -m25 -s40 -2K100m --heap-sort=yes --pairing=weak --sr=rna --min-dp-len=20
|
||||||
|
.BR --secondary=no ).
|
||||||
|
.TP
|
||||||
.B sr
|
.B sr
|
||||||
Short single-end reads without splicing
|
Short-read alignment without splicing
|
||||||
.RB ( -k21
|
.RB ( -k21
|
||||||
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -b0 -r100 -p.5 -N20 -f1000,5000 -n2 -m20
|
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -r100 -p.5 -N20 -f1000,5000 -n2 -m25
|
||||||
.B -s40 -g100 -2K50m --heap-sort=yes
|
.B -s40 -g100 -2K50m --heap-sort=yes
|
||||||
.BR --secondary=no ).
|
.BR --secondary=no ).
|
||||||
.TP
|
.TP
|
||||||
@@ -670,7 +797,7 @@ s2 i Chaining score of the best secondary chain
|
|||||||
NM i Total number of mismatches and gaps in the alignment
|
NM i Total number of mismatches and gaps in the alignment
|
||||||
MD Z To generate the ref sequence in the alignment
|
MD Z To generate the ref sequence in the alignment
|
||||||
AS i DP alignment score
|
AS i DP alignment score
|
||||||
SA Z List of other supplementary alignments
|
SA Z List of other supplementary alignments (with approximate CIGAR strings)
|
||||||
ms i DP score of the max scoring segment in the alignment
|
ms i DP score of the max scoring segment in the alignment
|
||||||
nn i Number of ambiguous bases in the alignment
|
nn i Number of ambiguous bases in the alignment
|
||||||
ts A Transcript strand (splice mode only)
|
ts A Transcript strand (splice mode only)
|
||||||
@@ -679,6 +806,7 @@ cs Z Difference string
|
|||||||
dv f Approximate per-base sequence divergence
|
dv f Approximate per-base sequence divergence
|
||||||
de f Gap-compressed per-base sequence divergence
|
de f Gap-compressed per-base sequence divergence
|
||||||
rl i Length of query regions harboring repetitive seeds
|
rl i Length of query regions harboring repetitive seeds
|
||||||
|
zd i Alignment broken due to Z-drop; bit 1: left broken; bit 2: right broken
|
||||||
.TE
|
.TE
|
||||||
|
|
||||||
.PP
|
.PP
|
||||||
|
|||||||
+2
-1
@@ -16,7 +16,8 @@ minimap2 -c test/MT-human.fa test/MT-orang.fa \
|
|||||||
| paftools.js liftover -l10000 - <(echo -e "MT_orang\t2000\t5000") # liftOver
|
| paftools.js liftover -l10000 - <(echo -e "MT_orang\t2000\t5000") # liftOver
|
||||||
# no test data for the following examples
|
# no test data for the following examples
|
||||||
paftools.js junceval -e anno.gtf splice.sam > out.txt # compare splice junctions to annotations
|
paftools.js junceval -e anno.gtf splice.sam > out.txt # compare splice junctions to annotations
|
||||||
paftools.js splice2bed anno.gtf > anno.bed # convert GTF/GFF3 to BED12
|
paftools.js splice2bed splice.sam > splice.bed # convert PAF/SAM to BED12
|
||||||
|
paftools.js gff2bed anno.gtf > anno.bed # convert GTF/GFF3 to BED12
|
||||||
```
|
```
|
||||||
|
|
||||||
## Table of Contents
|
## Table of Contents
|
||||||
|
|||||||
-335
@@ -1,335 +0,0 @@
|
|||||||
#!/usr/bin/env k8
|
|
||||||
|
|
||||||
var getopt = function(args, ostr) {
|
|
||||||
var oli; // option letter list index
|
|
||||||
if (typeof(getopt.place) == 'undefined')
|
|
||||||
getopt.ind = 0, getopt.arg = null, getopt.place = -1;
|
|
||||||
if (getopt.place == -1) { // update scanning pointer
|
|
||||||
if (getopt.ind >= args.length || args[getopt.ind].charAt(getopt.place = 0) != '-') {
|
|
||||||
getopt.place = -1;
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
if (getopt.place + 1 < args[getopt.ind].length && args[getopt.ind].charAt(++getopt.place) == '-') { // found "--"
|
|
||||||
++getopt.ind;
|
|
||||||
getopt.place = -1;
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
var optopt = args[getopt.ind].charAt(getopt.place++); // character checked for validity
|
|
||||||
if (optopt == ':' || (oli = ostr.indexOf(optopt)) < 0) {
|
|
||||||
if (optopt == '-') return null; // if the user didn't specify '-' as an option, assume it means null.
|
|
||||||
if (getopt.place < 0) ++getopt.ind;
|
|
||||||
return '?';
|
|
||||||
}
|
|
||||||
if (oli+1 >= ostr.length || ostr.charAt(++oli) != ':') { // don't need argument
|
|
||||||
getopt.arg = null;
|
|
||||||
if (getopt.place < 0 || getopt.place >= args[getopt.ind].length) ++getopt.ind, getopt.place = -1;
|
|
||||||
} else { // need an argument
|
|
||||||
if (getopt.place >= 0 && getopt.place < args[getopt.ind].length)
|
|
||||||
getopt.arg = args[getopt.ind].substr(getopt.place);
|
|
||||||
else if (args.length <= ++getopt.ind) { // no arg
|
|
||||||
getopt.place = -1;
|
|
||||||
if (ostr.length > 0 && ostr.charAt(0) == ':') return ':';
|
|
||||||
return '?';
|
|
||||||
} else getopt.arg = args[getopt.ind]; // white space
|
|
||||||
getopt.place = -1;
|
|
||||||
++getopt.ind;
|
|
||||||
}
|
|
||||||
return optopt;
|
|
||||||
}
|
|
||||||
|
|
||||||
function read_fastx(file, buf)
|
|
||||||
{
|
|
||||||
if (file.readline(buf) < 0) return null;
|
|
||||||
var m, line = buf.toString();
|
|
||||||
if ((m = /^([>@])(\S+)/.exec(line)) == null)
|
|
||||||
throw Error("wrong fastx format");
|
|
||||||
var is_fq = (m[1] == '@');
|
|
||||||
var name = m[2];
|
|
||||||
if (file.readline(buf) < 0)
|
|
||||||
throw Error("missing sequence line");
|
|
||||||
var seq = buf.toString();
|
|
||||||
if (is_fq) { // skip quality
|
|
||||||
file.readline(buf);
|
|
||||||
file.readline(buf);
|
|
||||||
}
|
|
||||||
return [name, seq];
|
|
||||||
}
|
|
||||||
|
|
||||||
function filter_paf(a, opt)
|
|
||||||
{
|
|
||||||
if (a.length == 0) return;
|
|
||||||
var k = 0;
|
|
||||||
for (var i = 0; i < a.length; ++i) {
|
|
||||||
var ai = a[i];
|
|
||||||
if (ai[10] < opt.min_blen) continue;
|
|
||||||
if (ai[9] < ai[10] * opt.min_iden) continue;
|
|
||||||
var clip = [0, 0];
|
|
||||||
if (ai[4] == '+') {
|
|
||||||
clip[0] = ai[2] < ai[7]? ai[2] : ai[7];
|
|
||||||
clip[1] = ai[1] - ai[3] < ai[6] - ai[8]? ai[1] - ai[3] : ai[6] - ai[8];
|
|
||||||
} else {
|
|
||||||
clip[0] = ai[2] < ai[6] - ai[8]? ai[2] : ai[6] - ai[8];
|
|
||||||
clip[1] = ai[1] - ai[3] < ai[7]? ai[1] - ai[3] : ai[7];
|
|
||||||
}
|
|
||||||
if (clip[0] > opt.max_clip_len || clip[1] > opt.max_clip_len) continue;
|
|
||||||
a[k++] = ai;
|
|
||||||
}
|
|
||||||
a.length = k;
|
|
||||||
}
|
|
||||||
|
|
||||||
function parse_events(t, ev, id, buf)
|
|
||||||
{
|
|
||||||
var re = /(:(\d+))|(([\+\-\*])([a-z]+))/g;
|
|
||||||
var m, cs = null;
|
|
||||||
for (var j = 12; j < t.length; ++j) {
|
|
||||||
if ((m = /^cs:Z:(\S+)/.exec(t[j])) != null) {
|
|
||||||
cs = m[1].toLowerCase();
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (cs == null) {
|
|
||||||
warn("Warning: no cs tag for read '" + t[0] + "'");
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
var st = t[2], en = t[3];
|
|
||||||
var x = st;
|
|
||||||
while ((m = re.exec(cs)) != null) {
|
|
||||||
var l;
|
|
||||||
if (m[2] != null) { // an identitcal match ":\d+"
|
|
||||||
l = parseInt(m[2]);
|
|
||||||
// [start, end, type, index, changed_base]
|
|
||||||
ev.push([x, x + l, 0, id]);
|
|
||||||
} else {
|
|
||||||
if (m[4] == '*') {
|
|
||||||
l = 1;
|
|
||||||
ev.push([x, x + 1, 1, id, m[5][0]]);
|
|
||||||
} else if (m[4] == '+') {
|
|
||||||
l = m[5].length;
|
|
||||||
ev.push([x, x + l, 2, id]);
|
|
||||||
} else if (m[4] == '-') {
|
|
||||||
l = 0;
|
|
||||||
ev.push([x, x, -1, id, m[5]]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
x += l;
|
|
||||||
}
|
|
||||||
if (x != en)
|
|
||||||
throw Error("inconsistent cs for read '" + t[0] + "'");
|
|
||||||
}
|
|
||||||
|
|
||||||
function find_het_sub(ev, a, opt)
|
|
||||||
{
|
|
||||||
var n = a.length, last0_i = -1, h = [], d = [];
|
|
||||||
for (var i = 0; i < n; ++i) h[i] = [], d[i] = [];
|
|
||||||
for (var i = 0; i < ev.length; ++i) {
|
|
||||||
if (ev[i][2] == 0) {
|
|
||||||
if (last0_i < 0 || ev[i][0] != ev[last0_i][0]) last0_i = i;
|
|
||||||
else if (ev[i][1] > ev[last0_i][1])
|
|
||||||
last0_i = i;
|
|
||||||
} else if (ev[i][2] == 1 && last0_i >= 0 && ev[i][0] < ev[last0_i][1]) {
|
|
||||||
if (ev[last0_i][1] - ev[last0_i][0] >= opt.min_mlen) {
|
|
||||||
if (opt.dbg_ev) print("EV", ev[last0_i].join("\t"), "|", ev[i].join("\t"));
|
|
||||||
var e0 = ev[last0_i], hl = h[e0[3]];
|
|
||||||
if (hl.length == 0 || hl[hl.length-1][0] != e0[0])
|
|
||||||
hl.push([e0[0], e0[1]]);
|
|
||||||
d[ev[i][3]].push([ev[i][0], e0[1] - e0[0]]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
var b = [];
|
|
||||||
for (var i = 0; i < n; ++i) {
|
|
||||||
var sh = 0, dh = 0;
|
|
||||||
for (var j = 0; j < h[i].length; ++j)
|
|
||||||
sh += h[i][j][1] - h[i][j][0];
|
|
||||||
for (var j = 0; j < d[i].length; ++j)
|
|
||||||
dh += d[i][j][1];
|
|
||||||
// [start, end, index, #consistent, lenConsistent, #conflictive, lenConflictive, identity, mlen]
|
|
||||||
b[i] = [a[i][2], a[i][3], i, h[i].length, sh, d[i].length, dh, a[i][9] / a[i][10], a[i][9]];
|
|
||||||
}
|
|
||||||
return b;
|
|
||||||
}
|
|
||||||
|
|
||||||
function flt_utg_for_ec(b, opt)
|
|
||||||
{
|
|
||||||
var k = 0;
|
|
||||||
for (var i = 0; i < b.length; ++i) {
|
|
||||||
var bi = b[i];
|
|
||||||
if (bi[4] == 0 && bi[6] == 0) b[k++] = bi; // entirely ambiguous
|
|
||||||
else if (bi[6] < (bi[4] + bi[6]) * opt.max_ratio0) b[k++] = bi;
|
|
||||||
}
|
|
||||||
b.length = k;
|
|
||||||
if (b.length == 0) return;
|
|
||||||
// find the longest contiguous segment
|
|
||||||
b.sort(function(x,y) { return x[0]-y[0] });
|
|
||||||
var st = b[0][0], en = b[0][1], max_st = 0, max_en = 0, max_max_en = en;
|
|
||||||
for (var i = 1; i < b.length; ++i) {
|
|
||||||
if (b[i][0] > en) {
|
|
||||||
if (en - st > max_en - max_st)
|
|
||||||
max_st = st, max_en = en;
|
|
||||||
st = b[i][0], en = b[i][1];
|
|
||||||
} else {
|
|
||||||
en = en > b[i][1]? en : b[i][1];
|
|
||||||
}
|
|
||||||
max_max_en = max_max_en > b[i][1]? max_max_en : b[i][1];
|
|
||||||
}
|
|
||||||
if (en - st > max_en - max_st)
|
|
||||||
max_st = st, max_en = en;
|
|
||||||
if (max_max_en != en || st != b[0][0]) {
|
|
||||||
var k = 0;
|
|
||||||
for (var i = 0; i < b.length; ++i)
|
|
||||||
if (b[i][0] < max_en && b[i][1] > max_st)
|
|
||||||
b[k++] = b[i];
|
|
||||||
b.length = k;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function flt_utg_for_bin(b, opt) // filter out alignments clearly on the wrong phase
|
|
||||||
{
|
|
||||||
var k = 0;
|
|
||||||
for (var i = 0; i < b.length; ++i) {
|
|
||||||
var bi = b[i];
|
|
||||||
if (bi[4] + bi[6] == 0 || bi[4] >= (bi[4] + bi[6]) * opt.max_ratio0) b[k++] = bi;
|
|
||||||
}
|
|
||||||
b.length = k;
|
|
||||||
}
|
|
||||||
|
|
||||||
function ec_core(b, n_a, ev, buf, ecb) // error correction
|
|
||||||
{
|
|
||||||
var intv = [];
|
|
||||||
for (var i = 0; i < n_a; ++i)
|
|
||||||
intv[i] = null;
|
|
||||||
intv[b[0][2]] = [b[0][0], b[0][1]];
|
|
||||||
var en = b[0][1];
|
|
||||||
for (var i = 1; i < b.length; ++i) {
|
|
||||||
if (b[i][1] <= en) continue;
|
|
||||||
intv[b[i][2]] = [en, b[i][1]];
|
|
||||||
en = b[i][1];
|
|
||||||
}
|
|
||||||
var k = 0;
|
|
||||||
ecb.capacity = buf.capacity;
|
|
||||||
ecb.length = 0;
|
|
||||||
for (var i = 0; i < ev.length; ++i) {
|
|
||||||
var e = ev[i], I = intv[e[3]];
|
|
||||||
if (I == null) continue;
|
|
||||||
if (e[0] >= I[0] && e[0] < I[1]) { // this is to reduce duplicated events around junctions
|
|
||||||
//print("X", e.join("\t"));
|
|
||||||
if (e[2] == 0) {
|
|
||||||
ecb.length += e[1] - e[0];
|
|
||||||
for (var j = e[0]; j < e[1]; ++j)
|
|
||||||
ecb[k++] = buf[j];
|
|
||||||
} else if (e[2] == 1) {
|
|
||||||
++ecb.length;
|
|
||||||
ecb[k++] = e[4].charCodeAt(0);
|
|
||||||
} else if (e[2] < 0) {
|
|
||||||
ecb.length += e[4].length;
|
|
||||||
for (var j = 0; j < e[4].length; ++j)
|
|
||||||
ecb[k++] = e[4].charCodeAt(j);
|
|
||||||
} // else, skip e[2] == 2
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (ecb.length != k) throw Error("BUG!");
|
|
||||||
}
|
|
||||||
|
|
||||||
function process_paf(a, opt, fp_seq, buf, ecb)
|
|
||||||
{
|
|
||||||
if (a.length == 0) return;
|
|
||||||
var len = a[0][1], name = a[0][0], seq = null;
|
|
||||||
if (len < opt.min_rlen) return;
|
|
||||||
if (fp_seq) {
|
|
||||||
var ret;
|
|
||||||
while ((ret = read_fastx(fp_seq, buf)) != null)
|
|
||||||
if (ret[0] == a[0][0])
|
|
||||||
break;
|
|
||||||
if (ret == null)
|
|
||||||
throw Error("failed to find sequence for read '" + a[0][0] + "'");
|
|
||||||
name = ret[0], seq = ret[1];
|
|
||||||
if (seq.length != len)
|
|
||||||
throw Error("inconsistent length for read '" + name + "'");
|
|
||||||
}
|
|
||||||
filter_paf(a, opt);
|
|
||||||
if (a.length == 0) return;
|
|
||||||
var ev = [];
|
|
||||||
for (var i = 0; i < a.length; ++i)
|
|
||||||
parse_events(a[i], ev, i, buf);
|
|
||||||
ev.sort(function(x,y) { return x[0]!=y[0]? x[0]-y[0] : x[2]-y[2] });
|
|
||||||
if (seq == null) print("SQ", name, a[0][1], a.length);
|
|
||||||
var b = find_het_sub(ev, a, opt);
|
|
||||||
if (opt.ec) flt_utg_for_ec(b, opt);
|
|
||||||
else flt_utg_for_bin(b, opt);
|
|
||||||
if (seq == null) {
|
|
||||||
for (var i = 0; i < b.length; ++i) {
|
|
||||||
var m, ai = a[b[i][2]], score = 0;
|
|
||||||
for (var j = 10; j < ai.length; ++j)
|
|
||||||
if ((m = /^AS:i:(\d+)/.exec(ai[j])) != null)
|
|
||||||
score = m[1];
|
|
||||||
print("TS", b[i][2], b[i][0], b[i][1], ai.slice(5, 9).join("\t"), b[i].slice(3, 7).join("\t"), score);
|
|
||||||
}
|
|
||||||
print("//");
|
|
||||||
} else { // error correction
|
|
||||||
if (b.length == 0) return;
|
|
||||||
buf.set(seq, 0);
|
|
||||||
ec_core(b, a.length, ev, buf, ecb);
|
|
||||||
print(">" + name);
|
|
||||||
print(ecb);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function main(args)
|
|
||||||
{
|
|
||||||
var c, opt = { min_rlen:5000, min_blen:5000, min_iden:0.8, min_mlen:5, max_clip_len:500, max_ratio0:0.25, dbg_ev:false };
|
|
||||||
while ((c = getopt(args, "l:b:d:m:c:r:E")) != null) {
|
|
||||||
if (c == 'l') opt.min_rlen = parseInt(getopt.arg);
|
|
||||||
else if (c == 'b') opt.min_blen = parseInt(getopt.arg);
|
|
||||||
else if (c == 'd') opt.min_iden = parseFloat(getopt.arg);
|
|
||||||
else if (c == 'm') opt.min_slen = parseInt(getopt.arg);
|
|
||||||
else if (c == 'c') opt.max_clip_len = parseInt(getopt.arg);
|
|
||||||
else if (c == 'r') opt.max_ratio0 = parseFloat(getopt.arg);
|
|
||||||
else if (c == 'E') opt.dbg_ev = true;
|
|
||||||
}
|
|
||||||
if (args.length - getopt.ind < 1) {
|
|
||||||
print("Usage: mmphase.js [options] <map-with-cs.paf> [reads.fa]");
|
|
||||||
print("Options:");
|
|
||||||
print(" -l INT min read length [" + opt.min_rlen + "]");
|
|
||||||
print(" -b INT min alignment length [" + opt.min_blen + "]");
|
|
||||||
print(" -d FLOAT min identity [" + opt.min_iden + "]");
|
|
||||||
print(" -s INT min match length [" + opt.min_mlen + "]");
|
|
||||||
print(" -c INT max clip length [" + opt.max_clip_len + "]");
|
|
||||||
print(" -r FLOAT initial ratio for haplotype filtering [" + opt.max_ratio0 + "]");
|
|
||||||
return 0;
|
|
||||||
}
|
|
||||||
|
|
||||||
opt.ec = args.length - getopt.ind < 2? false : true;
|
|
||||||
if (!opt.ec) {
|
|
||||||
print("CC");
|
|
||||||
print("CC", "SQ qName qLen nHits");
|
|
||||||
print("CC", "TS index qStart qEnd tName tLen tStart tEnd nConsistent lCons nConflictive lConf score");
|
|
||||||
print("CC");
|
|
||||||
}
|
|
||||||
|
|
||||||
var buf = new Bytes(), ecb = new Bytes();
|
|
||||||
var fp_paf = new File(args[getopt.ind]);
|
|
||||||
var fp_seq = args.length - getopt.ind >= 2? new File(args[getopt.ind+1]) : null;
|
|
||||||
var a = [];
|
|
||||||
while (fp_paf.readline(buf) >= 0) {
|
|
||||||
var t = buf.toString().split("\t");
|
|
||||||
if (a.length > 0 && a[0][0] != t[0]) {
|
|
||||||
process_paf(a, opt, fp_seq, buf, ecb);
|
|
||||||
a.length = 0;
|
|
||||||
}
|
|
||||||
for (var i = 1; i <= 3; ++i) t[i] = parseInt(t[i]);
|
|
||||||
if (t[1] < opt.min_rlen) continue;
|
|
||||||
for (var i = 6; i <= 10; ++i) t[i] = parseInt(t[i]);
|
|
||||||
if (t[10] < opt.min_blen) continue;
|
|
||||||
a.push(t);
|
|
||||||
}
|
|
||||||
if (a.length >= 0)
|
|
||||||
process_paf(a, opt, fp_seq, buf, ecb);
|
|
||||||
if (fp_seq) fp_seq.close();
|
|
||||||
fp_paf.close();
|
|
||||||
ecb.destroy();
|
|
||||||
buf.destroy();
|
|
||||||
}
|
|
||||||
|
|
||||||
var ret = main(arguments)
|
|
||||||
exit(ret)
|
|
||||||
Executable
+241
@@ -0,0 +1,241 @@
|
|||||||
|
#!/usr/bin/env k8
|
||||||
|
|
||||||
|
"use strict";
|
||||||
|
|
||||||
|
Array.prototype.delete_at = function(i) {
|
||||||
|
for (let j = i; j < this.length - 1; ++j)
|
||||||
|
this[j] = this[j + 1];
|
||||||
|
--this.length;
|
||||||
|
}
|
||||||
|
|
||||||
|
function* getopt(argv, ostr, longopts) {
|
||||||
|
if (argv.length == 0) return;
|
||||||
|
let pos = 0, cur = 0;
|
||||||
|
while (cur < argv.length) {
|
||||||
|
let lopt = "", opt = "?", arg = "";
|
||||||
|
while (cur < argv.length) { // skip non-option arguments
|
||||||
|
if (argv[cur][0] == "-" && argv[cur].length > 1) {
|
||||||
|
if (argv[cur] == "--") cur = argv.length;
|
||||||
|
break;
|
||||||
|
} else ++cur;
|
||||||
|
}
|
||||||
|
if (cur == argv.length) break;
|
||||||
|
let a = argv[cur];
|
||||||
|
if (a[0] == "-" && a[1] == "-") { // a long option
|
||||||
|
pos = -1;
|
||||||
|
let c = 0, k = -1, tmp = "", o;
|
||||||
|
const pos_eq = a.indexOf("=");
|
||||||
|
if (pos_eq > 0) {
|
||||||
|
o = a.substring(2, pos_eq);
|
||||||
|
arg = a.substring(pos_eq + 1);
|
||||||
|
} else o = a.substring(2);
|
||||||
|
for (let i = 0; i < longopts.length; ++i) {
|
||||||
|
let y = longopts[i];
|
||||||
|
if (y[y.length - 1] == "=") y = y.substring(0, y.length - 1);
|
||||||
|
if (o.length <= y.length && o == y.substring(0, o.length)) {
|
||||||
|
k = i, tmp = y;
|
||||||
|
++c; // c is the number of matches
|
||||||
|
if (o == y) { // exact match
|
||||||
|
c = 1;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (c == 1) { // find a unique match
|
||||||
|
lopt = tmp;
|
||||||
|
if (pos_eq < 0 && longopts[k][longopts[k].length-1] == "=" && cur + 1 < argv.length) {
|
||||||
|
arg = argv[cur+1];
|
||||||
|
argv.delete_at(cur + 1);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
} else { // a short option
|
||||||
|
if (pos == 0) pos = 1;
|
||||||
|
opt = a[pos++];
|
||||||
|
let k = ostr.indexOf(opt);
|
||||||
|
if (k < 0) {
|
||||||
|
opt = "?";
|
||||||
|
} else if (k + 1 < ostr.length && ostr[k+1] == ":") { // requiring an argument
|
||||||
|
if (pos >= a.length) {
|
||||||
|
arg = argv[cur+1];
|
||||||
|
argv.delete_at(cur + 1);
|
||||||
|
} else arg = a.substring(pos);
|
||||||
|
pos = -1;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (pos < 0 || pos >= argv[cur].length) {
|
||||||
|
argv.delete_at(cur);
|
||||||
|
pos = 0;
|
||||||
|
}
|
||||||
|
if (lopt != "") yield { opt: `--${lopt}`, arg: arg };
|
||||||
|
else if (opt != "?") yield { opt: `-${opt}`, arg: arg };
|
||||||
|
else yield { opt: "?", arg: "" };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function* k8_readline(fn) {
|
||||||
|
let buf = new Bytes();
|
||||||
|
let file = new File(fn);
|
||||||
|
while (file.readline(buf) >= 0) {
|
||||||
|
yield buf.toString();
|
||||||
|
}
|
||||||
|
file.close();
|
||||||
|
buf.destroy();
|
||||||
|
}
|
||||||
|
|
||||||
|
function merge_hits(b) {
|
||||||
|
if (b.length == 1)
|
||||||
|
return { name1:b[0].name1, name2:b[0].name2, len1:b[0].len1, len2:b[0].len2, min_cov:b[0].min_cov, max_cov:b[0].max_cov, cov1:b[0].cov1, cov2:b[0].cov2, s1:b[0].s1, dv:b[0].dv };
|
||||||
|
b.sort(function(x, y) { return x.st1 - y.st1 });
|
||||||
|
let f = [], bt = [];
|
||||||
|
for (let i = 0; i < b.length; ++i)
|
||||||
|
f[i] = b[i].s1, bt[i] = -1;
|
||||||
|
for (let i = 0; i < b.length; ++i) {
|
||||||
|
for (let j = 0; j < i; ++j) {
|
||||||
|
if (b[j].st2 < b[i].st2) {
|
||||||
|
if (b[j].en1 >= b[i].en1) continue;
|
||||||
|
if (b[j].en2 >= b[i].en2) continue;
|
||||||
|
const ov1 = b[j].en1 <= b[i].st1? 0 : b[i].st1 - b[j].en1;
|
||||||
|
const li1 = b[i].en1 - b[i].st1;
|
||||||
|
const s11 = b[i].s1 / li1 * (li1 - ov1);
|
||||||
|
const ov2 = b[j].en2 <= b[i].st2? 0 : b[i].st2 - b[j].en2;
|
||||||
|
const li2 = b[i].en2 - b[i].st2;
|
||||||
|
const s12 = b[i].s1 / li2 * (li2 - ov2);
|
||||||
|
const s1 = s11 < s12? s11 : s12;
|
||||||
|
if (f[i] < f[j] + s1)
|
||||||
|
f[i] = f[j] + s1, bt[i] = j;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
let max_i = -1, max_f = 0, d = [];
|
||||||
|
for (let i = 0; i < b.length; ++i)
|
||||||
|
if (max_f < f[i])
|
||||||
|
max_f = f[i], max_i = i;
|
||||||
|
for (let k = max_i; k >= 0; k = bt[k])
|
||||||
|
d.push(k);
|
||||||
|
d = d.reverse();
|
||||||
|
let dv = 0, tot = 0, cov1 = 0, cov2 = 0, st1 = 0, en1 = 0, st2 = 0, en2 = 0;
|
||||||
|
for (let k = 0; k < d.length; ++k) {
|
||||||
|
const i = d[k];
|
||||||
|
tot += b[i].blen;
|
||||||
|
dv += b[i].dv * b[i].blen;
|
||||||
|
if (b[i].st1 > en1) {
|
||||||
|
cov1 += en1 - st1;
|
||||||
|
st1 = b[i].st1, en1 = b[i].en1;
|
||||||
|
} else en1 = en1 > b[i].en1? en1 : b[i].en1;
|
||||||
|
if (b[i].st2 > en2) {
|
||||||
|
cov2 += en2 - st2;
|
||||||
|
st2 = b[i].st2, en2 = b[i].en2;
|
||||||
|
} else en2 = en2 > b[i].en2? en2 : b[i].en2;
|
||||||
|
}
|
||||||
|
dv /= tot;
|
||||||
|
cov1 = (cov1 + (en1 - st1)) / b[0].len1;
|
||||||
|
cov2 = (cov2 + (en2 - st2)) / b[0].len2;
|
||||||
|
const min_cov = cov1 < cov2? cov1 : cov2;
|
||||||
|
const max_cov = cov1 > cov2? cov1 : cov2;
|
||||||
|
//warn(d.length, b[0].name1, b[0].name2, min_cov, max_cov);
|
||||||
|
return { name1:b[0].name1, name2:b[0].name2, len1:b[0].len1, len2:b[0].len2, min_cov:min_cov, max_cov:max_cov, cov1:cov1, cov2:cov2, s1:max_f, dv:dv };
|
||||||
|
}
|
||||||
|
|
||||||
|
function main(args) {
|
||||||
|
let opt = { min_cov:.9, max_dv:.015, max_diff:20000 };
|
||||||
|
for (const o of getopt(args, "c:d:e:", [])) {
|
||||||
|
if (o.opt == '-c') opt.min_cov = parseFloat(o.arg);
|
||||||
|
else if (o.opt == '-d') opt.max_dv = parseFloat(o.arg);
|
||||||
|
else if (o.opt == '-e') opt.max_diff = parseFloat(o.arg);
|
||||||
|
}
|
||||||
|
if (args.length == 0) {
|
||||||
|
print("Usage: pafcluster.js [options] <ava.paf>");
|
||||||
|
print("Options:");
|
||||||
|
print(` -c FLOAT min coverage [${opt.min_cov}]`);
|
||||||
|
print(` -d FLOAT max divergence [${opt.max_dv}]`);
|
||||||
|
print(` -e FLOAT max difference [${opt.max_diff}]`);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// read
|
||||||
|
let a = [], len = {}, name2len = {};
|
||||||
|
for (const line of k8_readline(args[0])) {
|
||||||
|
let m, t = line.split("\t");
|
||||||
|
if (t[4] != "+") continue;
|
||||||
|
for (let i = 1; i < 4; ++i) t[i] = parseInt(t[i]);
|
||||||
|
for (let i = 6; i < 11; ++i) t[i] = parseInt(t[i]);
|
||||||
|
const len1 = t[1], len2 = t[6];
|
||||||
|
let s1 = -1, dv = -1.0;
|
||||||
|
for (let i = 12; i < t.length; ++i) {
|
||||||
|
if ((m = /^(s1|dv):\S:(\S+)/.exec(t[i])) != null) {
|
||||||
|
if (m[1] == "s1") s1 = parseInt(m[2]);
|
||||||
|
else if (m[1] == "dv") dv = parseFloat(m[2]);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (s1 < 0 || dv < 0) continue;
|
||||||
|
const cov1 = (parseInt(t[3]) - parseInt(t[2])) / len1;
|
||||||
|
const cov2 = (parseInt(t[8]) - parseInt(t[7])) / len2;
|
||||||
|
const min_cov = cov1 < cov2? cov1 : cov2;
|
||||||
|
const max_cov = cov1 > cov2? cov1 : cov2;
|
||||||
|
name2len[t[0]] = len1;
|
||||||
|
name2len[t[5]] = len2;
|
||||||
|
a.push({ name1:t[0], name2:t[5], len1:len1, len2:len2, min_cov:min_cov, max_cov:max_cov, s1:s1, dv:dv, cov1:cov1, cov2:cov2, st1:t[2], en1:t[3], st2:t[7], en2:t[8], blen:t[10] });
|
||||||
|
len[t[0]] = len1, len[t[5]] = len2;
|
||||||
|
}
|
||||||
|
warn(`Read ${a.length} hits`);
|
||||||
|
|
||||||
|
// merge duplicated hits
|
||||||
|
let h = {};
|
||||||
|
for (let i = 0; i < a.length; ++i) {
|
||||||
|
const key = `${a[i].name1}\t${a[i].name2}`;
|
||||||
|
if (h[key] == null) h[key] = [];
|
||||||
|
h[key].push(a[i]);
|
||||||
|
}
|
||||||
|
a = [];
|
||||||
|
for (const key in h)
|
||||||
|
a.push(merge_hits(h[key]));
|
||||||
|
|
||||||
|
// core loop
|
||||||
|
while (a.length > 1) {
|
||||||
|
// select the sequence with the highest sum of s1
|
||||||
|
let h = {};
|
||||||
|
for (let i = 0; i < a.length; ++i) {
|
||||||
|
if (h[a[i].name1] == null) h[a[i].name1] = 0;
|
||||||
|
h[a[i].name1] += a[i].s1;
|
||||||
|
}
|
||||||
|
let max_s1 = 0, max_name = "";
|
||||||
|
for (const name in h)
|
||||||
|
if (max_s1 < h[name])
|
||||||
|
max_s1 = h[name], max_name = name;
|
||||||
|
// find contigs in the same group
|
||||||
|
h = {};
|
||||||
|
h[max_name] = 1;
|
||||||
|
for (let i = 0; i < a.length; ++i) {
|
||||||
|
if (a[i].name1 != max_name && a[i].name2 != max_name)
|
||||||
|
continue;
|
||||||
|
const diff1 = a[i].len1 * (1.0 - a[i].cov1);
|
||||||
|
const diff2 = a[i].len2 * (1.0 - a[i].cov2);
|
||||||
|
if (a[i].min_cov >= opt.min_cov && a[i].dv <= opt.max_dv && diff1 <= opt.max_diff && diff2 <= opt.max_diff)
|
||||||
|
h[a[i].name1] = h[a[i].name2] = 1;
|
||||||
|
}
|
||||||
|
let n = 0;
|
||||||
|
for (const key in h) {
|
||||||
|
++n;
|
||||||
|
delete name2len[key];
|
||||||
|
}
|
||||||
|
print(`SD\t${max_name}\t${n}`);
|
||||||
|
for (const key in h) print(`CL\t${key}\t${len[key]}`);
|
||||||
|
print("//");
|
||||||
|
// filter out redundant hits
|
||||||
|
let b = [];
|
||||||
|
for (let i = 0; i < a.length; ++i)
|
||||||
|
if (h[a[i].name1] == null && h[a[i].name2] == null)
|
||||||
|
b.push(a[i]);
|
||||||
|
warn(`Reduced the number of hits from ${a.length} to ${b.length}`);
|
||||||
|
a = b;
|
||||||
|
}
|
||||||
|
|
||||||
|
// output remaining singletons
|
||||||
|
for (const key in name2len) {
|
||||||
|
print(`SD\t${key}\t1`);
|
||||||
|
print(`CL\t${key}\t${name2len[key]}`);
|
||||||
|
print(`//`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
main(arguments);
|
||||||
+767
-63
File diff suppressed because it is too large
Load Diff
@@ -13,6 +13,8 @@
|
|||||||
#define MM_DBG_PRINT_QNAME 0x2
|
#define MM_DBG_PRINT_QNAME 0x2
|
||||||
#define MM_DBG_PRINT_SEED 0x4
|
#define MM_DBG_PRINT_SEED 0x4
|
||||||
#define MM_DBG_PRINT_ALN_SEQ 0x8
|
#define MM_DBG_PRINT_ALN_SEQ 0x8
|
||||||
|
#define MM_DBG_PRINT_CHAIN 0x10
|
||||||
|
#define MM_DBG_SEED_FREQ 0x20
|
||||||
|
|
||||||
#define MM_SEED_LONG_JOIN (1ULL<<40)
|
#define MM_SEED_LONG_JOIN (1ULL<<40)
|
||||||
#define MM_SEED_IGNORE (1ULL<<41)
|
#define MM_SEED_IGNORE (1ULL<<41)
|
||||||
@@ -22,6 +24,9 @@
|
|||||||
#define MM_SEED_SEG_SHIFT 48
|
#define MM_SEED_SEG_SHIFT 48
|
||||||
#define MM_SEED_SEG_MASK (0xffULL<<(MM_SEED_SEG_SHIFT))
|
#define MM_SEED_SEG_MASK (0xffULL<<(MM_SEED_SEG_SHIFT))
|
||||||
|
|
||||||
|
#define MM_JUNC_ANNO 0x1
|
||||||
|
#define MM_JUNC_MISC 0x2
|
||||||
|
|
||||||
#ifndef kroundup32
|
#ifndef kroundup32
|
||||||
#define kroundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
#define kroundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
||||||
#endif
|
#endif
|
||||||
@@ -31,6 +36,7 @@
|
|||||||
|
|
||||||
#define MALLOC(type, len) ((type*)malloc((len) * sizeof(type)))
|
#define MALLOC(type, len) ((type*)malloc((len) * sizeof(type)))
|
||||||
#define CALLOC(type, len) ((type*)calloc((len), sizeof(type)))
|
#define CALLOC(type, len) ((type*)calloc((len), sizeof(type)))
|
||||||
|
#define REALLOC(type, ptr, cnt) ((type*)realloc((ptr), (cnt) * sizeof(type)))
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
extern "C" {
|
extern "C" {
|
||||||
@@ -50,6 +56,12 @@ typedef struct {
|
|||||||
mm128_t *a;
|
mm128_t *a;
|
||||||
} mm_seg_t;
|
} mm_seg_t;
|
||||||
|
|
||||||
|
typedef struct {
|
||||||
|
int32_t off, off2, cnt;
|
||||||
|
int16_t strand;
|
||||||
|
uint16_t flag;
|
||||||
|
} mm_idx_jjump1_t;
|
||||||
|
|
||||||
double cputime(void);
|
double cputime(void);
|
||||||
double realtime(void);
|
double realtime(void);
|
||||||
long peakrss(void);
|
long peakrss(void);
|
||||||
@@ -61,24 +73,29 @@ uint32_t ks_ksmall_uint32_t(size_t n, uint32_t arr[], size_t kk);
|
|||||||
void mm_sketch(void *km, const char *str, int len, int w, int k, uint32_t rid, int is_hpc, mm128_v *p);
|
void mm_sketch(void *km, const char *str, int len, int w, int k, uint32_t rid, int is_hpc, mm128_v *p);
|
||||||
|
|
||||||
mm_seed_t *mm_collect_matches(void *km, int *_n_m, int qlen, int max_occ, int max_max_occ, int dist, const mm_idx_t *mi, const mm128_v *mv, int64_t *n_a, int *rep_len, int *n_mini_pos, uint64_t **mini_pos);
|
mm_seed_t *mm_collect_matches(void *km, int *_n_m, int qlen, int max_occ, int max_max_occ, int dist, const mm_idx_t *mi, const mm128_v *mv, int64_t *n_a, int *rep_len, int *n_mini_pos, uint64_t **mini_pos);
|
||||||
|
void mm_seed_mz_flt(void *km, mm128_v *mv, int32_t q_occ_max, float q_occ_frac);
|
||||||
|
|
||||||
double mm_event_identity(const mm_reg1_t *r);
|
double mm_event_identity(const mm_reg1_t *r);
|
||||||
int mm_write_sam_hdr(const mm_idx_t *mi, const char *rg, const char *ver, int argc, char *argv[]);
|
int mm_write_sam_hdr(const mm_idx_t *mi, const char *rg, const char *ver, int argc, char *argv[]);
|
||||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag);
|
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag);
|
||||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len);
|
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len);
|
||||||
|
void mm_write_paf4(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len, int n_seg, int seg_idx);
|
||||||
void mm_write_sam(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int n_regs, const mm_reg1_t *regs);
|
void mm_write_sam(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int n_regs, const mm_reg1_t *regs);
|
||||||
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regs, const mm_reg1_t *const* regs, void *km, int64_t opt_flag);
|
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regs, const mm_reg1_t *const* regs, void *km, int64_t opt_flag);
|
||||||
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int64_t opt_flag, int rep_len);
|
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int64_t opt_flag, int rep_len);
|
||||||
|
void mm_write_junc(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r);
|
||||||
|
|
||||||
|
// indexing related in index.c
|
||||||
void mm_idxopt_init(mm_idxopt_t *opt);
|
void mm_idxopt_init(mm_idxopt_t *opt);
|
||||||
const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n);
|
const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n);
|
||||||
int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f);
|
int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f);
|
||||||
int mm_idx_getseq2(const mm_idx_t *mi, int is_rev, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq);
|
int mm_idx_getseq2(const mm_idx_t *mi, int is_rev, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq);
|
||||||
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a);
|
|
||||||
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a, int is_qstrand);
|
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a, int is_qstrand);
|
||||||
|
int mm_idx_bed_read(mm_idx_t *mi, const char *fn, int read_junc);
|
||||||
|
int mm_idx_jjump_read(mm_idx_t *mi, const char *fn, int flag, int min_sc);
|
||||||
|
const mm_idx_jjump1_t *mm_idx_jump_get(const mm_idx_t *db, int32_t cid, int32_t st, int32_t en, int32_t *n);
|
||||||
|
|
||||||
mm128_t *mm_chain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float gap_scale,
|
// chaining in lchain.c
|
||||||
int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
|
||||||
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||||
int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
||||||
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||||
@@ -90,12 +107,17 @@ void mm_sync_regs(void *km, int n_regs, mm_reg1_t *regs);
|
|||||||
int mm_squeeze_a(void *km, int n_regs, mm_reg1_t *regs, mm128_t *a);
|
int mm_squeeze_a(void *km, int n_regs, mm_reg1_t *regs, mm128_t *a);
|
||||||
int mm_set_sam_pri(int n, mm_reg1_t *r);
|
int mm_set_sam_pri(int n, mm_reg1_t *r);
|
||||||
void mm_set_parent(void *km, float mask_level, int mask_len, int n, mm_reg1_t *r, int sub_diff, int hard_mask_level, float alt_diff_frac);
|
void mm_set_parent(void *km, float mask_level, int mask_len, int n, mm_reg1_t *r, int sub_diff, int hard_mask_level, float alt_diff_frac);
|
||||||
void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int *n_, mm_reg1_t *r);
|
void mm_select_sub(void *km, float pri_ratio, int min_diff, int best_n, int check_strand, int min_strand_sc, int *n_, mm_reg1_t *r);
|
||||||
void mm_select_sub_multi(void *km, float pri_ratio, float pri1, float pri2, int max_gap_ref, int min_diff, int best_n, int n_segs, const int *qlens, int *n_, mm_reg1_t *r);
|
void mm_select_sub_multi(void *km, float pri_ratio, float pri1, float pri2, int max_gap_ref, int min_diff, int best_n, int n_segs, const int *qlens, int *n_, mm_reg1_t *r);
|
||||||
|
int mm_filter_strand_retained(int n_regs, mm_reg1_t *r);
|
||||||
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs);
|
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs);
|
||||||
void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r, float alt_diff_frac);
|
void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r, float alt_diff_frac);
|
||||||
void mm_set_mapq(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr);
|
void mm_set_mapq2(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr, int is_splice);
|
||||||
void mm_update_dp_max(int qlen, int n_regs, mm_reg1_t *regs, float frac, int a, int b);
|
void mm_update_dp_max(int qlen, int n_regs, mm_reg1_t *regs, float frac, int a, int b);
|
||||||
|
void mm_jump_split(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq, mm_reg1_t *r, int32_t ts_strand);
|
||||||
|
|
||||||
|
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a);
|
||||||
|
void mm_enlarge_cigar(mm_reg1_t *r, uint32_t n_cigar);
|
||||||
|
|
||||||
void mm_est_err(const mm_idx_t *mi, int qlen, int n_regs, mm_reg1_t *regs, const mm128_t *a, int32_t n, const uint64_t *mini_pos);
|
void mm_est_err(const mm_idx_t *mi, int qlen, int n_regs, mm_reg1_t *regs, const mm128_t *a, int32_t n, const uint64_t *mini_pos);
|
||||||
|
|
||||||
@@ -103,6 +125,8 @@ mm_seg_t *mm_seg_gen(void *km, uint32_t hash, int n_segs, const int *qlens, int
|
|||||||
void mm_seg_free(void *km, int n_segs, mm_seg_t *segs);
|
void mm_seg_free(void *km, int n_segs, mm_seg_t *segs);
|
||||||
void mm_pair(void *km, int max_gap_ref, int dp_bonus, int sub_diff, int match_sc, const int *qlens, int *n_regs, mm_reg1_t **regs);
|
void mm_pair(void *km, int max_gap_ref, int dp_bonus, int sub_diff, int match_sc, const int *qlens, int *n_regs, mm_reg1_t **regs);
|
||||||
|
|
||||||
|
void mm_jump_split(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq, mm_reg1_t *r, int32_t ts_strand);
|
||||||
|
|
||||||
FILE *mm_split_init(const char *prefix, const mm_idx_t *mi);
|
FILE *mm_split_init(const char *prefix, const mm_idx_t *mi);
|
||||||
mm_idx_t *mm_split_merge_prep(const char *prefix, int n_splits, FILE **fp, uint32_t *n_seq_part);
|
mm_idx_t *mm_split_merge_prep(const char *prefix, int n_splits, FILE **fp, uint32_t *n_seq_part);
|
||||||
int mm_split_merge(int n_segs, const char **fn, const mm_mapopt_t *opt, int n_split_idx);
|
int mm_split_merge(int n_segs, const char **fn, const mm_mapopt_t *opt, int n_split_idx);
|
||||||
|
|||||||
@@ -8,7 +8,7 @@ void mm_idxopt_init(mm_idxopt_t *opt)
|
|||||||
opt->k = 15, opt->w = 10, opt->flag = 0;
|
opt->k = 15, opt->w = 10, opt->flag = 0;
|
||||||
opt->bucket_bits = 14;
|
opt->bucket_bits = 14;
|
||||||
opt->mini_batch_size = 50000000;
|
opt->mini_batch_size = 50000000;
|
||||||
opt->batch_size = 4000000000ULL;
|
opt->batch_size = 8000000000ULL;
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_mapopt_init(mm_mapopt_t *opt)
|
void mm_mapopt_init(mm_mapopt_t *opt)
|
||||||
@@ -19,6 +19,7 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
|||||||
opt->min_mid_occ = 10;
|
opt->min_mid_occ = 10;
|
||||||
opt->max_mid_occ = 1000000;
|
opt->max_mid_occ = 1000000;
|
||||||
opt->sdust_thres = 0; // no SDUST masking
|
opt->sdust_thres = 0; // no SDUST masking
|
||||||
|
opt->q_occ_frac = 0.01f;
|
||||||
|
|
||||||
opt->min_cnt = 3;
|
opt->min_cnt = 3;
|
||||||
opt->min_chain_score = 40;
|
opt->min_chain_score = 40;
|
||||||
@@ -32,6 +33,7 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
|||||||
opt->rmq_rescue_size = 1000;
|
opt->rmq_rescue_size = 1000;
|
||||||
opt->rmq_rescue_ratio = 0.1f;
|
opt->rmq_rescue_ratio = 0.1f;
|
||||||
opt->chain_gap_scale = 0.8f;
|
opt->chain_gap_scale = 0.8f;
|
||||||
|
opt->chain_skip_scale = 0.0f;
|
||||||
opt->max_max_occ = 4095;
|
opt->max_max_occ = 4095;
|
||||||
opt->occ_dist = 500;
|
opt->occ_dist = 500;
|
||||||
|
|
||||||
@@ -43,6 +45,7 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
|||||||
opt->alt_drop = 0.15f;
|
opt->alt_drop = 0.15f;
|
||||||
|
|
||||||
opt->a = 2, opt->b = 4, opt->q = 4, opt->e = 2, opt->q2 = 24, opt->e2 = 1;
|
opt->a = 2, opt->b = 4, opt->q = 4, opt->e = 2, opt->q2 = 24, opt->e2 = 1;
|
||||||
|
opt->transition = 0;
|
||||||
opt->sc_ambi = 1;
|
opt->sc_ambi = 1;
|
||||||
opt->zdrop = 400, opt->zdrop_inv = 200;
|
opt->zdrop = 400, opt->zdrop_inv = 200;
|
||||||
opt->end_bonus = -1;
|
opt->end_bonus = -1;
|
||||||
@@ -52,12 +55,15 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
|||||||
opt->max_clip_ratio = 1.0f;
|
opt->max_clip_ratio = 1.0f;
|
||||||
opt->mini_batch_size = 500000000;
|
opt->mini_batch_size = 500000000;
|
||||||
opt->max_sw_mat = 100000000;
|
opt->max_sw_mat = 100000000;
|
||||||
|
opt->cap_kalloc = 500000000;
|
||||||
|
|
||||||
opt->rank_min_len = 500;
|
opt->rank_min_len = 500;
|
||||||
opt->rank_frac = 0.9f;
|
opt->rank_frac = 0.9f;
|
||||||
|
|
||||||
opt->pe_ori = 0; // FF
|
opt->pe_ori = 0; // FF
|
||||||
opt->pe_bonus = 33;
|
opt->pe_bonus = 33;
|
||||||
|
|
||||||
|
opt->jump_min_match = 3;
|
||||||
}
|
}
|
||||||
|
|
||||||
void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
||||||
@@ -71,6 +77,7 @@ void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
|||||||
if (opt->max_mid_occ > opt->min_mid_occ && opt->mid_occ > opt->max_mid_occ)
|
if (opt->max_mid_occ > opt->min_mid_occ && opt->mid_occ > opt->max_mid_occ)
|
||||||
opt->mid_occ = opt->max_mid_occ;
|
opt->mid_occ = opt->max_mid_occ;
|
||||||
}
|
}
|
||||||
|
if (opt->bw_long < opt->bw) opt->bw_long = opt->bw;
|
||||||
if (mm_verbose >= 3)
|
if (mm_verbose >= 3)
|
||||||
fprintf(stderr, "[M::%s::%.3f*%.2f] mid_occ = %d\n", __func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), opt->mid_occ);
|
fprintf(stderr, "[M::%s::%.3f*%.2f] mid_occ = %d\n", __func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), opt->mid_occ);
|
||||||
}
|
}
|
||||||
@@ -86,7 +93,7 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
|||||||
if (preset == 0) {
|
if (preset == 0) {
|
||||||
mm_idxopt_init(io);
|
mm_idxopt_init(io);
|
||||||
mm_mapopt_init(mo);
|
mm_mapopt_init(mo);
|
||||||
} else if (strcmp(preset, "map-ont") == 0) { // this is the same as the default
|
} else if (strcmp(preset, "lr") == 0 || strcmp(preset, "map-ont") == 0) { // this is the same as the default
|
||||||
} else if (strcmp(preset, "ava-ont") == 0) {
|
} else if (strcmp(preset, "ava-ont") == 0) {
|
||||||
io->flag = 0, io->k = 15, io->w = 5;
|
io->flag = 0, io->k = 15, io->w = 5;
|
||||||
mo->flag |= MM_F_ALL_CHAINS | MM_F_NO_DIAG | MM_F_NO_DUAL | MM_F_NO_LJOIN;
|
mo->flag |= MM_F_ALL_CHAINS | MM_F_NO_DIAG | MM_F_NO_DUAL | MM_F_NO_LJOIN;
|
||||||
@@ -101,16 +108,33 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
|||||||
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_chain_skip = 25;
|
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_chain_skip = 25;
|
||||||
mo->bw_long = mo->bw;
|
mo->bw_long = mo->bw;
|
||||||
mo->occ_dist = 0;
|
mo->occ_dist = 0;
|
||||||
} else if (strcmp(preset, "map-hifi") == 0 || strcmp(preset, "map-ccs") == 0) {
|
} else if (strcmp(preset, "lr:hq") == 0 || strcmp(preset, "map-hifi") == 0 || strcmp(preset, "map-ccs") == 0) {
|
||||||
io->flag = 0, io->k = 19, io->w = 19;
|
io->flag = 0, io->k = 19, io->w = 19;
|
||||||
mo->max_gap = 10000;
|
mo->max_gap = 10000;
|
||||||
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1;
|
|
||||||
mo->occ_dist = 500;
|
|
||||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
||||||
mo->min_dp_max = 200;
|
if (strcmp(preset, "map-hifi") == 0 || strcmp(preset, "map-ccs") == 0) {
|
||||||
|
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1;
|
||||||
|
mo->min_dp_max = 200;
|
||||||
|
}
|
||||||
|
} else if (strcmp(preset, "lr:hqae") == 0) { // high-quality assembly evaluation
|
||||||
|
io->flag = 0, io->k = 25, io->w = 51;
|
||||||
|
mo->flag |= MM_F_RMQ;
|
||||||
|
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
||||||
|
mo->rmq_inner_dist = 5000;
|
||||||
|
mo->occ_dist = 200;
|
||||||
|
mo->best_n = 100;
|
||||||
|
mo->chain_gap_scale = 5.0f;
|
||||||
|
} else if (strcmp(preset, "map-iclr-prerender") == 0) {
|
||||||
|
io->flag = 0, io->k = 15;
|
||||||
|
mo->b = 6, mo->transition = 1;
|
||||||
|
mo->q = 10, mo->q2 = 50;
|
||||||
|
} else if (strcmp(preset, "map-iclr") == 0) {
|
||||||
|
io->flag = 0, io->k = 19;
|
||||||
|
mo->b = 6, mo->transition = 4;
|
||||||
|
mo->q = 10, mo->q2 = 50;
|
||||||
} else if (strncmp(preset, "asm", 3) == 0) {
|
} else if (strncmp(preset, "asm", 3) == 0) {
|
||||||
io->flag = 0, io->k = 19, io->w = 19;
|
io->flag = 0, io->k = 19, io->w = 19;
|
||||||
mo->bw = mo->bw_long = 100000;
|
mo->bw = 1000, mo->bw_long = 100000;
|
||||||
mo->max_gap = 10000;
|
mo->max_gap = 10000;
|
||||||
mo->flag |= MM_F_RMQ;
|
mo->flag |= MM_F_RMQ;
|
||||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
||||||
@@ -142,7 +166,7 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
|||||||
mo->mid_occ = 1000;
|
mo->mid_occ = 1000;
|
||||||
mo->max_occ = 5000;
|
mo->max_occ = 5000;
|
||||||
mo->mini_batch_size = 50000000;
|
mo->mini_batch_size = 50000000;
|
||||||
} else if (strncmp(preset, "splice", 6) == 0 || strcmp(preset, "cdna") == 0) {
|
} else if (strcmp(preset, "splice") == 0 || strcmp(preset, "splice:hq") == 0 || strcmp(preset, "splice:sr") == 0 || strcmp(preset, "cdna") == 0) {
|
||||||
io->flag = 0, io->k = 15, io->w = 5;
|
io->flag = 0, io->k = 15, io->w = 5;
|
||||||
mo->flag |= MM_F_SPLICE | MM_F_SPLICE_FOR | MM_F_SPLICE_REV | MM_F_SPLICE_FLANK;
|
mo->flag |= MM_F_SPLICE | MM_F_SPLICE_FOR | MM_F_SPLICE_REV | MM_F_SPLICE_FLANK;
|
||||||
mo->max_sw_mat = 0;
|
mo->max_sw_mat = 0;
|
||||||
@@ -150,13 +174,31 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
|||||||
mo->a = 1, mo->b = 2, mo->q = 2, mo->e = 1, mo->q2 = 32, mo->e2 = 0;
|
mo->a = 1, mo->b = 2, mo->q = 2, mo->e = 1, mo->q2 = 32, mo->e2 = 0;
|
||||||
mo->noncan = 9;
|
mo->noncan = 9;
|
||||||
mo->junc_bonus = 9;
|
mo->junc_bonus = 9;
|
||||||
|
mo->junc_pen = 5;
|
||||||
mo->zdrop = 200, mo->zdrop_inv = 100; // because mo->a is halved
|
mo->zdrop = 200, mo->zdrop_inv = 100; // because mo->a is halved
|
||||||
if (strcmp(preset, "splice:hq") == 0)
|
if (strcmp(preset, "splice:hq") == 0) {
|
||||||
mo->junc_bonus = 5, mo->b = 4, mo->q = 6, mo->q2 = 24;
|
mo->noncan = 5, mo->b = 4, mo->q = 6, mo->q2 = 24;
|
||||||
|
} else if (strcmp(preset, "splice:sr") == 0) {
|
||||||
|
mo->flag |= MM_F_NO_PRINT_2ND | MM_F_2_IO_THREADS | MM_F_HEAP_SORT | MM_F_FRAG_MODE | MM_F_WEAK_PAIRING | MM_F_SR_RNA;
|
||||||
|
mo->noncan = 5, mo->b = 4, mo->q = 6, mo->q2 = 24;
|
||||||
|
mo->min_chain_score = 25;
|
||||||
|
mo->min_dp_max = 40;
|
||||||
|
mo->min_ksw_len = 20;
|
||||||
|
mo->pe_ori = 0<<1|1; // FR
|
||||||
|
mo->best_n = 10;
|
||||||
|
mo->mini_batch_size = 100000000;
|
||||||
|
}
|
||||||
} else return -1;
|
} else return -1;
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
int mm_max_spsc_bonus(const mm_mapopt_t *mo)
|
||||||
|
{
|
||||||
|
int max_sc = (mo->q2 + 1) / 2 - 1;
|
||||||
|
max_sc = max_sc > mo->q2 - mo->q? max_sc : mo->q2 - mo->q;
|
||||||
|
return max_sc;
|
||||||
|
}
|
||||||
|
|
||||||
int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
||||||
{
|
{
|
||||||
if (mo->bw > mo->bw_long) {
|
if (mo->bw > mo->bw_long) {
|
||||||
@@ -211,6 +253,11 @@ int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
|||||||
fprintf(stderr, "[ERROR]\033[1;31m scoring system violating ({-O}+{-E})+({-O2}+{-E2}) <= 127\033[0m\n");
|
fprintf(stderr, "[ERROR]\033[1;31m scoring system violating ({-O}+{-E})+({-O2}+{-E2}) <= 127\033[0m\n");
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
|
if (mo->sc_ambi < 0 || mo->sc_ambi >= mo->b) {
|
||||||
|
if (mm_verbose >= 1)
|
||||||
|
fprintf(stderr, "[ERROR]\033[1;31m --score-N should be within [0,{-B})\033[0m\n");
|
||||||
|
return -1;
|
||||||
|
}
|
||||||
if (mo->zdrop < mo->zdrop_inv) {
|
if (mo->zdrop < mo->zdrop_inv) {
|
||||||
if (mm_verbose >= 1)
|
if (mm_verbose >= 1)
|
||||||
fprintf(stderr, "[ERROR]\033[1;31m Z-drop should not be less than inversion-Z-drop\033[0m\n");
|
fprintf(stderr, "[ERROR]\033[1;31m Z-drop should not be less than inversion-Z-drop\033[0m\n");
|
||||||
|
|||||||
@@ -8,7 +8,8 @@ void mm_select_sub_multi(void *km, float pri_ratio, float pri1, float pri2, int
|
|||||||
if (pri_ratio > 0.0f && *n_ > 0) {
|
if (pri_ratio > 0.0f && *n_ > 0) {
|
||||||
int i, k, n = *n_, n_2nd = 0;
|
int i, k, n = *n_, n_2nd = 0;
|
||||||
int max_dist = n_segs == 2? qlens[0] + qlens[1] + max_gap_ref : 0;
|
int max_dist = n_segs == 2? qlens[0] + qlens[1] + max_gap_ref : 0;
|
||||||
for (i = k = 0; i < n; ++i) {
|
uint8_t *keep = (uint8_t*)kmalloc(km, n);
|
||||||
|
for (i = 0; i < n; ++i) {
|
||||||
int to_keep = 0;
|
int to_keep = 0;
|
||||||
if (r[i].parent == i) { // primary
|
if (r[i].parent == i) { // primary
|
||||||
to_keep = 1;
|
to_keep = 1;
|
||||||
@@ -34,9 +35,13 @@ void mm_select_sub_multi(void *km, float pri_ratio, float pri1, float pri2, int
|
|||||||
if (to_keep && r[i].parent != i) {
|
if (to_keep && r[i].parent != i) {
|
||||||
if (n_2nd++ >= best_n) to_keep = 0; // don't keep if there are too many secondary hits
|
if (n_2nd++ >= best_n) to_keep = 0; // don't keep if there are too many secondary hits
|
||||||
}
|
}
|
||||||
if (to_keep) r[k++] = r[i];
|
keep[i] = to_keep;
|
||||||
|
}
|
||||||
|
for (i = k = 0; i < n; ++i) {
|
||||||
|
if (keep[i]) r[k++] = r[i];
|
||||||
else if (r[i].p) free(r[i].p);
|
else if (r[i].p) free(r[i].p);
|
||||||
}
|
}
|
||||||
|
kfree(km, keep);
|
||||||
if (k != n) mm_sync_regs(km, k, r); // removing hits requires sync()
|
if (k != n) mm_sync_regs(km, k, r); // removing hits requires sync()
|
||||||
*n_ = k;
|
*n_ = k;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,2 @@
|
|||||||
|
[build-system]
|
||||||
|
requires = ["setuptools", "wheel", "Cython"]
|
||||||
+3
-1
@@ -77,7 +77,9 @@ This constructor accepts the following arguments:
|
|||||||
|
|
||||||
* **min_chain_score**: minimum chaing score
|
* **min_chain_score**: minimum chaing score
|
||||||
|
|
||||||
* **bw**: chaining and alignment band width
|
* **bw**: chaining and alignment band width (initial chaining and extension)
|
||||||
|
|
||||||
|
* **bw_long**: chaining and alignment band width (RMQ-based rechaining and closing gaps)
|
||||||
|
|
||||||
* **best_n**: max number of alignments to return
|
* **best_n**: max number of alignments to return
|
||||||
|
|
||||||
|
|||||||
+3
-3
@@ -71,13 +71,13 @@ static inline void mm_reset_timer(void)
|
|||||||
}
|
}
|
||||||
|
|
||||||
extern unsigned char seq_comp_table[256];
|
extern unsigned char seq_comp_table[256];
|
||||||
static inline mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char *seq1, const char *seq2, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt)
|
static inline mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char* seqname, const char *seq1, const char *seq2, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt)
|
||||||
{
|
{
|
||||||
mm_reg1_t *r;
|
mm_reg1_t *r;
|
||||||
|
|
||||||
Py_BEGIN_ALLOW_THREADS
|
Py_BEGIN_ALLOW_THREADS
|
||||||
if (seq2 == 0) {
|
if (seq2 == 0) {
|
||||||
r = mm_map(mi, strlen(seq1), seq1, n_regs, b, opt, NULL);
|
r = mm_map(mi, strlen(seq1), seq1, n_regs, b, opt, seqname);
|
||||||
} else {
|
} else {
|
||||||
int _n_regs[2];
|
int _n_regs[2];
|
||||||
mm_reg1_t *regs[2];
|
mm_reg1_t *regs[2];
|
||||||
@@ -94,7 +94,7 @@ static inline mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char *seq1, const
|
|||||||
seq[1][i] = seq_comp_table[t];
|
seq[1][i] = seq_comp_table[t];
|
||||||
}
|
}
|
||||||
if (len[1]&1) seq[1][len[1]>>1] = seq_comp_table[(uint8_t)seq[1][len[1]>>1]];
|
if (len[1]&1) seq[1][len[1]>>1] = seq_comp_table[(uint8_t)seq[1][len[1]>>1]];
|
||||||
mm_map_frag(mi, 2, len, (const char**)seq, _n_regs, regs, b, opt, NULL);
|
mm_map_frag(mi, 2, len, (const char**)seq, _n_regs, regs, b, opt, seqname);
|
||||||
for (i = 0; i < _n_regs[1]; ++i)
|
for (i = 0; i < _n_regs[1]; ++i)
|
||||||
regs[1][i].rev = !regs[1][i].rev;
|
regs[1][i].rev = !regs[1][i].rev;
|
||||||
*n_regs = _n_regs[0] + _n_regs[1];
|
*n_regs = _n_regs[0] + _n_regs[1];
|
||||||
|
|||||||
+8
-2
@@ -23,6 +23,7 @@ cdef extern from "minimap.h":
|
|||||||
int min_cnt
|
int min_cnt
|
||||||
int min_chain_score
|
int min_chain_score
|
||||||
float chain_gap_scale
|
float chain_gap_scale
|
||||||
|
float chain_skip_scale
|
||||||
int rmq_size_cap, rmq_inner_dist
|
int rmq_size_cap, rmq_inner_dist
|
||||||
int rmq_rescue_size
|
int rmq_rescue_size
|
||||||
float rmq_rescue_ratio
|
float rmq_rescue_ratio
|
||||||
@@ -35,9 +36,10 @@ cdef extern from "minimap.h":
|
|||||||
float alt_drop
|
float alt_drop
|
||||||
|
|
||||||
int a, b, q, e, q2, e2
|
int a, b, q, e, q2, e2
|
||||||
|
int transition
|
||||||
int sc_ambi
|
int sc_ambi
|
||||||
int noncan
|
int noncan
|
||||||
int junc_bonus
|
int junc_bonus, junc_pen
|
||||||
int zdrop, zdrop_inv
|
int zdrop, zdrop_inv
|
||||||
int end_bonus
|
int end_bonus
|
||||||
int min_dp_max
|
int min_dp_max
|
||||||
@@ -50,7 +52,10 @@ cdef extern from "minimap.h":
|
|||||||
|
|
||||||
int pe_ori, pe_bonus
|
int pe_ori, pe_bonus
|
||||||
|
|
||||||
|
int jump_min_match;
|
||||||
|
|
||||||
float mid_occ_frac
|
float mid_occ_frac
|
||||||
|
float q_occ_frac
|
||||||
int32_t min_mid_occ
|
int32_t min_mid_occ
|
||||||
int32_t mid_occ
|
int32_t mid_occ
|
||||||
int32_t max_occ
|
int32_t max_occ
|
||||||
@@ -107,6 +112,7 @@ cdef extern from "minimap.h":
|
|||||||
void mm_tbuf_destroy(mm_tbuf_t *b)
|
void mm_tbuf_destroy(mm_tbuf_t *b)
|
||||||
void *mm_tbuf_get_km(mm_tbuf_t *b)
|
void *mm_tbuf_get_km(mm_tbuf_t *b)
|
||||||
int mm_gen_cs(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden)
|
int mm_gen_cs(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden)
|
||||||
|
int mm_gen_ds(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq, int no_iden)
|
||||||
int mm_gen_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq)
|
int mm_gen_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, const mm_reg1_t *r, const char *seq)
|
||||||
|
|
||||||
#
|
#
|
||||||
@@ -126,7 +132,7 @@ cdef extern from "cmappy.h":
|
|||||||
|
|
||||||
void mm_reg2hitpy(const mm_idx_t *mi, mm_reg1_t *r, mm_hitpy_t *h)
|
void mm_reg2hitpy(const mm_idx_t *mi, mm_reg1_t *r, mm_hitpy_t *h)
|
||||||
void mm_free_reg1(mm_reg1_t *r)
|
void mm_free_reg1(mm_reg1_t *r)
|
||||||
mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char *seq1, const char *seq2, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt)
|
mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char* seqname, const char *seq1, const char *seq2, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt)
|
||||||
char *mappy_fetch_seq(const mm_idx_t *mi, const char *name, int st, int en, int *l)
|
char *mappy_fetch_seq(const mm_idx_t *mi, const char *name, int st, int en, int *l)
|
||||||
mm_idx_t *mappy_idx_seq(int w, int k, int is_hpc, int bucket_bits, const char *seq, int l)
|
mm_idx_t *mappy_idx_seq(int w, int k, int is_hpc, int bucket_bits, const char *seq, int l)
|
||||||
|
|
||||||
|
|||||||
+37
-13
@@ -3,7 +3,7 @@ from libc.stdlib cimport free
|
|||||||
cimport cmappy
|
cimport cmappy
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
__version__ = '2.22'
|
__version__ = '2.31'
|
||||||
|
|
||||||
cmappy.mm_reset_timer()
|
cmappy.mm_reset_timer()
|
||||||
|
|
||||||
@@ -14,9 +14,9 @@ cdef class Alignment:
|
|||||||
cdef int8_t _strand, _trans_strand
|
cdef int8_t _strand, _trans_strand
|
||||||
cdef uint8_t _mapq, _is_primary
|
cdef uint8_t _mapq, _is_primary
|
||||||
cdef int _seg_id
|
cdef int _seg_id
|
||||||
cdef _ctg, _cigar, _cs, _MD # these are python objects
|
cdef _ctg, _cigar, _cs, _ds, _MD # these are python objects
|
||||||
|
|
||||||
def __cinit__(self, ctg, cl, cs, ce, strand, qs, qe, mapq, cigar, is_primary, mlen, blen, NM, trans_strand, seg_id, cs_str, MD_str):
|
def __cinit__(self, ctg, cl, cs, ce, strand, qs, qe, mapq, cigar, is_primary, mlen, blen, NM, trans_strand, seg_id, cs_str, ds_str, MD_str):
|
||||||
self._ctg = ctg if isinstance(ctg, str) else ctg.decode()
|
self._ctg = ctg if isinstance(ctg, str) else ctg.decode()
|
||||||
self._ctg_len, self._r_st, self._r_en = cl, cs, ce
|
self._ctg_len, self._r_st, self._r_en = cl, cs, ce
|
||||||
self._strand, self._q_st, self._q_en = strand, qs, qe
|
self._strand, self._q_st, self._q_en = strand, qs, qe
|
||||||
@@ -27,6 +27,7 @@ cdef class Alignment:
|
|||||||
self._trans_strand = trans_strand
|
self._trans_strand = trans_strand
|
||||||
self._seg_id = seg_id
|
self._seg_id = seg_id
|
||||||
self._cs = cs_str
|
self._cs = cs_str
|
||||||
|
self._ds = ds_str
|
||||||
self._MD = MD_str
|
self._MD = MD_str
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -77,6 +78,9 @@ cdef class Alignment:
|
|||||||
@property
|
@property
|
||||||
def cs(self): return self._cs
|
def cs(self): return self._cs
|
||||||
|
|
||||||
|
@property
|
||||||
|
def ds(self): return self._ds
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def MD(self): return self._MD
|
def MD(self): return self._MD
|
||||||
|
|
||||||
@@ -96,6 +100,8 @@ cdef class Alignment:
|
|||||||
a = [str(self._q_st), str(self._q_en), strand, self._ctg, str(self._ctg_len), str(self._r_st), str(self._r_en),
|
a = [str(self._q_st), str(self._q_en), strand, self._ctg, str(self._ctg_len), str(self._r_st), str(self._r_en),
|
||||||
str(self._mlen), str(self._blen), str(self._mapq), tp, ts, "cg:Z:" + self.cigar_str]
|
str(self._mlen), str(self._blen), str(self._mapq), tp, ts, "cg:Z:" + self.cigar_str]
|
||||||
if self._cs != "": a.append("cs:Z:" + self._cs)
|
if self._cs != "": a.append("cs:Z:" + self._cs)
|
||||||
|
if self._ds != "": a.append("ds:Z:" + self._ds)
|
||||||
|
if self._MD != "": a.append("MD:Z:" + self._MD)
|
||||||
return "\t".join(a)
|
return "\t".join(a)
|
||||||
|
|
||||||
cdef class ThreadBuffer:
|
cdef class ThreadBuffer:
|
||||||
@@ -112,7 +118,7 @@ cdef class Aligner:
|
|||||||
cdef cmappy.mm_idxopt_t idx_opt
|
cdef cmappy.mm_idxopt_t idx_opt
|
||||||
cdef cmappy.mm_mapopt_t map_opt
|
cdef cmappy.mm_mapopt_t map_opt
|
||||||
|
|
||||||
def __cinit__(self, fn_idx_in=None, preset=None, k=None, w=None, min_cnt=None, min_chain_score=None, min_dp_score=None, bw=None, best_n=None, n_threads=3, fn_idx_out=None, max_frag_len=None, extra_flags=None, seq=None, scoring=None):
|
def __cinit__(self, fn_idx_in=None, preset=None, k=None, w=None, min_cnt=None, min_chain_score=None, min_dp_score=None, bw=None, bw_long=None, best_n=None, n_threads=3, fn_idx_out=None, max_frag_len=None, extra_flags=None, seq=None, scoring=None, sc_ambi=None, max_chain_skip=None):
|
||||||
self._idx = NULL
|
self._idx = NULL
|
||||||
cmappy.mm_set_opt(NULL, &self.idx_opt, &self.map_opt) # set the default options
|
cmappy.mm_set_opt(NULL, &self.idx_opt, &self.map_opt) # set the default options
|
||||||
if preset is not None:
|
if preset is not None:
|
||||||
@@ -125,6 +131,7 @@ cdef class Aligner:
|
|||||||
if min_chain_score is not None: self.map_opt.min_chain_score = min_chain_score
|
if min_chain_score is not None: self.map_opt.min_chain_score = min_chain_score
|
||||||
if min_dp_score is not None: self.map_opt.min_dp_max = min_dp_score
|
if min_dp_score is not None: self.map_opt.min_dp_max = min_dp_score
|
||||||
if bw is not None: self.map_opt.bw = bw
|
if bw is not None: self.map_opt.bw = bw
|
||||||
|
if bw_long is not None: self.map_opt.bw_long = bw_long
|
||||||
if best_n is not None: self.map_opt.best_n = best_n
|
if best_n is not None: self.map_opt.best_n = best_n
|
||||||
if max_frag_len is not None: self.map_opt.max_frag_len = max_frag_len
|
if max_frag_len is not None: self.map_opt.max_frag_len = max_frag_len
|
||||||
if extra_flags is not None: self.map_opt.flag |= extra_flags
|
if extra_flags is not None: self.map_opt.flag |= extra_flags
|
||||||
@@ -136,6 +143,8 @@ cdef class Aligner:
|
|||||||
self.map_opt.q2, self.map_opt.e2 = scoring[4], scoring[5]
|
self.map_opt.q2, self.map_opt.e2 = scoring[4], scoring[5]
|
||||||
if len(scoring) >= 7:
|
if len(scoring) >= 7:
|
||||||
self.map_opt.sc_ambi = scoring[6]
|
self.map_opt.sc_ambi = scoring[6]
|
||||||
|
if sc_ambi is not None: self.map_opt.sc_ambi = sc_ambi
|
||||||
|
if max_chain_skip is not None: self.map_opt.max_chain_skip = max_chain_skip
|
||||||
|
|
||||||
cdef cmappy.mm_idx_reader_t *r;
|
cdef cmappy.mm_idx_reader_t *r;
|
||||||
|
|
||||||
@@ -161,7 +170,7 @@ cdef class Aligner:
|
|||||||
def __bool__(self):
|
def __bool__(self):
|
||||||
return (self._idx != NULL)
|
return (self._idx != NULL)
|
||||||
|
|
||||||
def map(self, seq, seq2=None, buf=None, cs=False, MD=False, max_frag_len=None, extra_flags=None):
|
def map(self, seq, seq2=None, name=None, buf=None, cs=False, ds=False, MD=False, max_frag_len=None, extra_flags=None):
|
||||||
cdef cmappy.mm_reg1_t *regs
|
cdef cmappy.mm_reg1_t *regs
|
||||||
cdef cmappy.mm_hitpy_t h
|
cdef cmappy.mm_hitpy_t h
|
||||||
cdef ThreadBuffer b
|
cdef ThreadBuffer b
|
||||||
@@ -172,6 +181,7 @@ cdef class Aligner:
|
|||||||
cdef cmappy.mm_mapopt_t map_opt
|
cdef cmappy.mm_mapopt_t map_opt
|
||||||
|
|
||||||
if self._idx == NULL: return
|
if self._idx == NULL: return
|
||||||
|
if ((self.map_opt.flag & 4) and (self._idx.flag & 2)): return
|
||||||
map_opt = self.map_opt
|
map_opt = self.map_opt
|
||||||
if max_frag_len is not None: map_opt.max_frag_len = max_frag_len
|
if max_frag_len is not None: map_opt.max_frag_len = max_frag_len
|
||||||
if extra_flags is not None: map_opt.flag |= extra_flags
|
if extra_flags is not None: map_opt.flag |= extra_flags
|
||||||
@@ -179,31 +189,44 @@ cdef class Aligner:
|
|||||||
if self._idx is NULL: return None
|
if self._idx is NULL: return None
|
||||||
if buf is None: b = ThreadBuffer()
|
if buf is None: b = ThreadBuffer()
|
||||||
else: b = buf
|
else: b = buf
|
||||||
km = cmappy.mm_tbuf_get_km(b._b)
|
|
||||||
|
|
||||||
_seq = seq if isinstance(seq, bytes) else seq.encode()
|
_seq = seq if isinstance(seq, bytes) else seq.encode()
|
||||||
|
if name is not None:
|
||||||
|
_name = name if isinstance(name, bytes) else name.encode()
|
||||||
|
|
||||||
if seq2 is None:
|
if seq2 is None:
|
||||||
regs = cmappy.mm_map_aux(self._idx, _seq, NULL, &n_regs, b._b, &map_opt)
|
if name is None:
|
||||||
|
regs = cmappy.mm_map_aux(self._idx, NULL, _seq, NULL, &n_regs, b._b, &map_opt)
|
||||||
|
else:
|
||||||
|
regs = cmappy.mm_map_aux(self._idx, _name, _seq, NULL, &n_regs, b._b, &map_opt)
|
||||||
else:
|
else:
|
||||||
_seq2 = seq2 if isinstance(seq2, bytes) else seq2.encode()
|
_seq2 = seq2 if isinstance(seq2, bytes) else seq2.encode()
|
||||||
regs = cmappy.mm_map_aux(self._idx, _seq, _seq2, &n_regs, b._b, &map_opt)
|
if name is None:
|
||||||
|
regs = cmappy.mm_map_aux(self._idx, NULL, _seq, _seq2, &n_regs, b._b, &map_opt)
|
||||||
|
else:
|
||||||
|
regs = cmappy.mm_map_aux(self._idx, _name, _seq, _seq2, &n_regs, b._b, &map_opt)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
i = 0
|
i = 0
|
||||||
while i < n_regs:
|
while i < n_regs:
|
||||||
cmappy.mm_reg2hitpy(self._idx, ®s[i], &h)
|
cmappy.mm_reg2hitpy(self._idx, ®s[i], &h)
|
||||||
cigar, _cs, _MD = [], '', ''
|
cigar, _cs, _ds, _MD = [], '', '', ''
|
||||||
for k in range(h.n_cigar32): # convert the 32-bit CIGAR encoding to Python array
|
for k in range(h.n_cigar32): # convert the 32-bit CIGAR encoding to Python array
|
||||||
c = h.cigar32[k]
|
c = h.cigar32[k]
|
||||||
cigar.append([c>>4, c&0xf])
|
cigar.append([c>>4, c&0xf])
|
||||||
if cs or MD: # generate the cs and/or the MD tag, if requested
|
if cs or ds or MD: # generate the cs/ds and/or the MD tag, if requested
|
||||||
|
km = cmappy.mm_tbuf_get_km(b._b)
|
||||||
|
_cur_seq = _seq2 if h.seg_id > 0 and seq2 is not None else _seq
|
||||||
if cs:
|
if cs:
|
||||||
l_cs_str = cmappy.mm_gen_cs(km, &cs_str, &m_cs_str, self._idx, ®s[i], _seq, 1)
|
l_cs_str = cmappy.mm_gen_cs(km, &cs_str, &m_cs_str, self._idx, ®s[i], _cur_seq, 1)
|
||||||
_cs = cs_str[:l_cs_str] if isinstance(cs_str, str) else cs_str[:l_cs_str].decode()
|
_cs = cs_str[:l_cs_str] if isinstance(cs_str, str) else cs_str[:l_cs_str].decode()
|
||||||
|
if ds:
|
||||||
|
l_cs_str = cmappy.mm_gen_ds(km, &cs_str, &m_cs_str, self._idx, ®s[i], _cur_seq, 1)
|
||||||
|
_ds = cs_str[:l_cs_str] if isinstance(cs_str, str) else cs_str[:l_cs_str].decode()
|
||||||
if MD:
|
if MD:
|
||||||
l_cs_str = cmappy.mm_gen_MD(km, &cs_str, &m_cs_str, self._idx, ®s[i], _seq)
|
l_cs_str = cmappy.mm_gen_MD(km, &cs_str, &m_cs_str, self._idx, ®s[i], _cur_seq)
|
||||||
_MD = cs_str[:l_cs_str] if isinstance(cs_str, str) else cs_str[:l_cs_str].decode()
|
_MD = cs_str[:l_cs_str] if isinstance(cs_str, str) else cs_str[:l_cs_str].decode()
|
||||||
yield Alignment(h.ctg, h.ctg_len, h.ctg_start, h.ctg_end, h.strand, h.qry_start, h.qry_end, h.mapq, cigar, h.is_primary, h.mlen, h.blen, h.NM, h.trans_strand, h.seg_id, _cs, _MD)
|
yield Alignment(h.ctg, h.ctg_len, h.ctg_start, h.ctg_end, h.strand, h.qry_start, h.qry_end, h.mapq, cigar, h.is_primary, h.mlen, h.blen, h.NM, h.trans_strand, h.seg_id, _cs, _ds, _MD)
|
||||||
cmappy.mm_free_reg1(®s[i])
|
cmappy.mm_free_reg1(®s[i])
|
||||||
i += 1
|
i += 1
|
||||||
finally:
|
finally:
|
||||||
@@ -217,6 +240,7 @@ cdef class Aligner:
|
|||||||
cdef int l
|
cdef int l
|
||||||
cdef char *s
|
cdef char *s
|
||||||
if self._idx == NULL: return
|
if self._idx == NULL: return
|
||||||
|
if ((self.map_opt.flag & 4) and (self._idx.flag & 2)): return
|
||||||
s = cmappy.mappy_fetch_seq(self._idx, name.encode(), start, end, &l)
|
s = cmappy.mappy_fetch_seq(self._idx, name.encode(), start, end, &l)
|
||||||
if l == 0: return None
|
if l == 0: return None
|
||||||
r = s[:l] if isinstance(s, str) else s[:l].decode()
|
r = s[:l] if isinstance(s, str) else s[:l].decode()
|
||||||
|
|||||||
+7
-3
@@ -5,7 +5,7 @@ import getopt
|
|||||||
import mappy as mp
|
import mappy as mp
|
||||||
|
|
||||||
def main(argv):
|
def main(argv):
|
||||||
opts, args = getopt.getopt(argv[1:], "x:n:m:k:w:r:c")
|
opts, args = getopt.getopt(argv[1:], "x:n:m:k:w:r:cdM")
|
||||||
if len(args) < 2:
|
if len(args) < 2:
|
||||||
print("Usage: minimap2.py [options] <ref.fa>|<ref.mmi> <query.fq>")
|
print("Usage: minimap2.py [options] <ref.fa>|<ref.mmi> <query.fq>")
|
||||||
print("Options:")
|
print("Options:")
|
||||||
@@ -16,10 +16,12 @@ def main(argv):
|
|||||||
print(" -w INT minimizer window length")
|
print(" -w INT minimizer window length")
|
||||||
print(" -r INT band width")
|
print(" -r INT band width")
|
||||||
print(" -c output the cs tag")
|
print(" -c output the cs tag")
|
||||||
|
print(" -d output the ds tag")
|
||||||
|
print(" -M output the MD tag")
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
||||||
preset = min_cnt = min_sc = k = w = bw = None
|
preset = min_cnt = min_sc = k = w = bw = None
|
||||||
out_cs = False
|
out_cs = out_ds = out_MD = False
|
||||||
for opt, arg in opts:
|
for opt, arg in opts:
|
||||||
if opt == '-x': preset = arg
|
if opt == '-x': preset = arg
|
||||||
elif opt == '-n': min_cnt = int(arg)
|
elif opt == '-n': min_cnt = int(arg)
|
||||||
@@ -28,11 +30,13 @@ def main(argv):
|
|||||||
elif opt == '-k': k = int(arg)
|
elif opt == '-k': k = int(arg)
|
||||||
elif opt == '-w': w = int(arg)
|
elif opt == '-w': w = int(arg)
|
||||||
elif opt == '-c': out_cs = True
|
elif opt == '-c': out_cs = True
|
||||||
|
elif opt == '-d': out_ds = True
|
||||||
|
elif opt == '-M': out_MD = True
|
||||||
|
|
||||||
a = mp.Aligner(args[0], preset=preset, min_cnt=min_cnt, min_chain_score=min_sc, k=k, w=w, bw=bw)
|
a = mp.Aligner(args[0], preset=preset, min_cnt=min_cnt, min_chain_score=min_sc, k=k, w=w, bw=bw)
|
||||||
if not a: raise Exception("ERROR: failed to load/build index file '{}'".format(args[0]))
|
if not a: raise Exception("ERROR: failed to load/build index file '{}'".format(args[0]))
|
||||||
for name, seq, qual in mp.fastx_read(args[1]): # read one sequence
|
for name, seq, qual in mp.fastx_read(args[1]): # read one sequence
|
||||||
for h in a.map(seq, cs=out_cs): # traverse hits
|
for h in a.map(seq, cs=out_cs, ds=out_ds, MD=out_MD): # traverse hits
|
||||||
print('{}\t{}\t{}'.format(name, len(seq), h))
|
print('{}\t{}\t{}'.format(name, len(seq), h))
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
|
|||||||
@@ -2,6 +2,31 @@
|
|||||||
#include "kalloc.h"
|
#include "kalloc.h"
|
||||||
#include "ksort.h"
|
#include "ksort.h"
|
||||||
|
|
||||||
|
void mm_seed_mz_flt(void *km, mm128_v *mv, int32_t q_occ_max, float q_occ_frac)
|
||||||
|
{
|
||||||
|
mm128_t *a;
|
||||||
|
size_t i, j, st;
|
||||||
|
if (mv->n <= q_occ_max || q_occ_frac <= 0.0f || q_occ_max <= 0) return;
|
||||||
|
a = Kmalloc(km, mm128_t, mv->n);
|
||||||
|
for (i = 0; i < mv->n; ++i)
|
||||||
|
a[i].x = mv->a[i].x, a[i].y = i;
|
||||||
|
radix_sort_128x(a, a + mv->n);
|
||||||
|
for (st = 0, i = 1; i <= mv->n; ++i) {
|
||||||
|
if (i == mv->n || a[i].x != a[st].x) {
|
||||||
|
int32_t cnt = i - st;
|
||||||
|
if (cnt > q_occ_max && cnt > mv->n * q_occ_frac)
|
||||||
|
for (j = st; j < i; ++j)
|
||||||
|
mv->a[a[j].y].x = 0;
|
||||||
|
st = i;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
kfree(km, a);
|
||||||
|
for (i = j = 0; i < mv->n; ++i)
|
||||||
|
if (mv->a[i].x != 0)
|
||||||
|
mv->a[j++] = mv->a[i];
|
||||||
|
mv->n = j;
|
||||||
|
}
|
||||||
|
|
||||||
mm_seed_t *mm_seed_collect_all(void *km, const mm_idx_t *mi, const mm128_v *mv, int32_t *n_m_)
|
mm_seed_t *mm_seed_collect_all(void *km, const mm_idx_t *mi, const mm128_v *mv, int32_t *n_m_)
|
||||||
{
|
{
|
||||||
mm_seed_t *m;
|
mm_seed_t *m;
|
||||||
@@ -87,7 +112,8 @@ mm_seed_t *mm_collect_matches(void *km, int *_n_m, int qlen, int max_occ, int ma
|
|||||||
}
|
}
|
||||||
for (i = 0, n_m = 0, *rep_len = 0, *n_a = 0; i < n_m0; ++i) {
|
for (i = 0, n_m = 0, *rep_len = 0, *n_a = 0; i < n_m0; ++i) {
|
||||||
mm_seed_t *q = &m[i];
|
mm_seed_t *q = &m[i];
|
||||||
//fprintf(stderr, "X\t%d\t%d\t%d\n", q->q_pos>>1, q->n, q->flt);
|
if (mm_dbg_flag & MM_DBG_SEED_FREQ)
|
||||||
|
fprintf(stderr, "SF\t%d\t%d\t%d\n", q->q_pos>>1, q->n, q->flt);
|
||||||
if (q->flt) {
|
if (q->flt) {
|
||||||
int en = (q->q_pos >> 1) + 1, st = en - q->q_span;
|
int en = (q->q_pos >> 1) + 1, st = en - q->q_span;
|
||||||
if (st > rep_en) {
|
if (st > rep_en) {
|
||||||
|
|||||||
@@ -23,7 +23,7 @@ def readme():
|
|||||||
|
|
||||||
setup(
|
setup(
|
||||||
name = 'mappy',
|
name = 'mappy',
|
||||||
version = '2.22',
|
version = '2.31',
|
||||||
url = 'https://github.com/lh3/minimap2',
|
url = 'https://github.com/lh3/minimap2',
|
||||||
description = 'Minimap2 python binding',
|
description = 'Minimap2 python binding',
|
||||||
long_description = readme(),
|
long_description = readme(),
|
||||||
@@ -33,7 +33,7 @@ setup(
|
|||||||
keywords = 'sequence-alignment',
|
keywords = 'sequence-alignment',
|
||||||
scripts = ['python/minimap2.py'],
|
scripts = ['python/minimap2.py'],
|
||||||
ext_modules = [Extension('mappy',
|
ext_modules = [Extension('mappy',
|
||||||
sources = ['python/mappy.pyx', 'align.c', 'bseq.c', 'lchain.c', 'seed.c', 'format.c', 'hit.c', 'index.c', 'pe.c', 'options.c',
|
sources = ['python/mappy.pyx', 'align.c', 'bseq.c', 'lchain.c', 'seed.c', 'format.c', 'hit.c', 'index.c', 'pe.c', 'jump.c', 'options.c',
|
||||||
'ksw2_extd2_sse.c', 'ksw2_exts2_sse.c', 'ksw2_extz2_sse.c', 'ksw2_ll_sse.c',
|
'ksw2_extd2_sse.c', 'ksw2_exts2_sse.c', 'ksw2_extz2_sse.c', 'ksw2_ll_sse.c',
|
||||||
'kalloc.c', 'kthread.c', 'map.c', 'misc.c', 'sdust.c', 'sketch.c', 'esterr.c', 'splitidx.c'],
|
'kalloc.c', 'kthread.c', 'map.c', 'misc.c', 'sdust.c', 'sketch.c', 'esterr.c', 'splitidx.c'],
|
||||||
depends = ['minimap.h', 'bseq.h', 'kalloc.h', 'kdq.h', 'khash.h', 'kseq.h', 'ksort.h',
|
depends = ['minimap.h', 'bseq.h', 'kalloc.h', 'kdq.h', 'khash.h', 'kseq.h', 'ksort.h',
|
||||||
|
|||||||
@@ -0,0 +1,5 @@
|
|||||||
|
mm2: TGTTATCCCTAGGGTAACTTGTTCCGTTGGTCAAGTTATTGGATCAATTGAGTATAGTAGTGCACTCAC......................................................................................................................................CACTTGGAGCCATTCATACAGGTCCCTATTTAAGGAACAAGTGATTATGCTACCTTTGCACGGTT
|
||||||
|
||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||| |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
|
||||||
|
ref: TGTTATCCCTAGGGTAACTTGTTCCGTTGGTCAAGTTATTGGATCAATTGAGTATAGTAGTGCACTCACctGCTTCGCTTTGACTGGTGAAGTCTTAGCATGTACTGCTCGGAGGTTGGGTTCTGCTCCGAGGTCGCCCCAACCGAAATTTTTAATGCAGGTTTGGTAGTTTAGGACCTGTGGGTTTGTTAGGCTAACCTCacCACTTGGAGCCATTCATACAGGTCCCTATTTAAGGAACAAGTGATTATGCTACCTTTGCACGGTT
|
||||||
|
|||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||| ||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
|
||||||
|
sta: TGTTATCCCTAGGGTAACTTGTTCCGTTGGTCAAGTTATTGGATCAATTGAGTATAGTAGTGCA......................................................................................................................................CTCACCACTTGGAGCCATTCATACAGGTCCCTATTTAAGGAACAAGTGATTATGCTACCTTTGCACGGTT
|
||||||
@@ -0,0 +1,5 @@
|
|||||||
|
>query
|
||||||
|
AACCGTGCAAAGGTAGCATAATCACTTGTTCCTTAAATAGGGACCTGTATGAATGGCTCC
|
||||||
|
AAGTG
|
||||||
|
GTGAGTGCA
|
||||||
|
CTACTATACTCAATTGATCCAATAACTTGACCAACGGAACAAGTTACCCTAGGGATAACA
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
>ref
|
||||||
|
TGATCCAACATCGAGGTCGTAAACCCTATTGTTGATATGGACTCTAGAATAGGATTGCGC
|
||||||
|
TGTTATCCCTAGGGTAACTTGTTCCGTTGGTCAAGTTATTGGATCAATTGAGTATAGTAG
|
||||||
|
TGCACTCAC
|
||||||
|
ctGCTTCGCTTTGACTGGTGAAGTCTTAGCATGTACTGCTCGGAGGTTGGGTTCTGCTCC
|
||||||
|
GAGGTCGCCCCAACCGAAATTTTTAATGCAGGTTTGGTAGTTTAGGACCTGTGGGTTTGT
|
||||||
|
TAGGCTAACCTCac
|
||||||
|
CACTTGGAGCCATTCATACAGGTCCCTATTTAAGGAACAAGTGATTATGCTACCTTTGCA
|
||||||
|
CGGTTAGGGTACCGCGGCCGTTAAACATGTGTCACTGGGCAGGCGGTGCCTCTAATACTG
|
||||||
|
GTGAT
|
||||||
+10
-1
@@ -370,7 +370,6 @@
|
|||||||
year = {2020},
|
year = {2020},
|
||||||
doi = {10.1101/2020.11.01.363887},
|
doi = {10.1101/2020.11.01.363887},
|
||||||
publisher = {Cold Spring Harbor Laboratory},
|
publisher = {Cold Spring Harbor Laboratory},
|
||||||
abstract = {About 5-10\% of the human genome remains inaccessible for functional analysis due to the presence of repetitive sequences such as segmental duplications and tandem repeat arrays. To enable high-quality resequencing of personal genomes, it is crucial to support end-to-end genome variant discovery using repeat-aware read mapping methods. In this study, we highlight the fact that existing long read mappers often yield incorrect alignments and variant calls within long, near-identical repeats, as they remain vulnerable to allelic bias. In the presence of a non-reference allele within a repeat, a read sampled from that region could be mapped to an incorrect repeat copy because the standard pairwise sequence alignment scoring system penalizes true variants.To address the above problem, we propose a novel, long read mapping method that addresses allelic bias by making use of minimal confidently alignable substrings (MCASs). MCASs are formulated as minimal length substrings of a read that have unique alignments to a reference locus with sufficient mapping confidence (i.e., a mapping quality score above a user-specified threshold). This approach treats each read mapping as a collection of confident sub-alignments, which is more tolerant of structural variation and more sensitive to paralog-specific variants (PSVs) within repeats. We mathematically define MCASs and discuss an exact algorithm as well as a practical heuristic to compute them. The proposed method, referred to as Winnowmap2, is evaluated using simulated as well as real long read benchmarks using the recently completed gapless assemblies of human chromosomes X and 8 as a reference. We show that Winnowmap2 successfully addresses the issue of allelic bias, enabling more accurate downstream variant calls in repetitive sequences. As an example, using simulated PacBio HiFi reads and structural variants in chromosome 8, Winnowmap2 alignments achieved the lowest false-negative and false-positive rates (1.89\%, 1.89\%) for calling structural variants within near-identical repeats compared to minimap2 (39.62\%, 5.88\%) and NGMLR (56.60\%, 36.11\%) respectively.Winnowmap2 code is accessible at https://github.com/marbl/WinnowmapCompeting Interest StatementThe authors have declared no competing interest.},
|
|
||||||
URL = {https://www.biorxiv.org/content/early/2020/11/02/2020.11.01.363887},
|
URL = {https://www.biorxiv.org/content/early/2020/11/02/2020.11.01.363887},
|
||||||
eprint = {https://www.biorxiv.org/content/early/2020/11/02/2020.11.01.363887.full.pdf},
|
eprint = {https://www.biorxiv.org/content/early/2020/11/02/2020.11.01.363887.full.pdf},
|
||||||
journal = {bioRxiv}
|
journal = {bioRxiv}
|
||||||
@@ -449,3 +448,13 @@
|
|||||||
Title = {A synthetic-diploid benchmark for accurate variant-calling evaluation},
|
Title = {A synthetic-diploid benchmark for accurate variant-calling evaluation},
|
||||||
Volume = {15},
|
Volume = {15},
|
||||||
Year = {2018}}
|
Year = {2018}}
|
||||||
|
|
||||||
|
@article{Gu:1995wt,
|
||||||
|
author = {Gu, X and Li, W H},
|
||||||
|
journal = {J Mol Evol},
|
||||||
|
month = {Apr},
|
||||||
|
number = {4},
|
||||||
|
pages = {464-73},
|
||||||
|
title = {The size distribution of insertions and deletions in human and rodent pseudogenes suggests the logarithmic gap penalty for sequence alignment},
|
||||||
|
volume = {40},
|
||||||
|
year = {1995}}
|
||||||
|
|||||||
+60
-45
@@ -57,8 +57,8 @@ in v2.19 through v2.22 to improve mapping results.
|
|||||||
\begin{methods}
|
\begin{methods}
|
||||||
\section{Methods}
|
\section{Methods}
|
||||||
|
|
||||||
\subsection{Rescuing high-occurrence $k$-mers}
|
\subsection{Rescuing high-occurrence $k$-mers}\label{sec:high-occ}
|
||||||
Minimap2 keeps all $k$-mer minimizers during indexing. Its original
|
Minimap2 keeps all $k$-mer minimizers~\citep{Roberts:2004fv} during indexing. Its original
|
||||||
implementation only selected low-occurrence minimizers during mapping. The
|
implementation only selected low-occurrence minimizers during mapping. The
|
||||||
cutoff is a few hundred for mapping long reads against a human genome. If a
|
cutoff is a few hundred for mapping long reads against a human genome. If a
|
||||||
read habors only a few or even no low-occurrence minimizers, it will fail
|
read habors only a few or even no low-occurrence minimizers, it will fail
|
||||||
@@ -66,23 +66,24 @@ chaining due to insufficient anchors.
|
|||||||
|
|
||||||
To resolve this issue, we implemented a new heuristic to add additional
|
To resolve this issue, we implemented a new heuristic to add additional
|
||||||
minimizers. Suppose we are looking at two adjacent low-occurence $k$-mers
|
minimizers. Suppose we are looking at two adjacent low-occurence $k$-mers
|
||||||
located at position $x_1$ and $x_2$, respectively. If $|x_1-x_2|\ge500$,
|
located at position $x_1$ and $x_2$, respectively. If $|x_1-x_2|\ge L$,
|
||||||
minimap2 v2.22 additionally selects $\lfloor|x_1-x_2|/500\rfloor$ minimizers
|
minimap2 v2.22 additionally selects $\lfloor|x_1-x_2|/L\rfloor$ minimizers
|
||||||
of the lowest occurrence among minimizers between $x_1$ and $x_2$.
|
of the lowest occurrence among minimizers between $x_1$ and $x_2$. Here
|
||||||
We use a binary heap data
|
parameter $L$ controls the frequency of sampling. It defaults to 500.
|
||||||
structure to select minimizers of the lowest occurrence in this interval.
|
|
||||||
This strategy adds necessary anchors at the cost of increasing total alignment
|
This strategy adds necessary anchors at the cost of increasing total alignment
|
||||||
time by a few percent on real data.
|
time by a few percent on real data.
|
||||||
|
|
||||||
\subsection{Aligning through longer INDELs}
|
\subsection{Aligning through longer INDELs}
|
||||||
The original minimap2 may fail to align long INDELs due to its chaining
|
The original minimap2 may fail to align long INDELs due to its chaining
|
||||||
heuristics. Briefly, minimap2 applies dynamic programming (DP) to chain
|
heuristics. Briefly, minimap2 applies dynamic programming (DP) to chain
|
||||||
minimizer anchors. This is a quadratic algorithm, which is slow for chaining
|
minimizer anchors. This is a quadratic algorithm, slow for chaining
|
||||||
contigs. For acceptable performance, the original minimap2 uses a 500bp band by
|
contigs. For acceptable performance, the original minimap2 uses a 500bp band by
|
||||||
default. If there is an INDEL longer than 500bp and the two chains around the INDEL
|
default, which means a gap longer than 500bp will stop chaining.
|
||||||
|
To align through longer gaps, older minimap2 implemented a long-join heurstic as follows.
|
||||||
|
If there is an INDEL longer than 500bp and the two chains around the INDEL
|
||||||
have no overlaps on either the query or the reference sequence, minimap2 may
|
have no overlaps on either the query or the reference sequence, minimap2 may
|
||||||
join the two short chains later at a later step. We call it the
|
join the two short chains later.
|
||||||
long-join heuristic. This heuristic may fail around VNTRs because short chains
|
This heuristic may fail around VNTRs because short chains
|
||||||
often have overlaps in VNTRs. More subtly, minimap2 may escape the inner DP
|
often have overlaps in VNTRs. More subtly, minimap2 may escape the inner DP
|
||||||
loop early, again for performance, if the chaining result is not improved for
|
loop early, again for performance, if the chaining result is not improved for
|
||||||
50 iterations. When there is a copy number change in a long segmental
|
50 iterations. When there is a copy number change in a long segmental
|
||||||
@@ -90,13 +91,13 @@ duplication, the early escape may break around the event even if users
|
|||||||
specify a large band.
|
specify a large band.
|
||||||
|
|
||||||
In minigraph~\citep{Li:2020aa}, we developed a new chaining algorithm that
|
In minigraph~\citep{Li:2020aa}, we developed a new chaining algorithm that
|
||||||
finds short INDELs with DP-based chaining and goes through long INDELs with a
|
finds up to 1kb INDELs with DP-based chaining and goes through longer INDELs with a
|
||||||
subquadratic algorithm~\citep{DBLP:conf/wabi/AbouelhodaO03}. We ported the same
|
subquadratic algorithm~\citep{DBLP:conf/wabi/AbouelhodaO03}. We ported the same
|
||||||
algorithm to minimap2 for contig mapping. For long-read mapping, the minigraph
|
algorithm to minimap2 for contig mapping. For long-read mapping, the minigraph
|
||||||
algorithm is slower. Minimap2 v2.22 now still uses the DP-based algorithm to
|
algorithm is slower. Minimap2 v2.22 still uses the DP-based algorithm to
|
||||||
find short chains and then invokes the minigraph algorithm to rechain anchors in
|
find short chains and then invokes the minigraph algorithm to rechain anchors in
|
||||||
these short chains. The rechaining step achieves the same goal as long-join
|
these short chains. The rechaining step achieves the same goal as long-join
|
||||||
but is more reliable as it can resolve overlaps between short chains. The old
|
but is more reliable because it can resolve overlaps between short chains. The old
|
||||||
long-join heuristic has since been removed.
|
long-join heuristic has since been removed.
|
||||||
|
|
||||||
\subsection{Properly mapping long reads with SVs}
|
\subsection{Properly mapping long reads with SVs}
|
||||||
@@ -106,25 +107,25 @@ the best scoring alignment is sometimes not the correct alignment.
|
|||||||
\citet{Jain2020.11.01.363887} resolved this dilemma by altering the mapping
|
\citet{Jain2020.11.01.363887} resolved this dilemma by altering the mapping
|
||||||
algorithm.
|
algorithm.
|
||||||
|
|
||||||
In our view, this problem is rooted in impropriate scoring: affine-gap penalty
|
In our view, this problem is rooted in inapropriate scoring: affine-gap penalty
|
||||||
over-penalizes a long INDEL that was often evolutionarily created in one event.
|
over-penalizes a long INDEL that was often evolutionarily created in one event.
|
||||||
We should not penalize a SV linearly in its length. Minimap2 v2.22 rescores
|
We should not penalize a SV by a function linear in the SV length. Minimap2 v2.22 instead rescores
|
||||||
an alignment with the following scoring function. Suppose an alignment consists
|
an alignment with the following scoring function. Suppose an alignment consists
|
||||||
of $M$ matching bases, $N$ substitutions and $G$ gap opens, we empirically
|
of $M$ matching bases, $N$ substitutions and $G$ gap opens, we empirically
|
||||||
score the alignment with
|
score the alignment with
|
||||||
$$
|
$$
|
||||||
M-\frac{N+G}{2d}-\sum_{i=1}^G\log_2(1+g_i)
|
S=M-\frac{N+G}{2d}-\sum_{i=1}^G\log_2(1+g_i)
|
||||||
$$
|
$$
|
||||||
where $g_i\ge1$ is the length of the $i$-th gap and
|
where $g_i\ge1$ is the length of the $i$-th gap and
|
||||||
$$
|
$$
|
||||||
d=\max\left\{\frac{N+G}{M+N+G},0.02\right\}
|
d=\max\left\{\frac{N+G}{M+N+G},0.02\right\}
|
||||||
$$
|
$$
|
||||||
Here $d$ approximates per-base sequence divergence with the smallest value set
|
It approximates per-base sequence divergence except with the smallest value set
|
||||||
to 2\%. As an analogy to affine-gap scoring, the matching score in our scheme
|
to 2\%. As an analogy to affine-gap scoring, the matching score in our scheme
|
||||||
is 1, the mismatch and gap open penalties are both $1/2d$ and the gap extension
|
is 1, the mismatch and gap open penalties are both $1/2d$ and the gap extension
|
||||||
penalty is a logarithm function of the gap length. Our scoring gives a long SV
|
penalty is a logarithm function of the gap length~\citep{Gu:1995wt}. Our scoring gives a long SV
|
||||||
a much milder penalty. In terms of time complexity, scoring an alignment is
|
a much milder penalty. In terms of time complexity, scoring an alignment is
|
||||||
linear in the length of the alignment. Time spent on rescoring is negligible in
|
linear in the length of the alignment. The time spent on rescoring is negligible in
|
||||||
practice.
|
practice.
|
||||||
|
|
||||||
%If we assume sequences evolve under a duplication-mutation model, we may have a
|
%If we assume sequences evolve under a duplication-mutation model, we may have a
|
||||||
@@ -144,13 +145,15 @@ practice.
|
|||||||
\toprule
|
\toprule
|
||||||
$[$Benchmark$]$ Metric & v2.22 & v2.18 & Winno & lra \\
|
$[$Benchmark$]$ Metric & v2.22 & v2.18 & Winno & lra \\
|
||||||
\midrule
|
\midrule
|
||||||
$[$sim-map$]$ \% mapped reads at Q10 & 97.9 & 97.6 & {\bf 99.0} & 97.3 \\
|
$[$sim-map$]$ \% mapped reads at Q10 & 97.9 & 97.6 & {\bf 99.0}& 97.3 \\
|
||||||
$[$sim-map$]$ err. rate at Q10 (phredQ) & {\bf 52} & {\bf 52} & 38 & 24 \\
|
$[$sim-map$]$ err. rate at Q10 (phredQ) & {\bf 52} & {\bf 52} & 38 & 24 \\
|
||||||
$[$winno-cmp$]$ rate of diff. (phredQ) & {\bf 41} & 37 & N/A & 18 \\
|
$[$winno-cmp$]$ rate of diff. (phredQ) & {\bf 41} & 37 & truth & 18 \\
|
||||||
$[$sim-sv$]$ \% false negative rate & {\bf 0.5} & 2.0 & {\bf 0.5} & 1.4 \\
|
$[$winno-cmp$]$ CPU time (hour) & {\bf 5.0} & 5.3 & 71.8 & 13.1 \\
|
||||||
$[$sim-sv$]$ \% false discovery rate & {\bf 0.0} & 0.1 & {\bf 0.0} & 0.1 \\
|
$[$winno-cmp$]$ peak RAM (Gb) & 17.1 & 14.4 & {\bf 9.6} & 12.4 \\
|
||||||
$[$real-sv-1k$]$ \% false negative rate & {\bf 7.3} & 20.0 & 13.0 & N/A \\
|
$[$sim-sv$]$ \% false negative rate & {\bf 0.5} & 2.0 & {\bf 0.5} & 1.4 \\
|
||||||
$[$real-sv-1k$]$ \% false discovery rate & 2.7 & {\bf 2.4} & 2.7 & N/A \\
|
$[$sim-sv$]$ \% false discovery rate & {\bf 0.0} & 0.1 & {\bf 0.0} & 0.1 \\
|
||||||
|
$[$real-sv-1k$]$ \% false negative rate & {\bf 7.3} & 20.0 & 13.0 & N/A \\
|
||||||
|
$[$real-sv-1k$]$ \% false discovery rate & 2.7 & {\bf 2.4} & 2.7 & N/A \\
|
||||||
\botrule
|
\botrule
|
||||||
\end{tabular}}
|
\end{tabular}}
|
||||||
{In $[$sim-map$]$, 152,713 reads were simulated from the CHM13 telomere-to-telomere assembly v1.1
|
{In $[$sim-map$]$, 152,713 reads were simulated from the CHM13 telomere-to-telomere assembly v1.1
|
||||||
@@ -159,11 +162,11 @@ $[$real-sv-1k$]$ \% false discovery rate & 2.7 & {\bf 2.4} & 2.7 & N/A \\
|
|||||||
10 or higher were evaluated by ``paftools.js mapeval''. The mapping error rate
|
10 or higher were evaluated by ``paftools.js mapeval''. The mapping error rate
|
||||||
is measured in the phred scale: if the error rate is $e$, $-10\log_{10}e$ is
|
is measured in the phred scale: if the error rate is $e$, $-10\log_{10}e$ is
|
||||||
reported in the table. In $[$winno-cmp$]$, 1.39 million CHM13 HiFi reads from
|
reported in the table. In $[$winno-cmp$]$, 1.39 million CHM13 HiFi reads from
|
||||||
SRR11292121 were mapped against CHM13. 99.3\% of them were mapped by Winnowmap2
|
SRR11292121 were mapped against the same CHM13 assembly. 99.3\% of them were mapped by Winnowmap2
|
||||||
at mapping quality 10 or higher and were taken as ground truth to evaluate
|
at mapping quality 10 or higher and were taken as ground truth to evaluate
|
||||||
minimap2 and lra with ``paftools.js pafcmp''. $[$sim-sv$]$ simulated 1,000
|
minimap2 and lra with ``paftools.js pafcmp''. $[$sim-sv$]$ simulated 1,000
|
||||||
50bp to 1000bp INDELs from chr8 in CHM13 using SURVIVOR~\citep{Jeffares:2017aa} and simulated Nanopore
|
50bp to 1000bp INDELs from chr8 in CHM13 using SURVIVOR~\citep{Jeffares:2017aa} and simulated Nanopore
|
||||||
reads at 30 folds with the same pbsim2 command line. SVs were called with
|
reads at 30-fold coverage with the same pbsim2 command line. SVs were called with
|
||||||
``sniffles -q 10''~\citep{Sedlazeck:2018ab} and compared to the simulated truth with ``SURVIVOR eval
|
``sniffles -q 10''~\citep{Sedlazeck:2018ab} and compared to the simulated truth with ``SURVIVOR eval
|
||||||
call.vcf truth.bed 50''. In $[$real-sv-1k$]$, small and long variants were
|
call.vcf truth.bed 50''. In $[$real-sv-1k$]$, small and long variants were
|
||||||
called by dipcall-0.3~\citep{Li:2018aa} for HG002 assemblies (AC: GCA\_018852605.1 and
|
called by dipcall-0.3~\citep{Li:2018aa} for HG002 assemblies (AC: GCA\_018852605.1 and
|
||||||
@@ -172,31 +175,44 @@ GCA\_018852615.1) and compared to the GIAB truth~\citep{Zook:2020aa} using ``tru
|
|||||||
\end{table}
|
\end{table}
|
||||||
|
|
||||||
We evaluated minimap2 v2.22 along with v2.18, Winnowmap2 v2.03 and lra v1.3.2
|
We evaluated minimap2 v2.22 along with v2.18, Winnowmap2 v2.03 and lra v1.3.2
|
||||||
(Table~\ref{tab:1}). Both versions of minimap2 achieved high mapping accuracy on
|
(Table~\ref{tab:1}), using the default setting of each mapper according to the input data types.
|
||||||
|
Both versions of minimap2 achieved high mapping accuracy on
|
||||||
simulated Nanopore reads (sim-map). Winnowmap2 aligned more reads at mapping
|
simulated Nanopore reads (sim-map). Winnowmap2 aligned more reads at mapping
|
||||||
quality 10 or higher (mapQ10). However, it may occasionally assign a high mapping
|
quality 10 or higher (mapQ10). However, it may occasionally assign a high mapping
|
||||||
quality to a read with multiple identical best alignments. This reduced its
|
quality to a read with multiple identical best alignments. This reduced its
|
||||||
mapping accuracy.
|
mapping accuracy.
|
||||||
|
|
||||||
In lack of groud truth for real data, so we took Winnowmap2 mapping as ground
|
In lack of groud truth for real data, we took Winnowmap2 mapping as ground
|
||||||
truth to evaluate other mappers (winno-cmp). Out of 1,378,092 reads with mapQ10
|
truth to evaluate other mappers (winno-cmp in Table~\ref{tab:1}). Out of 1,378,092 reads with mapQ10
|
||||||
alignments by Winnowmap2, minimap2 v2.22 could map all of them. 118 reads, less
|
alignments by Winnowmap2, minimap2 v2.22 could map all of them. 118 reads, less
|
||||||
than 0.01\% of all reads, were mapped differently by v2.22. 51 of them have
|
than 0.01\% of all reads, were mapped differently by v2.22. 51 of them have
|
||||||
multiple identical best alignments. We believe these are more likely to be
|
multiple identical best alignments. We believe these are more likely to be
|
||||||
Winnowmap2 errors. Most of the remaining 67 (=118-51) reads have multiple
|
Winnowmap2 errors. Most of the remaining 67 (=118-51) reads have multiple
|
||||||
highly similar but not identical alignments. We are not sure what are real
|
highly similar but not identical alignments.
|
||||||
mapping errors.
|
Minimap2 v2.18 is less consistent with 275 differences including 30 unmapped
|
||||||
|
reads mappable by both Winnowmap2 and v2.22.
|
||||||
|
|
||||||
The two benchmarks above only evaluate read mappings without variations.
|
For the minimizer rescuing parameter $L$ in Section~\ref{sec:high-occ},
|
||||||
|
we set its default to 500 such that v2.22 has comparable performance to v2.18 given simulated PacBio and Nanopore human reads.
|
||||||
|
To see the effect of this parameter on real data, we tried several different $L$ values.
|
||||||
|
v2.22 gave 99 mapping differences at $L=200$,
|
||||||
|
118 at $L=500$ (default), 167 at $L=750$ and 224 differences at $L=1000$ in comparison to Winnowmap2.
|
||||||
|
$L=200$ is 28\% slower than the default while $L=1000$ is 9\% faster.
|
||||||
|
Changing the default minimizer window size (option ``-w'')
|
||||||
|
and the initial minimizer occurrence cutoff (option ``-f'')
|
||||||
|
also affects performance and accuracy to a similar magnitude.
|
||||||
|
|
||||||
|
The two benchmarks above only evaluate read mappings when there are no variations between the reads and the reference.
|
||||||
To measure the mapping accuracy in the presence of SVs (sim-sv), we reproduced
|
To measure the mapping accuracy in the presence of SVs (sim-sv), we reproduced
|
||||||
the results by~\citep{Jain2020.11.01.363887}. Minimap2 v2.22 is as good as
|
the results by~\citep{Jain2020.11.01.363887}. Minimap2 v2.22 is as good as
|
||||||
Winnowmap2 now. Note that we were setting the Sniffles mapping quality
|
Winnowmap2 now. Note that we were setting the Sniffles mapping quality
|
||||||
threshold to 10 in consistent with the benchmarks above. If we used the
|
threshold to 10 in consistent with the benchmarks above. If we used the
|
||||||
default threshold 20, v2.22 would miss additional 0.5\% SVs, suggesting
|
default threshold 20, v2.22 would miss additional five SVs (accounting for
|
||||||
minimap2 v2.22 could map variant reads correctly but with conservative mapping
|
0.5\% of simulated SVs). For four out of these five missing SVs, minimap2 v2.22
|
||||||
quality. This observation is more about the interaction between mappers and
|
mapped more variant reads than Winnowmap2. Sniffles did not call these SVs
|
||||||
callers. Furthermore, the simulation here only considers a simple scenario in
|
because minimap2 tended to give them conservative mapping quality. It is worth
|
||||||
evolution. Non-allelic gene conversions, which happen often in segmental
|
noting that the simulation here only considers a simple scenario in evolution.
|
||||||
|
Non-allelic gene conversions, which happen often in segmental
|
||||||
duplications~\citep{Harpak:2017aa}, would obscure the optimal mapping
|
duplications~\citep{Harpak:2017aa}, would obscure the optimal mapping
|
||||||
strategies. How much such simple SV simulation informs real-world SV calling
|
strategies. How much such simple SV simulation informs real-world SV calling
|
||||||
remains a question.
|
remains a question.
|
||||||
@@ -204,18 +220,17 @@ remains a question.
|
|||||||
To see if minimap2 v2.22 could improve long INDEL alignment, we ran dipcall on
|
To see if minimap2 v2.22 could improve long INDEL alignment, we ran dipcall on
|
||||||
contig-to-reference alignments and focused on INDELs longer than 1kb
|
contig-to-reference alignments and focused on INDELs longer than 1kb
|
||||||
(real-sv-1k). v2.22 is more sensitive at comparable specificity, confirming its
|
(real-sv-1k). v2.22 is more sensitive at comparable specificity, confirming its
|
||||||
advantage in more contiguous alignment. lra is supposed to handle long INDELs
|
advantage in more contiguous alignment. We could not get dipcall to work well with lra,
|
||||||
better, too. However, we could not get lra to work well with dipcall, so did
|
so did not report the numbers.
|
||||||
not report the numbers.
|
|
||||||
|
|
||||||
Minimap2 spends most computing time on base alignment. As recent improvements
|
Minimap2 spends most computing time on base alignment. As recent improvements
|
||||||
in v2.22 incur little additional computing and do not change the base alignment
|
in v2.22 incur little additional computing and do not change the base alignment
|
||||||
algorithm, the new version has similar performance to older verions. It is
|
algorithm, the new version has similar performance to older versions. It is
|
||||||
consistently faster than Winnowmap2 by several times. Sometimes simple
|
consistently faster than Winnowmap2 by several times. Sometimes simple
|
||||||
heuristics can be as effective as more sophisticated yet slower solutions.
|
heuristics can be as effective as more sophisticated yet slower solutions.
|
||||||
|
|
||||||
\section*{Acknowledgements}
|
\section*{Acknowledgements}
|
||||||
We thank Arang Rhie and Chirag Jain for providing motivating examples where
|
We thank Arang Rhie and Chirag Jain for providing motivating examples for which
|
||||||
older minimap2 underperforms.
|
older minimap2 underperforms.
|
||||||
|
|
||||||
\paragraph{Funding\textcolon} This work is funded by NHGRI grant R01HG010040.
|
\paragraph{Funding\textcolon} This work is funded by NHGRI grant R01HG010040.
|
||||||
|
|||||||
Reference in New Issue
Block a user