mirror of
https://github.com/lh3/minimap2.git
synced 2026-09-24 03:28:12 +08:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1fd85be6e2 | ||
|
|
e616b0dacf | ||
|
|
b58b97423a | ||
|
|
df9e650346 | ||
|
|
7a540c37ca | ||
|
|
c19e3ccb86 | ||
|
|
94d171b01e | ||
|
|
ff312a2957 | ||
|
|
01ccedd5a0 | ||
|
|
819b3bf017 | ||
|
|
e88110463a | ||
|
|
fb81e150f2 | ||
|
|
d930ea94ad | ||
|
|
a832a42f6f | ||
|
|
a5411fc3c0 | ||
|
|
bd03d975fc | ||
|
|
9c3c4b1ce8 | ||
|
|
3542a3d153 | ||
|
|
75619c7b51 | ||
|
|
fbb9c0fcba | ||
|
|
af094640e5 | ||
|
|
d43f356ef9 | ||
|
|
38acd6617f | ||
|
|
3d351267a0 | ||
|
|
54a4718c9b | ||
|
|
dbc12b2838 | ||
|
|
2ed264db4e | ||
|
|
a8094ad859 | ||
|
|
1877818239 | ||
|
|
9ede5c4255 | ||
|
|
405511fe8d | ||
|
|
dd90d9dde6 | ||
|
|
a8c567b5e9 | ||
|
|
d9d3c0cc3f | ||
|
|
cbe8d61ca4 | ||
|
|
9d06cef13e | ||
|
|
924fc4d671 | ||
|
|
fc2d1e95b3 | ||
|
|
bbf0bb871b | ||
|
|
83e9b2e28c | ||
|
|
e816fd071c | ||
|
|
a955b1f31d | ||
|
|
54fa925e2e | ||
|
|
d4a396c5c3 | ||
|
|
bdf46f5786 | ||
|
|
f536b69b81 | ||
|
|
4b8b4418df | ||
|
|
ce30004e02 | ||
|
|
54f8e5f7d6 | ||
|
|
2857de7dbd | ||
|
|
e18935fbad | ||
|
|
74ebfb2532 | ||
|
|
a0cbe2e4d2 | ||
|
|
618d33515e | ||
|
|
a10d4f4496 | ||
|
|
1eea2fee11 | ||
|
|
1d346d56bc | ||
|
|
46750de966 | ||
|
|
7d8bbb74a8 | ||
|
|
c6db201b38 | ||
|
|
358a39850f | ||
|
|
fcb5d5e6eb | ||
|
|
95807a2224 | ||
|
|
a4c93e9377 | ||
|
|
7d69334e69 | ||
|
|
3e1ab2951d | ||
|
|
68179ed195 | ||
|
|
d1f4c8d232 | ||
|
|
8efe83b744 | ||
|
|
042c8d4d71 | ||
|
|
e4e1f7843b | ||
|
|
69e3629916 | ||
|
|
0cc3cdca27 | ||
|
|
8170693de3 | ||
|
|
e3d8c708ac | ||
|
|
119bdc6029 | ||
|
|
89d4d219cd | ||
|
|
f51ff1abac | ||
|
|
27b254ed6f | ||
|
|
c881b14ba5 | ||
|
|
f18dadb1c4 | ||
|
|
a83b8fe7cc | ||
|
|
c22bfe7722 | ||
|
|
12d441ea22 | ||
|
|
c7433c2811 | ||
|
|
5279377544 | ||
|
|
acab05781e | ||
|
|
98c23bc6d2 | ||
|
|
9b0ff2418c | ||
|
|
b6762503a9 | ||
|
|
9667468e89 | ||
|
|
ba60aac6f6 | ||
|
|
fcd4df2a73 | ||
|
|
0efc886012 | ||
|
|
940388f8e4 | ||
|
|
23d2674c39 | ||
|
|
a12673611f | ||
|
|
8140259974 | ||
|
|
f3e59fc2a0 | ||
|
|
fc2e1607d7 | ||
|
|
bc588c0eeb | ||
|
|
ab717023b6 | ||
|
|
9506e7ac3f | ||
|
|
ce03fbc275 | ||
|
|
98a3aa1b39 | ||
|
|
ae05f8485f | ||
|
|
ace990c381 | ||
|
|
e28a55be86 | ||
|
|
f8d46a7a30 | ||
|
|
4483f89ee5 | ||
|
|
f1b3c7ad06 | ||
|
|
180faa3594 | ||
|
|
704fbc6f5c | ||
|
|
fc24c8a348 | ||
|
|
e68d868806 | ||
|
|
c3d461e22a | ||
|
|
c41518ae85 | ||
|
|
819d843e3c | ||
|
|
5e7242303c | ||
|
|
a026c69b89 | ||
|
|
ea2042a577 | ||
|
|
1834b1fd42 | ||
|
|
7ced0f16a0 | ||
|
|
35732f3025 | ||
|
|
a6fab118c5 | ||
|
|
6ce0dd8b70 | ||
|
|
1d3c3eef03 | ||
|
|
01b98e8e52 | ||
|
|
16b8d50199 | ||
|
|
226fd6114c | ||
|
|
822ccd1733 | ||
|
|
f67849c9af | ||
|
|
b0b199f503 | ||
|
|
c2f07ff2ac | ||
|
|
cefd0d9f6c | ||
|
|
2a319c89aa | ||
|
|
85a5260408 | ||
|
|
6c2cbf7903 | ||
|
|
5aa4355ca8 | ||
|
|
6ed7263670 | ||
|
|
315795eefd | ||
|
|
fc6869a9e8 | ||
|
|
6252e5e367 | ||
|
|
843729df1e | ||
|
|
195c98fa46 | ||
|
|
a2e6659d9b | ||
|
|
e450f161bb | ||
|
|
15cade0f06 | ||
|
|
767556b6f0 | ||
|
|
50a26a60a6 | ||
|
|
31de4fd1bc | ||
|
|
e018caea32 | ||
|
|
7c02742fa8 | ||
|
|
a41f5d1eeb | ||
|
|
ed3d0eb328 | ||
|
|
e6d166a314 | ||
|
|
06fedaadd0 | ||
|
|
fe35e679e9 | ||
|
|
e25aa5ee74 | ||
|
|
3bde3450a0 | ||
|
|
36942ff711 | ||
|
|
d3a89d34d4 | ||
|
|
c8f0a35c40 | ||
|
|
fcaadc22b7 | ||
|
|
db37fc43a7 | ||
|
|
a8f1fa8ea3 | ||
|
|
b276772890 |
@@ -7,7 +7,8 @@ on:
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
build:
|
||||
build-linux-x8664:
|
||||
name: Linux x86_64
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
matrix:
|
||||
@@ -15,7 +16,53 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout minimap2
|
||||
uses: actions/checkout@v2
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Compile with ${{ matrix.compiler }}
|
||||
run: make CC=${{ matrix.compiler }}
|
||||
run: |
|
||||
make CC=${{ matrix.compiler }}
|
||||
file minimap2 | grep x86-64
|
||||
|
||||
build-linux-aarch64:
|
||||
name: Linux aarch64
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
matrix:
|
||||
compiler: [gcc]
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Compile with ${{ matrix.compiler }}
|
||||
uses: uraimo/run-on-arch-action@v2
|
||||
with:
|
||||
arch: aarch64
|
||||
distro: ubuntu22.04
|
||||
githubToken: ${{ github.token }}
|
||||
dockerRunArgs: |
|
||||
--volume "${PWD}:/minimap2"
|
||||
install: |
|
||||
apt-get update -q -y
|
||||
apt-get install -q -y make ${{ matrix.compiler }} zlib1g-dev file
|
||||
run: |
|
||||
cd /minimap2
|
||||
make CC=${{ matrix.compiler }} arm_neon=1 aarch64=1 -j
|
||||
file minimap2 | grep aarch64
|
||||
|
||||
build-mac-arm64:
|
||||
name: Mac ARM64
|
||||
runs-on: macos-14
|
||||
strategy:
|
||||
matrix:
|
||||
compiler: [clang]
|
||||
|
||||
steps:
|
||||
- name: Checkout minimap2
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Compile with ${{ matrix.compiler }}
|
||||
run: |
|
||||
make CC=${{ matrix.compiler }} arm_neon=1 aarch64=1 -j
|
||||
file minimap2 | grep arm64
|
||||
|
||||
-24
@@ -1,24 +0,0 @@
|
||||
matrix:
|
||||
include:
|
||||
- language: c
|
||||
compiler: gcc
|
||||
script: make
|
||||
- language: c
|
||||
compiler: clang
|
||||
script: make
|
||||
- arch: arm64
|
||||
language: c
|
||||
compiler: gcc
|
||||
script: make arm_neon=1 aarch64=1
|
||||
- language: python
|
||||
python: "2.7"
|
||||
before_install: pip install cython
|
||||
script: python setup.py build_ext
|
||||
- language: python
|
||||
python: "3.5"
|
||||
before_install: pip install cython
|
||||
script: python setup.py build_ext
|
||||
- language: python
|
||||
python: "3.9"
|
||||
before_install: pip install cython
|
||||
script: python setup.py build_ext
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
Without `-a`, `-c` or `--cs`, minimap2 only finds *approximate* mapping
|
||||
locations without detailed base alignment. In particular, the start and end
|
||||
positions of the alignment are impricise. With one of those options, minimap2
|
||||
positions of the alignment are imprecise. With one of those options, minimap2
|
||||
will perform base alignment, which is generally more accurate but is much
|
||||
slower.
|
||||
|
||||
|
||||
@@ -2,12 +2,16 @@ CFLAGS= -g -Wall -O2 -Wc++-compat #-Wextra
|
||||
CPPFLAGS= -DHAVE_KALLOC
|
||||
INCLUDES=
|
||||
OBJS= kthread.o kalloc.o misc.o bseq.o sketch.o sdust.o options.o index.o \
|
||||
lchain.o align.o hit.o seed.o map.o format.o pe.o esterr.o splitidx.o \
|
||||
lchain.o align.o hit.o seed.o jump.o map.o format.o pe.o esterr.o splitidx.o \
|
||||
ksw2_ll_sse.o
|
||||
PROG= minimap2
|
||||
PROG_EXTRA= sdust minimap2-lite
|
||||
LIBS= -lm -lz -lpthread
|
||||
|
||||
ifneq ($(aarch64),)
|
||||
arm_neon=1
|
||||
endif
|
||||
|
||||
ifeq ($(arm_neon),) # if arm_neon is not defined
|
||||
ifeq ($(sse2only),) # if sse2only is not defined
|
||||
OBJS+=ksw2_extz2_sse41.o ksw2_extd2_sse41.o ksw2_exts2_sse41.o ksw2_extz2_sse2.o ksw2_extd2_sse2.o ksw2_exts2_sse2.o ksw2_dispatch.o
|
||||
@@ -26,12 +30,12 @@ endif
|
||||
|
||||
ifneq ($(asan),)
|
||||
CFLAGS+=-fsanitize=address
|
||||
LIBS+=-fsanitize=address
|
||||
LIBS+=-fsanitize=address -ldl
|
||||
endif
|
||||
|
||||
ifneq ($(tsan),)
|
||||
CFLAGS+=-fsanitize=thread
|
||||
LIBS+=-fsanitize=thread
|
||||
LIBS+=-fsanitize=thread -ldl
|
||||
endif
|
||||
|
||||
.PHONY:all extra clean depend
|
||||
@@ -111,8 +115,9 @@ esterr.o: mmpriv.h minimap.h bseq.h kseq.h
|
||||
example.o: minimap.h kseq.h
|
||||
format.o: kalloc.h mmpriv.h minimap.h bseq.h kseq.h
|
||||
hit.o: mmpriv.h minimap.h bseq.h kseq.h kalloc.h khash.h
|
||||
index.o: kthread.h bseq.h minimap.h mmpriv.h kseq.h kvec.h kalloc.h khash.h
|
||||
index.o: ksort.h
|
||||
index.o: kthread.h bseq.h minimap.h mmpriv.h kseq.h ksw2.h kalloc.h kvec.h
|
||||
index.o: khash.h ksort.h
|
||||
jump.o: mmpriv.h minimap.h bseq.h kseq.h
|
||||
kalloc.o: kalloc.h
|
||||
ksw2_extd2_sse.o: ksw2.h kalloc.h
|
||||
ksw2_exts2_sse.o: ksw2.h kalloc.h
|
||||
|
||||
@@ -1,3 +1,170 @@
|
||||
Release 2.29-r1283 (18 April 2025)
|
||||
----------------------------------
|
||||
|
||||
Notable changes to minimap2:
|
||||
|
||||
* New feature: added the `splice:sr` preset for short RNA-seq read alignment.
|
||||
Users may use `-j` to specify known gene annotation to improve spliced
|
||||
alignment close to the ends of short reads. Also added `--write-junc` and
|
||||
`--pass1` for 2-pass short-read RNA-seq alignment.
|
||||
|
||||
* Experimental feature: read splice scores from a file specified by `--spsc`
|
||||
and consider the scores during base alignment. The feature makes it possible
|
||||
to apply advanced splice models and to improve spliced alignment.
|
||||
|
||||
* Change: adjusted the mapping quality calculation for spliced alignment.
|
||||
|
||||
* Bugfixes: a) missing overlap alignment when base alignment is requested
|
||||
(#969); b) incorrect summary information for long genomes (#1192); c)
|
||||
missing parameter check for `--score-N` (#1226).
|
||||
|
||||
* Improvement: a) warn about absent junction files (#1229); b) report an error
|
||||
if a wrong preset prefixed with "splice" is specified (#589).
|
||||
|
||||
Notable changes to mappy:
|
||||
|
||||
* Improvement: allow passing read name (#1260)
|
||||
|
||||
* Improvement: exposed score for ambiguous bases (#1240)
|
||||
|
||||
Minimap2 now supports short/long genomic/RNA-seq read alignment along with
|
||||
contig alignment and all-vs-all read overlapping. It produces identical genomic
|
||||
long-read or contig alignment to v2.27. Short genomic read alignment and the
|
||||
mapping quality of long RNA-seq read alignment may slightly differ in very rare
|
||||
cases.
|
||||
|
||||
(2.29: 18 April 2025, r1283)
|
||||
|
||||
|
||||
|
||||
Release 2.28-r1209 (27 March 2024)
|
||||
----------------------------------
|
||||
|
||||
Notable changes to minimap2:
|
||||
|
||||
* Bugfix: `--MD` was not working properly due to the addition of `--ds` in the
|
||||
last release (#1181 and #1182).
|
||||
|
||||
* New feature: added an experimental preset `lq:hqae` for aligning accurate
|
||||
long reads back to their assembly. It has been observed that `map-hifi` and
|
||||
`lr:hq` may produce many wrong alignments around centromeres when accurate
|
||||
long reads (PacBio HiFi or Nanopore duplex/Q20+) are mapped to a diploid
|
||||
assembly constructed from them. This new preset produces much more accurate
|
||||
alignment. It is still experimental and may be subjective to changes in
|
||||
future.
|
||||
|
||||
* Change: reduced the default `--cap-kalloc` to 500m to lower the peak
|
||||
memory consumption (#855).
|
||||
|
||||
Notable changes to mappy:
|
||||
|
||||
* Bugfix: mappy option struct was out of sync with minimap2 (#1177).
|
||||
|
||||
Minimap2 should output identical alignments to v2.27.
|
||||
|
||||
(2.28: 27 March 2024, r1209)
|
||||
|
||||
|
||||
|
||||
Release 2.27-r1193 (12 March 2024)
|
||||
----------------------------------
|
||||
|
||||
Notable changes to minimap2:
|
||||
|
||||
* New feature: added the `lr:hq` preset for accurate long reads at ~1% error
|
||||
rate. This was suggested by Oxford Nanopore developers (#1127). It is not
|
||||
clear if this preset also works well for PacBio HiFi reads.
|
||||
|
||||
* New feature: added the `map-iclr` preset for Illumina Complete Long Reads
|
||||
(#1069), provided by Illumina developers.
|
||||
|
||||
* New feature: added option `-b` to specify mismatch penalty for base
|
||||
transitions (i.e. A-to-G or C-to-T changes).
|
||||
|
||||
* New feature: added option `--ds` to generate a new `ds:Z` tag that
|
||||
indicates uncertainty in INDEL positions. It is an extension to `cs`. The
|
||||
`mgutils-es6.js` script in minigraph parses `ds`.
|
||||
|
||||
* Bugfix: avoided a NULL pointer dereference (#1154). This would not have an
|
||||
effect on most systems but would still be good to fix.
|
||||
|
||||
* Bugfix: reverted the value of `ms:i` to pre-2.22 versions (#1146). This was
|
||||
an oversight. See fcd4df2 for details.
|
||||
|
||||
Notable changes to paftools.js and mappy:
|
||||
|
||||
* New feature: expose `bw_long` to mappy's Aligner class (#1124).
|
||||
|
||||
* Bugfix: fixed several compatibility issues with k8 v1.0 (#1161 and #1166).
|
||||
Subcommands "call", "pbsim2fq" and "mason2fq" were not working with v1.0.
|
||||
|
||||
Minimap2 should output identical alignments to v2.26, except the ms tag.
|
||||
|
||||
(2.27: 12 March 2024, r1193)
|
||||
|
||||
|
||||
|
||||
Release 2.26-r1175 (29 April 2023)
|
||||
----------------------------------
|
||||
|
||||
Fixed the broken Python package. This is the only change.
|
||||
|
||||
(2.26: 25 April 2023, r1173)
|
||||
|
||||
|
||||
|
||||
Release 2.25-r1173 (25 April 2023)
|
||||
----------------------------------
|
||||
|
||||
Notable changes:
|
||||
|
||||
* Improvement: use the miniprot splice model for RNA-seq alignment by default.
|
||||
This model considers non-GT-AG splice sites and leads to slightly higher
|
||||
(<0.1%) accuracy and sensitivity on real human data.
|
||||
|
||||
* Change: increased the default `-I` to `8G` such that minimap2 would create a
|
||||
uni-part index for a pair of mammalian genomes. This change may increase the
|
||||
memory for all-vs-all read overlap alignment given large datasets.
|
||||
|
||||
* New feature: output the sequences in secondary alignments with option
|
||||
`--secondary-seq` (#687).
|
||||
|
||||
* Bugfix: --rmq was not parsed correctly (#1010)
|
||||
|
||||
* Bugfix: possibly incorrect coordinate when applying end bonus to the target
|
||||
sequence (#1025). This is a ksw2 bug. It does not affect minimap2 as
|
||||
minimap2 is not using the affected feature.
|
||||
|
||||
* Improvement: incorporated several changes for better compatibility with
|
||||
Windows (#1051) and for minimap2 integration at Oxford Nanopore Technologies
|
||||
(#1048 and #1033).
|
||||
|
||||
* Improvement: output the HD-line in SAM output (#1019).
|
||||
|
||||
* Improvement: check minimap2 index file in mappy to prevent segmentation
|
||||
fault for certain indices (#1008).
|
||||
|
||||
For genomic sequences, minimap2 should give identical output to v2.24.
|
||||
Long-read RNA-seq alignment may occasionally differ from previous versions.
|
||||
|
||||
(2.25: 25 April 2023, r1173)
|
||||
|
||||
|
||||
|
||||
Release 2.24-r1122 (26 December 2021)
|
||||
-------------------------------------
|
||||
|
||||
This release improves alignment around long poorly aligned regions. Older
|
||||
minimap2 may chain through such regions in rare cases which may result in
|
||||
missing alignments later. The issue has become worse since the the change of
|
||||
the chaining algorithm in v2.19. v2.23 implements an incomplete remedy. This
|
||||
release provides a better solution with a X-drop-like heuristic and by enabling
|
||||
two-bandwidth chaining in the assembly mode.
|
||||
|
||||
(2.24: 26 December 2021, r1122)
|
||||
|
||||
|
||||
|
||||
Release 2.23-r1111 (18 November 2021)
|
||||
-------------------------------------
|
||||
|
||||
|
||||
@@ -14,13 +14,15 @@ cd minimap2 && make
|
||||
# use presets (no test data)
|
||||
./minimap2 -ax map-pb ref.fa pacbio.fq.gz > aln.sam # PacBio CLR genomic reads
|
||||
./minimap2 -ax map-ont ref.fa ont.fq.gz > aln.sam # Oxford Nanopore genomic reads
|
||||
./minimap2 -ax map-hifi ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.19 or later)
|
||||
./minimap2 -ax asm20 ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.18 or earlier)
|
||||
./minimap2 -ax map-hifi ref.fa pacbio-ccs.fq.gz > aln.sam # PacBio HiFi/CCS genomic reads (v2.19+)
|
||||
./minimap2 -ax lr:hq ref.fa ont-Q20.fq.gz > aln.sam # Nanopore Q20 genomic reads (v2.27+)
|
||||
./minimap2 -ax sr ref.fa read1.fa read2.fa > aln.sam # short genomic paired-end reads
|
||||
./minimap2 -ax splice ref.fa rna-reads.fa > aln.sam # spliced long reads (strand unknown)
|
||||
./minimap2 -ax splice -uf -k14 ref.fa reads.fa > aln.sam # noisy Nanopore Direct RNA-seq
|
||||
./minimap2 -ax splice:hq -uf ref.fa query.fa > aln.sam # Final PacBio Iso-seq or traditional cDNA
|
||||
./minimap2 -ax splice --junc-bed anno.bed12 ref.fa query.fa > aln.sam # prioritize on annotated junctions
|
||||
./minimap2 -ax splice -uf -k14 ref.fa reads.fa > aln.sam # noisy Nanopore direct RNA-seq
|
||||
./minimap2 -ax splice:hq -uf ref.fa query.fa > aln.sam # PacBio Kinnex/Iso-seq (RNA-seq)
|
||||
./minimap2 -ax splice --junc-bed=anno.bed12 ref.fa query.fa > aln.sam # use annotated junctions
|
||||
./minimap2 -ax splice:sr ref.fa r1.fq r2.fq > aln.sam # short-read RNA-seq (v2.29+)
|
||||
./minimap2 -ax splice:sr -j anno.bed12 ref.fa r1.fq r2.fq > aln.sam
|
||||
./minimap2 -cx asm5 asm1.fa asm2.fa > aln.paf # intra-species asm-to-asm alignment
|
||||
./minimap2 -x ava-pb reads.fa reads.fa > overlaps.paf # PacBio read overlap
|
||||
./minimap2 -x ava-ont reads.fa reads.fa > overlaps.paf # Nanopore read overlap
|
||||
@@ -38,7 +40,8 @@ man ./minimap2.1
|
||||
- [Map long noisy genomic reads](#map-long-genomic)
|
||||
- [Map long mRNA/cDNA reads](#map-long-splice)
|
||||
- [Find overlaps between long reads](#long-overlap)
|
||||
- [Map short accurate genomic reads](#short-genomic)
|
||||
- [Map short genomic reads](#short-genomic)
|
||||
- [Map short RNA-seq reads](#short-rna-seq)
|
||||
- [Full genome/assembly alignment](#full-genome)
|
||||
- [Advanced features](#advanced)
|
||||
- [Working with >65535 CIGAR operations](#long-cigar)
|
||||
@@ -74,8 +77,8 @@ Detailed evaluations are available from the [minimap2 paper][doi] or the
|
||||
Minimap2 is optimized for x86-64 CPUs. You can acquire precompiled binaries from
|
||||
the [release page][release] with:
|
||||
```sh
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.23/minimap2-2.23_x64-linux.tar.bz2 | tar -jxvf -
|
||||
./minimap2-2.23_x64-linux/minimap2
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.29/minimap2-2.29_x64-linux.tar.bz2 | tar -jxvf -
|
||||
./minimap2-2.29_x64-linux/minimap2
|
||||
```
|
||||
If you want to compile from the source, you need to have a C compiler, GNU make
|
||||
and zlib development files installed. Then type `make` in the source code
|
||||
@@ -139,12 +142,15 @@ parameters at the same time. The default setting is the same as `map-ont`.
|
||||
```sh
|
||||
minimap2 -ax map-pb ref.fa pacbio-reads.fq > aln.sam # for PacBio CLR reads
|
||||
minimap2 -ax map-ont ref.fa ont-reads.fq > aln.sam # for Oxford Nanopore reads
|
||||
minimap2 -ax map-iclr ref.fa iclr-reads.fq > aln.sam # for Illumina Complete Long Reads
|
||||
```
|
||||
The difference between `map-pb` and `map-ont` is that `map-pb` uses
|
||||
homopolymer-compressed (HPC) minimizers as seeds, while `map-ont` uses ordinary
|
||||
minimizers as seeds. Emperical evaluation suggests HPC minimizers improve
|
||||
minimizers as seeds. Empirical evaluation suggests HPC minimizers improve
|
||||
performance and sensitivity when aligning PacBio CLR reads, but hurt when aligning
|
||||
Nanopore reads.
|
||||
Nanopore reads. `map-iclr` uses an adjusted alignment scoring matrix that
|
||||
accounts for the low overall error rate in the reads, with transversion errors
|
||||
being less frequent than transitions.
|
||||
|
||||
#### <a name="map-long-splice"></a>Map long mRNA/cDNA reads
|
||||
|
||||
@@ -168,9 +174,8 @@ or the last exons.
|
||||
|
||||
Minimap2 rates an alignment by the score of the max-scoring sub-segment,
|
||||
*excluding* introns, and marks the best alignment as primary in SAM. When a
|
||||
spliced gene also has unspliced pseudogenes, minimap2 does not intentionally
|
||||
prefer spliced alignment, though in practice it more often marks the spliced
|
||||
alignment as the primary. By default, minimap2 outputs up to five secondary
|
||||
spliced gene also has unspliced pseudogenes, minimap2 slightly prefers
|
||||
the spliced alignment. By default, minimap2 outputs up to five secondary
|
||||
alignments (i.e. likely pseudogenes in the context of RNA-seq mapping). This
|
||||
can be tuned with option **-N**.
|
||||
|
||||
@@ -201,6 +206,10 @@ bonus score (tuned by `--junc-bonus`) if an aligned junction matches a junction
|
||||
in the annotation. Option `--junc-bed` also takes 5-column BED, including the
|
||||
strand field. In this case, each line indicates an oriented junction.
|
||||
|
||||
**Note:** `--junc-bed` is intended for long noisy RNA-seq reads only.
|
||||
Applying the option to short RNA-seq reads would increase run time with little
|
||||
improvement to junction accuracy.
|
||||
|
||||
#### <a name="long-overlap"></a>Find overlaps between long reads
|
||||
|
||||
```sh
|
||||
@@ -213,7 +222,7 @@ the overlapping mode because it is slow and may produce false positive
|
||||
overlaps. However, if performance is not a concern, you may try to add `-a` or
|
||||
`-c` anyway.
|
||||
|
||||
#### <a name="short-genomic"></a>Map short accurate genomic reads
|
||||
#### <a name="short-genomic"></a>Map short genomic reads
|
||||
|
||||
```sh
|
||||
minimap2 -ax sr ref.fa reads-se.fq > aln.sam # single-end alignment
|
||||
@@ -226,8 +235,18 @@ be paired if they are adjacent in the input stream and have the same name (with
|
||||
the `/[0-9]` suffix trimmed if present). Single- and paired-end reads can be
|
||||
mixed.
|
||||
|
||||
Minimap2 does not work well with short spliced reads. There are many capable
|
||||
RNA-seq mappers for short reads.
|
||||
#### <a name="short-rna-seq"></a>Map short RNA-seq reads
|
||||
|
||||
```sh
|
||||
minimap2 -ax splice:sr ref.fa reads-se.fq.gz > aln.sam # single-end
|
||||
minimap2 -ax splice:sr ref.fa r1.fq.gz r2.fq.gz > aln.sam # paired-end
|
||||
minimap2 -ax splice:sr -j anno.bed ref.fa r1.fq r2.fq > aln.sam # use annotation
|
||||
# 2-pass alignment
|
||||
minimap2 -x splice:sr -j anno.bed --write-junc ref.fa r1.fq r2.fq > junc.bed
|
||||
minimap2 -ax splice:sr -j anno.bed --pass1=junc.bed ref.fa r1.fq r2.fq > aln.sam
|
||||
```
|
||||
The new preset `splice:sr` was added in v2.29. It functions similarly to `sr`
|
||||
except that it performs spliced alignment.
|
||||
|
||||
#### <a name="full-genome"></a>Full genome/assembly alignment
|
||||
|
||||
@@ -350,6 +369,11 @@ If you use minimap2 in your work, please cite:
|
||||
> Li, H. (2018). Minimap2: pairwise alignment for nucleotide sequences.
|
||||
> *Bioinformatics*, **34**:3094-3100. [doi:10.1093/bioinformatics/bty191][doi]
|
||||
|
||||
and/or:
|
||||
|
||||
> Li, H. (2021). New strategies to improve minimap2 alignment accuracy.
|
||||
> *Bioinformatics*, **37**:4572-4574. [doi:10.1093/bioinformatics/btab705][doi2]
|
||||
|
||||
## <a name="dguide"></a>Developers' Guide
|
||||
|
||||
Minimap2 is not only a command line tool, but also a programming library.
|
||||
@@ -399,5 +423,6 @@ mappy` or [from BioConda][mappyconda] via `conda install -c bioconda mappy`.
|
||||
[manpage]: https://lh3.github.io/minimap2/minimap2.html
|
||||
[manpage-cs]: https://lh3.github.io/minimap2/minimap2.html#10
|
||||
[doi]: https://doi.org/10.1093/bioinformatics/bty191
|
||||
[smide]: https://github.com/nemequ/simde
|
||||
[doi2]: https://doi.org/10.1093/bioinformatics/btab705
|
||||
[simde]: https://github.com/nemequ/simde
|
||||
[unimap]: https://github.com/lh3/unimap
|
||||
|
||||
@@ -6,6 +6,8 @@
|
||||
#include "mmpriv.h"
|
||||
#include "ksw2.h"
|
||||
|
||||
#define MM_MAX_QLEN_FLANK 100
|
||||
|
||||
static void ksw_gen_simple_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t sc_ambi)
|
||||
{
|
||||
int i, j;
|
||||
@@ -21,6 +23,18 @@ static void ksw_gen_simple_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t sc
|
||||
mat[(m - 1) * m + j] = sc_ambi;
|
||||
}
|
||||
|
||||
static void ksw_gen_ts_mat(int m, int8_t *mat, int8_t a, int8_t b, int8_t transition, int8_t sc_ambi)
|
||||
{
|
||||
assert(m == 5);
|
||||
ksw_gen_simple_mat(m, mat, a, b, sc_ambi);
|
||||
if (transition == 0 || transition == b) return;
|
||||
transition = transition > 0? -transition : transition;
|
||||
mat[0 * m + 2] = transition; // A->G
|
||||
mat[1 * m + 3] = transition; // C->T
|
||||
mat[2 * m + 0] = transition; // G->A
|
||||
mat[3 * m + 1] = transition; // T->C
|
||||
}
|
||||
|
||||
static inline void mm_seq_rev(uint32_t len, uint8_t *seq)
|
||||
{
|
||||
uint32_t i;
|
||||
@@ -246,7 +260,7 @@ static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *ts
|
||||
if (p == 0) return;
|
||||
mm_fix_cigar(r, qseq, tseq, &qshift, &tshift);
|
||||
qseq += qshift, tseq += tshift; // qseq and tseq may be shifted due to the removal of leading I/D
|
||||
r->blen = r->mlen = 0;
|
||||
r->blen = r->mlen = 0, r->is_spliced = 0;
|
||||
for (k = 0; k < p->n_cigar; ++k) {
|
||||
uint32_t op = p->cigar[k]&0xf, len = p->cigar[k]>>4;
|
||||
if (op == MM_CIGAR_MATCH) {
|
||||
@@ -280,17 +294,16 @@ static void mm_update_extra(mm_reg1_t *r, const uint8_t *qseq, const uint8_t *ts
|
||||
if (s < 0) s = 0;
|
||||
toff += len;
|
||||
} else if (op == MM_CIGAR_N_SKIP) {
|
||||
toff += len;
|
||||
r->is_spliced = 1, toff += len;
|
||||
}
|
||||
}
|
||||
p->dp_max = (int32_t)(max + .499);
|
||||
p->dp_max = p->dp_max0 = (int32_t)(max + .499);
|
||||
assert(qoff == r->qe - r->qs && toff == r->re - r->rs);
|
||||
if (is_eqx) mm_update_cigar_eqx(r, qseq, tseq); // NB: it has to be called here as changes to qseq and tseq are not returned
|
||||
}
|
||||
|
||||
static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, uint32_t *cigar) // TODO: this calls the libc realloc()
|
||||
void mm_enlarge_cigar(mm_reg1_t *r, uint32_t n_cigar) // TODO: this calls the libc realloc()
|
||||
{
|
||||
mm_extra_t *p;
|
||||
if (n_cigar == 0) return;
|
||||
if (r->p == 0) {
|
||||
uint32_t capacity = n_cigar + sizeof(mm_extra_t)/4;
|
||||
@@ -302,6 +315,13 @@ static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, uint32_t *cigar) //
|
||||
kroundup32(r->p->capacity);
|
||||
r->p = (mm_extra_t*)realloc(r->p, r->p->capacity * 4);
|
||||
}
|
||||
}
|
||||
|
||||
static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, const uint32_t *cigar)
|
||||
{
|
||||
mm_extra_t *p;
|
||||
if (n_cigar == 0) return;
|
||||
mm_enlarge_cigar(r, n_cigar);
|
||||
p = r->p;
|
||||
if (p->n_cigar > 0 && (p->cigar[p->n_cigar-1]&0xf) == (cigar[0]&0xf)) { // same CIGAR op at the boundary
|
||||
p->cigar[p->n_cigar-1] += cigar[0]>>4<<4;
|
||||
@@ -313,25 +333,31 @@ static void mm_append_cigar(mm_reg1_t *r, uint32_t n_cigar, uint32_t *cigar) //
|
||||
}
|
||||
}
|
||||
|
||||
static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc, const int8_t *mat, int w, int end_bonus, int zdrop, int flag, ksw_extz_t *ez)
|
||||
static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc,
|
||||
const int8_t *mat, int w, int end_bonus, int zdrop, int ksw_flag, ksw_extz_t *ez)
|
||||
{
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
||||
int i;
|
||||
fprintf(stderr, "===> q=(%d,%d), e=(%d,%d), bw=%d, flag=%d, zdrop=%d <===\n", opt->q, opt->q2, opt->e, opt->e2, w, flag, opt->zdrop);
|
||||
fprintf(stderr, "===> q=(%d,%d), e=(%d,%d), bw=%d, ksw_flag=%d, zdrop=%d, end_bonus=%d <===\n", opt->q, opt->q2, opt->e, opt->e2, w, ksw_flag, opt->zdrop, end_bonus);
|
||||
for (i = 0; i < tlen; ++i) fputc("ACGTN"[tseq[i]], stderr);
|
||||
fputc('\n', stderr);
|
||||
for (i = 0; i < qlen; ++i) fputc("ACGTN"[qseq[i]], stderr);
|
||||
fputc('\n', stderr);
|
||||
}
|
||||
if (opt->max_sw_mat > 0 && (int64_t)tlen * qlen > opt->max_sw_mat) {
|
||||
if (opt->transition != 0 && opt->b != opt->transition)
|
||||
ksw_flag |= KSW_EZ_GENERIC_SC;
|
||||
if (opt->max_sw_mat > 0 && (int64_t)tlen * qlen > opt->max_sw_mat) { // too much memory; skip alignment
|
||||
ksw_reset_extz(ez);
|
||||
ez->zdropped = 1;
|
||||
} else if (opt->flag & MM_F_SPLICE)
|
||||
ksw_exts2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->noncan, zdrop, opt->junc_bonus, flag, junc, ez);
|
||||
else if (opt->q == opt->q2 && opt->e == opt->e2)
|
||||
ksw_extz2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, w, zdrop, end_bonus, flag, ez);
|
||||
else
|
||||
ksw_extd2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, flag, ez);
|
||||
} else if (opt->flag & MM_F_SPLICE) { // spliced alignment
|
||||
assert((ksw_flag & KSW_EZ_SPLICE_FOR) == 0 || (ksw_flag & KSW_EZ_SPLICE_REV) == 0);
|
||||
if (!(opt->flag & MM_F_SPLICE_OLD)) ksw_flag |= KSW_EZ_SPLICE_CMPLX;
|
||||
ksw_exts2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->noncan, zdrop, end_bonus, opt->junc_bonus, opt->junc_pen, ksw_flag, junc, ez);
|
||||
} else if (opt->q == opt->q2 && opt->e == opt->e2) { // affine gap
|
||||
ksw_extz2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, w, zdrop, end_bonus, ksw_flag, ez);
|
||||
} else { // dual affine gap
|
||||
ksw_extd2_sse(km, qlen, qseq, tlen, tseq, 5, mat, opt->q, opt->e, opt->q2, opt->e2, w, zdrop, end_bonus, ksw_flag, ez);
|
||||
}
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_ALN_SEQ) {
|
||||
int i;
|
||||
fprintf(stderr, "score=%d, cigar=", ez->score);
|
||||
@@ -341,6 +367,45 @@ static void mm_align_pair(void *km, const mm_mapopt_t *opt, int qlen, const uint
|
||||
}
|
||||
}
|
||||
|
||||
static int mm_align_sr_rna(void *km, const mm_mapopt_t *opt, int qlen, const uint8_t *qseq, int tlen, const uint8_t *tseq, const uint8_t *junc, uint8_t *tseq2, uint8_t *junc2,
|
||||
const int8_t *mat, int w, int end_bonus, int zdrop, int ksw_flag, ksw_extz_t *ez)
|
||||
{
|
||||
int32_t ilen = opt->q2 * 2, tlen2 = qlen * 2 + ilen;
|
||||
int32_t i, ll = 0, lr = 0, nn = 0, n_ins = 0;
|
||||
if (!(opt->flag & MM_F_SPLICE)) return 0; // only for spliced alignment
|
||||
if (qlen > MM_MAX_QLEN_FLANK || qlen * 2 + ilen > tlen) return 0; // the query sequence can't be too long and the target sequence must be long enough
|
||||
for (i = 0; i < qlen; ++i) // exact match length from the left
|
||||
if (qseq[i] == tseq[i] && qseq[i] < 4)
|
||||
++ll;
|
||||
for (i = 0; i < qlen; ++i) // exact match length from the right
|
||||
if (qseq[qlen - 1 - i] == tseq[tlen - 1 - i] && qseq[qlen - 1 - i] < 4)
|
||||
++lr;
|
||||
if (qlen - (ll + lr) > 9) return 0; // qlen may be smaller than ll+lr
|
||||
memcpy(tseq2, tseq, qlen);
|
||||
memset(&tseq2[qlen], 4, ilen);
|
||||
memcpy(&tseq2[qlen + ilen], &tseq[tlen - qlen], qlen);
|
||||
if (junc) {
|
||||
memcpy(junc2, junc, qlen);
|
||||
memset(&junc2[qlen], 0, ilen);
|
||||
memcpy(&junc2[qlen + ilen], &junc[tlen - qlen], qlen);
|
||||
}
|
||||
if (!(opt->flag & MM_F_SPLICE_OLD)) ksw_flag |= KSW_EZ_SPLICE_CMPLX;
|
||||
ksw_exts2_sse(km, qlen, qseq, tlen2, tseq2, 5, mat, opt->q, opt->e, opt->q2, opt->noncan, zdrop, end_bonus, opt->junc_bonus, opt->junc_pen, ksw_flag, junc2, ez);
|
||||
if (ez->zdropped) return 0;
|
||||
if ((ez->cigar[0]&0xf) != KSW_CIGAR_MATCH || (ez->cigar[ez->n_cigar-1]&0xf) != KSW_CIGAR_MATCH) return 0;
|
||||
for (i = 0; i < ez->n_cigar; ++i) { // count the number of introns in the alignment
|
||||
if ((ez->cigar[i]&0xf) == KSW_CIGAR_N_SKIP)
|
||||
++nn;
|
||||
else if ((ez->cigar[i]&0xf) == KSW_CIGAR_INS)
|
||||
++n_ins;
|
||||
}
|
||||
if (nn != 1 || n_ins > 0) return 0; // the heuristic only works when there is exactly one intron
|
||||
for (i = 0; i < ez->n_cigar; ++i)
|
||||
if ((ez->cigar[i]&0xf) == KSW_CIGAR_N_SKIP)
|
||||
ez->cigar[i] += (tlen - tlen2) << 4;
|
||||
return 1;
|
||||
}
|
||||
|
||||
static inline int mm_get_hplen_back(const mm_idx_t *mi, uint32_t rid, uint32_t x)
|
||||
{
|
||||
int64_t i, off0 = mi->seq[rid].offset, off = off0 + x;
|
||||
@@ -570,12 +635,19 @@ static void mm_fix_bad_ends_splice(void *km, const mm_mapopt_t *opt, const mm_id
|
||||
}
|
||||
}
|
||||
|
||||
static inline void mm_get_junc(const mm_idx_t *mi, int32_t ctg, int32_t st, int32_t en, int32_t rev, uint8_t *junc)
|
||||
{
|
||||
if (mi->spsc) mm_idx_spsc_get(mi, ctg, st, en, rev, junc);
|
||||
else if (mi->I) mm_idx_bed_junc(mi, ctg, st, en, junc);
|
||||
else memset(junc, 0, en - st);
|
||||
}
|
||||
|
||||
static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, uint8_t *qseq0[2], mm_reg1_t *r, mm_reg1_t *r2, int n_a, mm128_t *a, ksw_extz_t *ez, int splice_flag)
|
||||
{
|
||||
int is_sr = !!(opt->flag & MM_F_SR), is_splice = !!(opt->flag & MM_F_SPLICE);
|
||||
int is_sr = !!(opt->flag & MM_F_SR), is_splice = !!(opt->flag & MM_F_SPLICE), is_sr_rna = (!!(opt->flag & MM_F_SR_RNA) && is_splice);
|
||||
int32_t rid = a[r->as].x<<1>>33, rev = a[r->as].x>>63, as1, cnt1;
|
||||
uint8_t *tseq, *qseq, *junc;
|
||||
int32_t i, l, bw, bw_long, dropped = 0, extra_flag = 0, rs0, re0, qs0, qe0;
|
||||
uint8_t *tseq, *qseq, *junc, *tseq2 = 0, *junc2 = 0;
|
||||
int32_t i, l, bw, bw_long, dropped = 0, ksw_flag = 0, rs0, re0, qs0, qe0;
|
||||
int32_t rs, re, qs, qe;
|
||||
int32_t rs1, qs1, re1, qe1;
|
||||
int8_t mat[25];
|
||||
@@ -584,7 +656,7 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
|
||||
r2->cnt = 0;
|
||||
if (r->cnt == 0) return;
|
||||
ksw_gen_simple_mat(5, mat, opt->a, opt->b, opt->sc_ambi);
|
||||
ksw_gen_ts_mat(5, mat, opt->a, opt->b, opt->transition, opt->sc_ambi);
|
||||
bw = (int)(opt->bw * 1.5 + 1.);
|
||||
bw_long = (int)(opt->bw_long * 1.5 + 1.);
|
||||
if (bw_long < bw) bw_long = bw;
|
||||
@@ -610,9 +682,10 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
assert(cnt1 > 0);
|
||||
|
||||
if (is_splice) {
|
||||
if (splice_flag & MM_F_SPLICE_FOR) extra_flag |= rev? KSW_EZ_SPLICE_REV : KSW_EZ_SPLICE_FOR;
|
||||
if (splice_flag & MM_F_SPLICE_REV) extra_flag |= rev? KSW_EZ_SPLICE_FOR : KSW_EZ_SPLICE_REV;
|
||||
if (opt->flag & MM_F_SPLICE_FLANK) extra_flag |= KSW_EZ_SPLICE_FLANK;
|
||||
if (splice_flag & MM_F_SPLICE_FOR) ksw_flag |= rev? KSW_EZ_SPLICE_REV : KSW_EZ_SPLICE_FOR;
|
||||
if (splice_flag & MM_F_SPLICE_REV) ksw_flag |= rev? KSW_EZ_SPLICE_FOR : KSW_EZ_SPLICE_REV;
|
||||
if (opt->flag & MM_F_SPLICE_FLANK) ksw_flag |= KSW_EZ_SPLICE_FLANK;
|
||||
if (mi->spsc) ksw_flag |= KSW_EZ_SPLICE_SCORE;
|
||||
}
|
||||
|
||||
/* Look for the start and end of regions to perform DP. This sounds easy
|
||||
@@ -697,6 +770,12 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
tseq = (uint8_t*)kmalloc(km, re0 - rs0);
|
||||
junc = (uint8_t*)kmalloc(km, re0 - rs0);
|
||||
|
||||
if (is_sr_rna) {
|
||||
int32_t max_tlen2 = MM_MAX_QLEN_FLANK * 2 + opt->q2 * 2;
|
||||
tseq2 = Kmalloc(km, uint8_t, max_tlen2 * 2);
|
||||
junc2 = tseq2 + max_tlen2;
|
||||
}
|
||||
|
||||
if (qs > 0 && rs > 0) { // left extension; probably the condition can be changed to "qs > qs0 && rs > rs0"
|
||||
if (opt->flag & MM_F_QSTRAND) {
|
||||
qseq = &qseq0[0][qs0];
|
||||
@@ -705,11 +784,11 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
qseq = &qseq0[rev][qs0];
|
||||
mm_idx_getseq(mi, rid, rs0, rs, tseq);
|
||||
}
|
||||
mm_idx_bed_junc(mi, rid, rs0, rs, junc);
|
||||
mm_get_junc(mi, rid, rs0, rs, !!(ksw_flag&KSW_EZ_SPLICE_REV), junc);
|
||||
mm_seq_rev(qs - qs0, qseq);
|
||||
mm_seq_rev(rs - rs0, tseq);
|
||||
mm_seq_rev(rs - rs0, junc);
|
||||
mm_align_pair(km, opt, qs - qs0, qseq, rs - rs0, tseq, junc, mat, bw, opt->end_bonus, r->split_inv? opt->zdrop_inv : opt->zdrop, extra_flag|KSW_EZ_EXTZ_ONLY|KSW_EZ_RIGHT|KSW_EZ_REV_CIGAR, ez);
|
||||
mm_align_pair(km, opt, qs - qs0, qseq, rs - rs0, tseq, junc, mat, bw, opt->end_bonus, r->split_inv? opt->zdrop_inv : opt->zdrop, ksw_flag|KSW_EZ_EXTZ_ONLY|KSW_EZ_RIGHT|KSW_EZ_REV_CIGAR, ez);
|
||||
if (ez->n_cigar > 0) {
|
||||
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
||||
r->p->dp_score += ez->max;
|
||||
@@ -721,14 +800,14 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
re1 = rs, qe1 = qs;
|
||||
assert(qs1 >= 0 && rs1 >= 0);
|
||||
|
||||
for (i = is_sr? cnt1 - 1 : 1; i < cnt1; ++i) { // gap filling
|
||||
for (i = is_sr? cnt1 - 1 : 1; i < cnt1; ++i) { // gap filling; for short genomic reads, fill from the first seed to the last
|
||||
if ((a[as1+i].y & (MM_SEED_IGNORE|MM_SEED_TANDEM)) && i != cnt1 - 1) continue;
|
||||
if (is_sr && !(mi->flag & MM_I_HPC)) {
|
||||
re = (int32_t)a[as1 + i].x + 1;
|
||||
qe = (int32_t)a[as1 + i].y + 1;
|
||||
} else mm_adjust_minier(mi, qseq0, &a[as1 + i], &re, &qe);
|
||||
re1 = re, qe1 = qe;
|
||||
if (i == cnt1 - 1 || (a[as1+i].y&MM_SEED_LONG_JOIN) || (qe - qs >= opt->min_ksw_len && re - rs >= opt->min_ksw_len)) {
|
||||
if (i == cnt1 - 1 || (a[as1+i].y&MM_SEED_LONG_JOIN) || (qe - qs >= opt->min_ksw_len && re - rs >= opt->min_ksw_len)) { // gap filling
|
||||
int j, bw1 = bw_long, zdrop_code;
|
||||
if (a[as1+i].y & MM_SEED_LONG_JOIN)
|
||||
bw1 = qe - qs > re - rs? qe - qs : re - rs;
|
||||
@@ -740,21 +819,29 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
qseq = &qseq0[rev][qs];
|
||||
mm_idx_getseq(mi, rid, rs, re, tseq);
|
||||
}
|
||||
mm_idx_bed_junc(mi, rid, rs, re, junc);
|
||||
if (is_sr) { // perform ungapped alignment
|
||||
mm_get_junc(mi, rid, rs, re, !!(ksw_flag&KSW_EZ_SPLICE_REV), junc);
|
||||
if (is_sr || (is_sr_rna && qe - qs == re - rs)) { // perform ungapped alignment
|
||||
int32_t max_gapped_score = (qe - qs - 2) * opt->a - 2 * (opt->q + opt->e);
|
||||
assert(qe - qs == re - rs);
|
||||
ksw_reset_extz(ez);
|
||||
for (j = 0, ez->score = 0; j < qe - qs; ++j) {
|
||||
if (qseq[j] >= 4 || tseq[j] >= 4) ez->score += opt->e2;
|
||||
if (qseq[j] >= 4 || tseq[j] >= 4) ez->score += opt->sc_ambi > 0? -opt->sc_ambi : opt->sc_ambi;
|
||||
else ez->score += qseq[j] == tseq[j]? opt->a : -opt->b;
|
||||
}
|
||||
ez->cigar = ksw_push_cigar(km, &ez->n_cigar, &ez->m_cigar, ez->cigar, MM_CIGAR_MATCH, qe - qs);
|
||||
if (ez->score > max_gapped_score)
|
||||
ez->cigar = ksw_push_cigar(km, &ez->n_cigar, &ez->m_cigar, ez->cigar, MM_CIGAR_MATCH, qe - qs);
|
||||
else
|
||||
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, opt->zdrop, ksw_flag|KSW_EZ_APPROX_MAX, ez);
|
||||
} else { // perform normal gapped alignment
|
||||
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, opt->zdrop, extra_flag|KSW_EZ_APPROX_MAX, ez); // first pass: with approximate Z-drop
|
||||
int32_t skip_full = 0;
|
||||
if (is_sr_rna)
|
||||
skip_full = mm_align_sr_rna(km, opt, qe - qs, qseq, re - rs, tseq, junc, tseq2, junc2, mat, bw1, -1, opt->zdrop, ksw_flag|KSW_EZ_APPROX_MAX, ez);
|
||||
if (!skip_full)
|
||||
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, opt->zdrop, ksw_flag|KSW_EZ_APPROX_MAX, ez); // first pass: with approximate Z-drop
|
||||
}
|
||||
// test Z-drop and inversion Z-drop
|
||||
if ((zdrop_code = mm_test_zdrop(km, opt, qseq, tseq, ez->n_cigar, ez->cigar, mat)) != 0)
|
||||
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, zdrop_code == 2? opt->zdrop_inv : opt->zdrop, extra_flag, ez); // second pass: lift approximate
|
||||
mm_align_pair(km, opt, qe - qs, qseq, re - rs, tseq, junc, mat, bw1, -1, zdrop_code == 2? opt->zdrop_inv : opt->zdrop, ksw_flag, ez); // second pass: lift approximate
|
||||
// update CIGAR
|
||||
if (ez->n_cigar > 0)
|
||||
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
||||
@@ -792,8 +879,8 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
qseq = &qseq0[rev][qe];
|
||||
mm_idx_getseq(mi, rid, re, re0, tseq);
|
||||
}
|
||||
mm_idx_bed_junc(mi, rid, re, re0, junc);
|
||||
mm_align_pair(km, opt, qe0 - qe, qseq, re0 - re, tseq, junc, mat, bw, opt->end_bonus, opt->zdrop, extra_flag|KSW_EZ_EXTZ_ONLY, ez);
|
||||
mm_get_junc(mi, rid, re, re0, !!(ksw_flag&KSW_EZ_SPLICE_REV), junc);
|
||||
mm_align_pair(km, opt, qe0 - qe, qseq, re0 - re, tseq, junc, mat, bw, opt->end_bonus, opt->zdrop, ksw_flag|KSW_EZ_EXTZ_ONLY, ez);
|
||||
if (ez->n_cigar > 0) {
|
||||
mm_append_cigar(r, ez->n_cigar, ez->cigar);
|
||||
r->p->dp_score += ez->max;
|
||||
@@ -816,11 +903,12 @@ static void mm_align1(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int
|
||||
mm_idx_getseq(mi, rid, rs1, re1, tseq);
|
||||
qseq = &qseq0[r->rev][qs1];
|
||||
}
|
||||
mm_update_extra(r, qseq, tseq, mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & MM_F_SR));
|
||||
mm_update_extra(r, qseq, tseq, mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(is_sr || is_sr_rna));
|
||||
if (rev && r->p->trans_strand)
|
||||
r->p->trans_strand ^= 3; // flip to the read strand
|
||||
}
|
||||
|
||||
if (tseq2) kfree(km, tseq2);
|
||||
kfree(km, tseq);
|
||||
kfree(km, junc);
|
||||
}
|
||||
@@ -842,7 +930,7 @@ static int mm_align1_inv(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, i
|
||||
if (ql < opt->min_chain_score || ql > opt->max_gap) return 0;
|
||||
if (tl < opt->min_chain_score || tl > opt->max_gap) return 0;
|
||||
|
||||
ksw_gen_simple_mat(5, mat, opt->a, opt->b, opt->sc_ambi);
|
||||
ksw_gen_ts_mat(5, mat, opt->a, opt->b, opt->transition, opt->sc_ambi);
|
||||
tseq = (uint8_t*)kmalloc(km, tl);
|
||||
mm_idx_getseq(mi, r1->rid, r1->re, r2->rs, tseq);
|
||||
qseq = r1->rev? &qseq0[0][r2->qe] : &qseq0[1][qlen - r2->qs];
|
||||
@@ -875,7 +963,7 @@ static int mm_align1_inv(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, i
|
||||
}
|
||||
r_inv->rs = r1->re + t_off;
|
||||
r_inv->re = r_inv->rs + ez->max_t + 1;
|
||||
mm_update_extra(r_inv, &qseq[q_off], &tseq[t_off], mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & MM_F_SR));
|
||||
mm_update_extra(r_inv, &qseq[q_off], &tseq[t_off], mat, opt->q, opt->e, opt->flag & MM_F_EQX, !(opt->flag & (MM_F_SR|MM_F_SR_RNA)));
|
||||
ret = 1;
|
||||
end_align1_inv:
|
||||
kfree(km, tseq);
|
||||
@@ -917,14 +1005,14 @@ double mm_event_identity(const mm_reg1_t *r)
|
||||
static int32_t mm_recal_max_dp(const mm_reg1_t *r, double b2, int32_t match_sc)
|
||||
{
|
||||
uint32_t i;
|
||||
int32_t n_gap = 0, n_gapo = 0, n_mis;
|
||||
int32_t n_gap = 0, n_mis;
|
||||
double gap_cost = 0.0;
|
||||
if (r->p == 0) return -1;
|
||||
for (i = 0; i < r->p->n_cigar; ++i) {
|
||||
int32_t op = r->p->cigar[i] & 0xf, len = r->p->cigar[i] >> 4;
|
||||
if (op == MM_CIGAR_INS || op == MM_CIGAR_DEL) {
|
||||
gap_cost += b2 + (double)mg_log2(1.0 + len);
|
||||
++n_gapo, n_gap += len;
|
||||
n_gap += len;
|
||||
}
|
||||
}
|
||||
n_mis = r->blen + r->p->n_ambi - r->mlen - n_gap;
|
||||
@@ -976,24 +1064,36 @@ mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
||||
n_a = mm_squeeze_a(km, n_regs, regs, a);
|
||||
memset(&ez, 0, sizeof(ksw_extz_t));
|
||||
for (i = 0; i < n_regs; ++i) {
|
||||
mm_reg1_t r2;
|
||||
mm_reg1_t r2; // only used for inversion
|
||||
if ((opt->flag&MM_F_SPLICE) && (opt->flag&MM_F_SPLICE_FOR) && (opt->flag&MM_F_SPLICE_REV)) { // then do two rounds of alignments for both strands
|
||||
mm_reg1_t s[2], s2[2];
|
||||
int which, trans_strand;
|
||||
mm_reg1_t s[2], s2[2], *r;
|
||||
s[0] = s[1] = regs[i];
|
||||
mm_align1(km, opt, mi, qlen, qseq0, &s[0], &s2[0], n_a, a, &ez, MM_F_SPLICE_FOR);
|
||||
mm_align1(km, opt, mi, qlen, qseq0, &s[1], &s2[1], n_a, a, &ez, MM_F_SPLICE_REV);
|
||||
if (s[0].p->dp_score > s[1].p->dp_score) which = 0, trans_strand = 1;
|
||||
else if (s[0].p->dp_score < s[1].p->dp_score) which = 1, trans_strand = 2;
|
||||
else trans_strand = 3, which = (qlen + s[0].p->dp_score) & 1; // randomly choose a strand, effectively
|
||||
if (which == 0) {
|
||||
mm_align1(km, opt, mi, qlen, qseq0, &s[0], &s2[0], n_a, a, &ez, MM_F_SPLICE_FOR); // assume the transcript is on the + strand of the genome
|
||||
if ((opt->flag&MM_F_SR_RNA) && regs[i].qe - regs[i].qs == regs[i].re - regs[i].rs && s[0].qe - s[0].qs == s[0].re - s[0].rs && s[0].qs == 0 && s[0].qe == qlen) {
|
||||
regs[i] = s[0], r2 = s2[0];
|
||||
free(s[1].p);
|
||||
regs[i].p->trans_strand = 0;
|
||||
} else {
|
||||
regs[i] = s[1], r2 = s2[1];
|
||||
free(s[0].p);
|
||||
int which, trans_strand;
|
||||
mm_align1(km, opt, mi, qlen, qseq0, &s[1], &s2[1], n_a, a, &ez, MM_F_SPLICE_REV); // assume the transcript on the - strand
|
||||
if (s[0].p->dp_score > s[1].p->dp_score) which = 0, trans_strand = 1;
|
||||
else if (s[0].p->dp_score < s[1].p->dp_score) which = 1, trans_strand = 2;
|
||||
else trans_strand = 3, which = (qlen + s[0].p->dp_score) & 1; // randomly choose a strand, effectively
|
||||
if (which == 0) {
|
||||
regs[i] = s[0], r2 = s2[0];
|
||||
free(s[1].p);
|
||||
} else {
|
||||
regs[i] = s[1], r2 = s2[1];
|
||||
free(s[0].p);
|
||||
}
|
||||
r = ®s[i];
|
||||
r->p->trans_strand = trans_strand;
|
||||
if (r->is_spliced) {
|
||||
if (trans_strand == 1 || trans_strand == 2) // this is an *approximate* way to tell if there are splice signals.
|
||||
r->p->dp_max += (opt->a + opt->b) + ((opt->a + opt->b) >> 1);
|
||||
else if (trans_strand == 3)
|
||||
r->p->dp_max -= opt->a + opt->b;
|
||||
}
|
||||
}
|
||||
regs[i].p->trans_strand = trans_strand;
|
||||
} else { // one round of alignment
|
||||
mm_align1(km, opt, mi, qlen, qseq0, ®s[i], &r2, n_a, a, &ez, opt->flag);
|
||||
if (opt->flag&MM_F_SPLICE)
|
||||
@@ -1011,7 +1111,7 @@ mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *m
|
||||
kfree(km, qseq0[0]);
|
||||
kfree(km, ez.cigar);
|
||||
mm_filter_regs(opt, qlen, n_regs_, regs);
|
||||
if (!(opt->flag&MM_F_SR) && !opt->split_prefix && qlen >= opt->rank_min_len) {
|
||||
if (!(opt->flag&(MM_F_SR|MM_F_SR_RNA|MM_F_ALL_CHAINS)) && !opt->split_prefix && qlen >= opt->rank_min_len) {
|
||||
mm_update_dp_max(qlen, *n_regs_, regs, opt->rank_frac, opt->a, opt->b);
|
||||
mm_filter_regs(opt, qlen, n_regs_, regs);
|
||||
}
|
||||
|
||||
+2
-2
@@ -31,8 +31,8 @@ To acquire the data used in this cookbook and to install minimap2 and paftools,
|
||||
please follow the command lines below:
|
||||
```sh
|
||||
# install minimap2 executables
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.23/minimap2-2.23_x64-linux.tar.bz2 | tar jxf -
|
||||
cp minimap2-2.23_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.29/minimap2-2.29_x64-linux.tar.bz2 | tar jxf -
|
||||
cp minimap2-2.29_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
||||
export PATH="$PATH:"`pwd` # put the current directory on PATH
|
||||
# download example datasets
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.10/cookbook-data.tgz | tar zxf -
|
||||
|
||||
@@ -119,6 +119,7 @@ int mm_write_sam_hdr(const mm_idx_t *idx, const char *rg, const char *ver, int a
|
||||
{
|
||||
kstring_t str = {0,0,0};
|
||||
int ret = 0;
|
||||
mm_sprintf_lite(&str, "@HD\tVN:1.6\tSO:unsorted\tGO:query\n");
|
||||
if (idx) {
|
||||
uint32_t i;
|
||||
for (i = 0; i < idx->n_seq; ++i)
|
||||
@@ -138,10 +139,48 @@ int mm_write_sam_hdr(const mm_idx_t *idx, const char *rg, const char *ver, int a
|
||||
return ret;
|
||||
}
|
||||
|
||||
static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq, const mm_reg1_t *r, char *tmp, int no_iden, int write_tag)
|
||||
static void write_indel_ds(kstring_t *str, int64_t len, const uint8_t *seq, int64_t ll, int64_t lr) // write an indel to ds; adapted from minigraph
|
||||
{
|
||||
int i, q_off, t_off;
|
||||
if (write_tag) mm_sprintf_lite(s, "\tcs:Z:");
|
||||
int64_t i;
|
||||
if (ll + lr >= len) {
|
||||
mm_sprintf_lite(str, "[");
|
||||
for (i = 0; i < len; ++i)
|
||||
mm_sprintf_lite(str, "%c", "acgtn"[seq[i]]);
|
||||
mm_sprintf_lite(str, "]");
|
||||
} else {
|
||||
int64_t k = 0;
|
||||
if (ll > 0) {
|
||||
mm_sprintf_lite(str, "[");
|
||||
for (i = 0; i < ll; ++i)
|
||||
mm_sprintf_lite(str, "%c", "acgtn"[seq[k+i]]);
|
||||
mm_sprintf_lite(str, "]");
|
||||
k += ll;
|
||||
}
|
||||
for (i = 0; i < len - lr - ll; ++i)
|
||||
mm_sprintf_lite(str, "%c", "acgtn"[seq[k+i]]);
|
||||
k += len - lr - ll;
|
||||
if (lr > 0) {
|
||||
mm_sprintf_lite(str, "[");
|
||||
for (i = 0; i < lr; ++i)
|
||||
mm_sprintf_lite(str, "%c", "acgtn"[seq[k+i]]);
|
||||
mm_sprintf_lite(str, "]");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void write_cs_ds_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq, const mm_reg1_t *r, char *tmp, int no_iden, int is_ds, int write_tag)
|
||||
{
|
||||
int i, q_off, t_off, q_len = 0, t_len = 0;
|
||||
if (write_tag) mm_sprintf_lite(s, "\t%cs:Z:", is_ds? 'd' : 'c');
|
||||
for (i = 0; i < (int)r->p->n_cigar; ++i) {
|
||||
int op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||
if (op == MM_CIGAR_MATCH || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH)
|
||||
q_len += len, t_len += len;
|
||||
else if (op == MM_CIGAR_INS)
|
||||
q_len += len;
|
||||
else if (op == MM_CIGAR_DEL || op == MM_CIGAR_N_SKIP)
|
||||
t_len += len;
|
||||
}
|
||||
for (i = q_off = t_off = 0; i < (int)r->p->n_cigar; ++i) {
|
||||
int j, op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||
assert((op >= MM_CIGAR_MATCH && op <= MM_CIGAR_N_SKIP) || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH);
|
||||
@@ -167,14 +206,42 @@ static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
||||
}
|
||||
q_off += len, t_off += len;
|
||||
} else if (op == MM_CIGAR_INS) {
|
||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||
tmp[j] = "acgtn"[qseq[q_off + j]];
|
||||
mm_sprintf_lite(s, "+%s", tmp);
|
||||
if (is_ds) {
|
||||
int z, ll, lr, y = q_off;
|
||||
for (z = 1; z <= len; ++z)
|
||||
if (y - z < 0 || qseq[y + len - z] != qseq[y - z])
|
||||
break;
|
||||
lr = z - 1;
|
||||
for (z = 0; z < len; ++z)
|
||||
if (y + len + z >= q_len || qseq[y + len + z] != qseq[y + z])
|
||||
break;
|
||||
ll = z;
|
||||
mm_sprintf_lite(s, "+");
|
||||
write_indel_ds(s, len, &qseq[y], ll, lr);
|
||||
} else {
|
||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||
tmp[j] = "acgtn"[qseq[q_off + j]];
|
||||
mm_sprintf_lite(s, "+%s", tmp);
|
||||
}
|
||||
q_off += len;
|
||||
} else if (op == MM_CIGAR_DEL) {
|
||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||
tmp[j] = "acgtn"[tseq[t_off + j]];
|
||||
mm_sprintf_lite(s, "-%s", tmp);
|
||||
if (is_ds) {
|
||||
int z, ll, lr, x = t_off;
|
||||
for (z = 1; z <= len; ++z)
|
||||
if (x - z < 0 || tseq[x + len - z] != tseq[x - z])
|
||||
break;
|
||||
lr = z - 1;
|
||||
for (z = 0; z < len; ++z)
|
||||
if (x + len + z >= t_len || tseq[x + z] != tseq[x + len + z])
|
||||
break;
|
||||
ll = z;
|
||||
mm_sprintf_lite(s, "-");
|
||||
write_indel_ds(s, len, &tseq[x], ll, lr);
|
||||
} else {
|
||||
for (j = 0, tmp[len] = 0; j < len; ++j)
|
||||
tmp[j] = "acgtn"[tseq[t_off + j]];
|
||||
mm_sprintf_lite(s, "-%s", tmp);
|
||||
}
|
||||
t_off += len;
|
||||
} else { // intron
|
||||
assert(len >= 2);
|
||||
@@ -186,6 +253,52 @@ static void write_cs_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
||||
assert(t_off == r->re - r->rs && q_off == r->qe - r->qs);
|
||||
}
|
||||
|
||||
static inline void revcomp_splice(uint8_t s[2])
|
||||
{
|
||||
uint8_t c = s[1] < 4? 3 - s[1] : 4;
|
||||
s[1] = s[0] < 4? 3 - s[0] : 4;
|
||||
s[0] = c;
|
||||
}
|
||||
|
||||
void mm_write_junc(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r)
|
||||
{
|
||||
int32_t i, t_off, swritten = 0;
|
||||
s->l = 0;
|
||||
if (!r->is_spliced || r->p == 0) return; // no junctions
|
||||
if (r->p->trans_strand != 1 && r->p->trans_strand != 2) return; // no preferred strand
|
||||
for (i = 0, t_off = r->rs; i < (int)r->p->n_cigar; ++i) {
|
||||
int op = r->p->cigar[i]&0xf, len = r->p->cigar[i]>>4;
|
||||
if (op == MM_CIGAR_MATCH || op == MM_CIGAR_EQ_MATCH || op == MM_CIGAR_X_MISMATCH || op == MM_CIGAR_DEL) {
|
||||
t_off += len;
|
||||
} else if (op == MM_CIGAR_N_SKIP) { // intron
|
||||
uint8_t donor[2], acceptor[2];
|
||||
int32_t score1 = 0, score2 = 0, rev;
|
||||
assert(len >= 2);
|
||||
rev = (r->p->trans_strand == 2) ^ r->rev;
|
||||
if (!rev) {
|
||||
mm_idx_getseq(mi, r->rid, t_off, t_off + 2, donor);
|
||||
mm_idx_getseq(mi, r->rid, t_off + len - 2, t_off + len, acceptor);
|
||||
} else {
|
||||
mm_idx_getseq(mi, r->rid, t_off, t_off + 2, acceptor);
|
||||
mm_idx_getseq(mi, r->rid, t_off + len - 2, t_off + len, donor);
|
||||
revcomp_splice(donor);
|
||||
revcomp_splice(acceptor);
|
||||
}
|
||||
//fprintf(stderr, "%c%c-%c%c\n", "ACGTN"[donor[0]], "ACGTN"[donor[1]], "ACGTN"[acceptor[0]], "ACGTN"[acceptor[1]]);
|
||||
if (donor[0] == 2 && donor[1] == 3) score1 = 3;
|
||||
else if (donor[0] == 2 && donor[1] == 1) score1 = 2;
|
||||
else if (donor[0] == 0 && donor[1] == 3) score1 = 1;
|
||||
if (acceptor[0] == 0 && acceptor[1] == 2) score2 = 3;
|
||||
else if (acceptor[0] == 0 && acceptor[1] == 1) score2 = 1;
|
||||
if (swritten) mm_sprintf_lite(s, "\n");
|
||||
else swritten = 1;
|
||||
mm_sprintf_lite(s, "%s\t%d\t%d\t%s\t%d\t%c", mi->seq[r->rid].name, t_off, t_off + len, t->name, score1 + score2, "+-"[rev]);
|
||||
t_off += len;
|
||||
}
|
||||
}
|
||||
assert(t_off == r->re);
|
||||
}
|
||||
|
||||
static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq, const mm_reg1_t *r, char *tmp, int write_tag)
|
||||
{
|
||||
int i, q_off, t_off, l_MD = 0;
|
||||
@@ -217,7 +330,7 @@ static void write_MD_core(kstring_t *s, const uint8_t *tseq, const uint8_t *qseq
|
||||
assert(t_off == r->re - r->rs && q_off == r->qe - r->qs);
|
||||
}
|
||||
|
||||
static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int no_iden, int is_MD, int write_tag, int is_qstrand)
|
||||
static void write_cs_ds_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int no_iden, int is_MD, int is_ds, int write_tag, int is_qstrand)
|
||||
{
|
||||
extern unsigned char seq_nt4_table[256];
|
||||
int i;
|
||||
@@ -244,7 +357,7 @@ static void write_cs_or_MD(void *km, kstring_t *s, const mm_idx_t *mi, const mm_
|
||||
}
|
||||
}
|
||||
if (is_MD) write_MD_core(s, tseq, qseq, r, tmp, write_tag);
|
||||
else write_cs_core(s, tseq, qseq, r, tmp, no_iden, write_tag);
|
||||
else write_cs_ds_core(s, tseq, qseq, r, tmp, no_iden, is_ds, write_tag);
|
||||
kfree(km, qseq); kfree(km, tseq); kfree(km, tmp);
|
||||
}
|
||||
|
||||
@@ -255,7 +368,7 @@ int mm_gen_cs_or_MD(void *km, char **buf, int *max_len, const mm_idx_t *mi, cons
|
||||
str.s = *buf, str.l = 0, str.m = *max_len;
|
||||
t.l_seq = strlen(seq);
|
||||
t.seq = (char*)seq;
|
||||
write_cs_or_MD(km, &str, mi, &t, r, no_iden, is_MD, 0, is_qstrand);
|
||||
write_cs_ds_or_MD(km, &str, mi, &t, r, no_iden, is_MD, 0, 0, is_qstrand);
|
||||
*max_len = str.m;
|
||||
*buf = str.s;
|
||||
return str.l;
|
||||
@@ -277,7 +390,7 @@ static inline void write_tags(kstring_t *s, const mm_reg1_t *r)
|
||||
if (r->id == r->parent) type = r->inv? 'I' : 'P';
|
||||
else type = r->inv? 'i' : 'S';
|
||||
if (r->p) {
|
||||
mm_sprintf_lite(s, "\tNM:i:%d\tms:i:%d\tAS:i:%d\tnn:i:%d", r->blen - r->mlen + r->p->n_ambi, r->p->dp_max, r->p->dp_score, r->p->n_ambi);
|
||||
mm_sprintf_lite(s, "\tNM:i:%d\tms:i:%d\tAS:i:%d\tnn:i:%d", r->blen - r->mlen + r->p->n_ambi, r->p->dp_max0, r->p->dp_score, r->p->n_ambi);
|
||||
if (r->p->trans_strand == 1 || r->p->trans_strand == 2)
|
||||
mm_sprintf_lite(s, "\tts:A:%c", "?+-?"[r->p->trans_strand]);
|
||||
}
|
||||
@@ -299,15 +412,18 @@ static inline void write_tags(kstring_t *s, const mm_reg1_t *r)
|
||||
if (r->split) mm_sprintf_lite(s, "\tzd:i:%d", r->split);
|
||||
}
|
||||
|
||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len)
|
||||
void mm_write_paf4(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len, int n_seg, int seg_idx)
|
||||
{
|
||||
s->l = 0;
|
||||
mm_sprintf_lite(s, "%s", t->name);
|
||||
if ((opt_flag & MM_F_FRAG_MODE) && n_seg >= 2 && seg_idx >= 0)
|
||||
mm_sprintf_lite(s, "/%d", seg_idx + 1);
|
||||
if (r == 0) {
|
||||
mm_sprintf_lite(s, "%s\t%d\t0\t0\t*\t*\t0\t0\t0\t0\t0\t0", t->name, t->l_seq);
|
||||
mm_sprintf_lite(s, "\t%d\t0\t0\t*\t*\t0\t0\t0\t0\t0\t0", t->l_seq);
|
||||
if (rep_len >= 0) mm_sprintf_lite(s, "\trl:i:%d", rep_len);
|
||||
return;
|
||||
}
|
||||
mm_sprintf_lite(s, "%s\t%d\t%d\t%d\t%c\t", t->name, t->l_seq, r->qs, r->qe, "+-"[r->rev]);
|
||||
mm_sprintf_lite(s, "\t%d\t%d\t%d\t%c\t", t->l_seq, r->qs, r->qe, "+-"[r->rev]);
|
||||
if (mi->seq[r->rid].name) mm_sprintf_lite(s, "%s", mi->seq[r->rid].name);
|
||||
else mm_sprintf_lite(s, "%d", r->rid);
|
||||
mm_sprintf_lite(s, "\t%d", mi->seq[r->rid].len);
|
||||
@@ -325,12 +441,17 @@ void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const
|
||||
for (k = 0; k < r->p->n_cigar; ++k)
|
||||
mm_sprintf_lite(s, "%d%c", r->p->cigar[k]>>4, MM_CIGAR_STR[r->p->cigar[k]&0xf]);
|
||||
}
|
||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1, !!(opt_flag&MM_F_QSTRAND));
|
||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_DS|MM_F_OUT_MD)))
|
||||
write_cs_ds_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), !!(opt_flag&MM_F_OUT_MD), !!(opt_flag&MM_F_OUT_DS), 1, !!(opt_flag&MM_F_QSTRAND));
|
||||
if ((opt_flag & MM_F_COPY_COMMENT) && t->comment)
|
||||
mm_sprintf_lite(s, "\t%s", t->comment);
|
||||
}
|
||||
|
||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len)
|
||||
{
|
||||
mm_write_paf4(s, mi, t, r, km, opt_flag, rep_len, 0, 0);
|
||||
}
|
||||
|
||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag)
|
||||
{
|
||||
mm_write_paf3(s, mi, t, r, km, opt_flag, -1);
|
||||
@@ -369,14 +490,16 @@ static void write_sam_cigar(kstring_t *s, int sam_flag, int in_tag, int qlen, co
|
||||
clip_len[0] = r->rev? qlen - r->qe : r->qs;
|
||||
clip_len[1] = r->rev? r->qs : qlen - r->qe;
|
||||
if (in_tag) {
|
||||
int clip_char = (sam_flag&0x800) && !(opt_flag&MM_F_SOFTCLIP)? 5 : 4;
|
||||
int clip_char = (((sam_flag&0x800) || ((sam_flag&0x100) && (opt_flag&MM_F_SECONDARY_SEQ))) &&
|
||||
!(opt_flag&MM_F_SOFTCLIP)) ? 5 : 4;
|
||||
mm_sprintf_lite(s, "\tCG:B:I");
|
||||
if (clip_len[0]) mm_sprintf_lite(s, ",%u", clip_len[0]<<4|clip_char);
|
||||
for (k = 0; k < r->p->n_cigar; ++k)
|
||||
mm_sprintf_lite(s, ",%u", r->p->cigar[k]);
|
||||
if (clip_len[1]) mm_sprintf_lite(s, ",%u", clip_len[1]<<4|clip_char);
|
||||
} else {
|
||||
int clip_char = (sam_flag&0x800) && !(opt_flag&MM_F_SOFTCLIP)? 'H' : 'S';
|
||||
int clip_char = (((sam_flag&0x800) || ((sam_flag&0x100) && (opt_flag&MM_F_SECONDARY_SEQ))) &&
|
||||
!(opt_flag&MM_F_SOFTCLIP)) ? 'H' : 'S';
|
||||
assert(clip_len[0] < qlen && clip_len[1] < qlen);
|
||||
if (clip_len[0]) mm_sprintf_lite(s, "%d%c", clip_len[0], clip_char);
|
||||
for (k = 0; k < r->p->n_cigar; ++k)
|
||||
@@ -451,7 +574,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
||||
if (cigar_in_tag) {
|
||||
int slen;
|
||||
if ((flag & 0x900) == 0 || (opt_flag & MM_F_SOFTCLIP)) slen = t->l_seq;
|
||||
else if (flag & 0x100) slen = 0;
|
||||
else if ((flag & 0x100) && !(opt_flag & MM_F_SECONDARY_SEQ)) slen = 0;
|
||||
else slen = r->qe - r->qs;
|
||||
mm_sprintf_lite(s, "%dS%dN", slen, r->re - r->rs);
|
||||
} else write_sam_cigar(s, flag, 0, t->l_seq, r, opt_flag);
|
||||
@@ -492,7 +615,7 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
||||
mm_sprintf_lite(s, "\t");
|
||||
if (t->qual) sam_write_sq(s, t->qual, t->l_seq, r->rev, 0);
|
||||
else mm_sprintf_lite(s, "*");
|
||||
} else if (flag & 0x100) {
|
||||
} else if ((flag & 0x100) && !(opt_flag & MM_F_SECONDARY_SEQ)){
|
||||
mm_sprintf_lite(s, "*\t*");
|
||||
} else {
|
||||
sam_write_sq(s, t->seq + r->qs, r->qe - r->qs, r->rev, r->rev);
|
||||
@@ -532,8 +655,8 @@ void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int se
|
||||
}
|
||||
}
|
||||
}
|
||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_MD)))
|
||||
write_cs_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, 1, 0);
|
||||
if (r->p && (opt_flag & (MM_F_OUT_CS|MM_F_OUT_DS|MM_F_OUT_MD)))
|
||||
write_cs_ds_or_MD(km, s, mi, t, r, !(opt_flag&MM_F_OUT_CS_LONG), opt_flag&MM_F_OUT_MD, !!(opt_flag&MM_F_OUT_DS), 1, 0);
|
||||
if (cigar_in_tag)
|
||||
write_sam_cigar(s, flag, 1, t->l_seq, r, opt_flag);
|
||||
}
|
||||
|
||||
@@ -55,7 +55,7 @@ mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u,
|
||||
mm_reg1_t *r;
|
||||
int i, k;
|
||||
|
||||
if (n_u == 0) return 0;
|
||||
if (n_u <= 0) return 0;
|
||||
|
||||
// sort by score
|
||||
z = (mm128_t*)kmalloc(km, n_u * 16);
|
||||
@@ -279,7 +279,7 @@ int mm_filter_strand_retained(int n_regs, mm_reg1_t *r)
|
||||
int i, k;
|
||||
for (i = k = 0; i < n_regs; ++i) {
|
||||
int p = r[i].parent;
|
||||
if (!r[i].strand_retained || r[i].div < r[p].div * 5.0f) {
|
||||
if (!r[i].strand_retained || r[i].div < r[p].div * 5.0f || r[i].div < 0.01f) {
|
||||
if (k < i) r[k++] = r[i];
|
||||
else ++k;
|
||||
}
|
||||
@@ -418,16 +418,19 @@ static void mm_set_inv_mapq(void *km, int n_regs, mm_reg1_t *regs)
|
||||
kfree(km, aux);
|
||||
}
|
||||
|
||||
void mm_set_mapq(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr)
|
||||
void mm_set_mapq2(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr, int is_splice)
|
||||
{
|
||||
static const float q_coef = 40.0f;
|
||||
int64_t sum_sc = 0;
|
||||
float uniq_ratio;
|
||||
int i;
|
||||
int i, n_2nd_splice = 0;
|
||||
if (n_regs == 0) return;
|
||||
for (i = 0; i < n_regs; ++i)
|
||||
for (i = 0; i < n_regs; ++i) {
|
||||
if (regs[i].parent == regs[i].id)
|
||||
sum_sc += regs[i].score;
|
||||
else if (regs[i].is_spliced)
|
||||
++n_2nd_splice;
|
||||
}
|
||||
uniq_ratio = (float)sum_sc / (sum_sc + rep_len);
|
||||
for (i = 0; i < n_regs; ++i) {
|
||||
mm_reg1_t *r = ®s[i];
|
||||
@@ -440,13 +443,18 @@ void mm_set_mapq(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int ma
|
||||
pen_cm = pen_s1 < pen_cm? pen_s1 : pen_cm;
|
||||
subsc = r->subsc > min_chain_sc? r->subsc : min_chain_sc;
|
||||
if (r->p && r->p->dp_max2 > 0 && r->p->dp_max > 0) {
|
||||
float identity = (float)r->mlen / r->blen;
|
||||
float x = (float)r->p->dp_max2 * subsc / r->p->dp_max / r->score0;
|
||||
float x, identity = (float)r->mlen / r->blen;
|
||||
if (is_sr && is_splice)
|
||||
x = (float)r->p->dp_max2 / r->p->dp_max; // ignore chaining score; for short RNA-seq reads, unspliced chaining score tends to be higher
|
||||
else
|
||||
x = (float)r->p->dp_max2 * subsc / r->p->dp_max / r->score0;
|
||||
mapq = (int)(identity * pen_cm * q_coef * (1.0f - x * x) * logf((float)r->p->dp_max / match_sc));
|
||||
if (!is_sr) {
|
||||
int mapq_alt = (int)(6.02f * identity * identity * (r->p->dp_max - r->p->dp_max2) / match_sc + .499f); // BWA-MEM like mapQ, mostly for short reads
|
||||
mapq = mapq < mapq_alt? mapq : mapq_alt; // in case the long-read heuristic fails
|
||||
}
|
||||
if (is_splice && is_sr && r->is_spliced && n_2nd_splice == 0)
|
||||
mapq += 10;
|
||||
} else {
|
||||
float x = (float)subsc / r->score0;
|
||||
if (r->p) {
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
#include "bseq.h"
|
||||
#include "minimap.h"
|
||||
#include "mmpriv.h"
|
||||
#include "ksw2.h"
|
||||
#include "kvec.h"
|
||||
#include "khash.h"
|
||||
|
||||
@@ -32,7 +33,7 @@ typedef struct mm_idx_bucket_s {
|
||||
} mm_idx_bucket_t;
|
||||
|
||||
typedef struct {
|
||||
int32_t st, en, max; // max is not used for now
|
||||
int32_t st, en, cnt;
|
||||
int32_t score:30, strand:2;
|
||||
} mm_idx_intv1_t;
|
||||
|
||||
@@ -41,6 +42,11 @@ typedef struct mm_idx_intv_s {
|
||||
mm_idx_intv1_t *a;
|
||||
} mm_idx_intv_t;
|
||||
|
||||
typedef struct mm_idx_jjump_s {
|
||||
int32_t n, m;
|
||||
mm_idx_jjump1_t *a;
|
||||
} mm_idx_jjump_t;
|
||||
|
||||
mm_idx_t *mm_idx_init(int w, int k, int b, int flag)
|
||||
{
|
||||
mm_idx_t *mi;
|
||||
@@ -65,11 +71,17 @@ void mm_idx_destroy(mm_idx_t *mi)
|
||||
kh_destroy(idx, (idxhash_t*)mi->B[i].h);
|
||||
}
|
||||
}
|
||||
if (mi->spsc) free(mi->spsc);
|
||||
if (mi->I) {
|
||||
for (i = 0; i < mi->n_seq; ++i)
|
||||
free(mi->I[i].a);
|
||||
free(mi->I);
|
||||
}
|
||||
if (mi->J) {
|
||||
for (i = 0; i < mi->n_seq; ++i)
|
||||
free(mi->J[i].a);
|
||||
free(mi->J);
|
||||
}
|
||||
if (!mi->km) {
|
||||
for (i = 0; i < mi->n_seq; ++i)
|
||||
free(mi->seq[i].name);
|
||||
@@ -99,7 +111,7 @@ const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n)
|
||||
|
||||
void mm_idx_stat(const mm_idx_t *mi)
|
||||
{
|
||||
int n = 0, n1 = 0;
|
||||
int64_t n = 0, n1 = 0;
|
||||
uint32_t i;
|
||||
uint64_t sum = 0, len = 0;
|
||||
fprintf(stderr, "[M::%s] kmer size: %d; skip: %d; is_hpc: %d; #seq: %d\n", __func__, mi->k, mi->w, mi->flag&MM_I_HPC, mi->n_seq);
|
||||
@@ -117,8 +129,8 @@ void mm_idx_stat(const mm_idx_t *mi)
|
||||
if (kh_key(h, k)&1) ++n1;
|
||||
}
|
||||
}
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] distinct minimizers: %d (%.2f%% are singletons); average occurrences: %.3lf; average spacing: %.3lf; total length: %ld\n",
|
||||
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), n, 100.0*n1/n, (double)sum / n, (double)len / sum, (long)len);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] distinct minimizers: %ld (%.2f%% are singletons); average occurrences: %.3lf; average spacing: %.3lf; total length: %ld\n",
|
||||
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), (long)n, 100.0*n1/n, (double)sum / n, (double)len / sum, (long)len);
|
||||
}
|
||||
|
||||
int mm_idx_index_name(mm_idx_t *mi)
|
||||
@@ -192,6 +204,7 @@ int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f)
|
||||
if (f <= 0.) return INT32_MAX;
|
||||
for (i = 0; i < 1<<mi->b; ++i)
|
||||
if (mi->B[i].h) n += kh_size((idxhash_t*)mi->B[i].h);
|
||||
if (n == 0) return INT32_MAX;
|
||||
a = (uint32_t*)malloc(n * 4);
|
||||
for (i = n = 0; i < 1<<mi->b; ++i) {
|
||||
idxhash_t *h = (idxhash_t*)mi->B[i].h;
|
||||
@@ -656,10 +669,17 @@ int mm_idx_alt_read(mm_idx_t *mi, const char *fn)
|
||||
return n_alt;
|
||||
}
|
||||
|
||||
/***************
|
||||
* BED reading *
|
||||
***************/
|
||||
|
||||
#define sort_key_bed(a) ((a).st)
|
||||
KRADIX_SORT_INIT(bed, mm_idx_intv1_t, sort_key_bed, 4)
|
||||
|
||||
mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc)
|
||||
#define sort_key_end(a) ((a).en)
|
||||
KRADIX_SORT_INIT(end, mm_idx_intv1_t, sort_key_end, 4)
|
||||
|
||||
static mm_idx_intv_t *mm_idx_bed_read_core(const mm_idx_t *mi, const char *fn, int read_junc, int min_sc)
|
||||
{
|
||||
gzFile fp;
|
||||
kstream_t *ks;
|
||||
@@ -668,7 +688,7 @@ mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc
|
||||
|
||||
fp = fn && strcmp(fn, "-")? gzopen(fn, "r") : gzdopen(fileno(stdin), "r");
|
||||
if (fp == 0) return 0;
|
||||
I = (mm_idx_intv_t*)calloc(mi->n_seq, sizeof(*I));
|
||||
I = CALLOC(mm_idx_intv_t, mi->n_seq);
|
||||
ks = ks_init(fp);
|
||||
while (ks_getuntil(ks, KS_SEP_LINE, &str, 0) >= 0) {
|
||||
mm_idx_intv_t *r;
|
||||
@@ -689,7 +709,7 @@ mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc
|
||||
t.en = atol(q);
|
||||
if (t.en < 0) break;
|
||||
} else if (i == 4) { // BED score
|
||||
t.score = atol(q);
|
||||
t.score = *q >= '0' && *q <= '9'? atol(q) : -1;
|
||||
} else if (i == 5) { // strand
|
||||
t.strand = *q == '+'? 1 : *q == '-'? -1 : 0;
|
||||
} else if (i == 9) {
|
||||
@@ -705,7 +725,8 @@ mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc
|
||||
++i, q = p + 1;
|
||||
}
|
||||
}
|
||||
if (id < 0 || t.st < 0 || t.st >= t.en) continue;
|
||||
if (id < 0 || t.st < 0 || t.st >= t.en) continue; // contig ID not found, or other problems
|
||||
if (min_sc > 0 && t.score < min_sc) continue;
|
||||
r = &I[id];
|
||||
if (i >= 11 && read_junc) { // BED12
|
||||
int32_t st, sz, en;
|
||||
@@ -738,14 +759,44 @@ mm_idx_intv_t *mm_idx_read_bed(const mm_idx_t *mi, const char *fn, int read_junc
|
||||
return I;
|
||||
}
|
||||
|
||||
static mm_idx_intv_t *mm_idx_bed_read_merge(const mm_idx_t *mi, const char *fn, int read_junc, int min_sc)
|
||||
{
|
||||
long n = 0, n0 = 0;
|
||||
int32_t i;
|
||||
mm_idx_intv_t *I;
|
||||
I = mm_idx_bed_read_core(mi, fn, read_junc, min_sc);
|
||||
if (I == 0) return 0;
|
||||
for (i = 0; i < mi->n_seq; ++i) {
|
||||
int32_t j, j0, k;
|
||||
mm_idx_intv_t *intv = &I[i];
|
||||
n0 += intv->n;
|
||||
radix_sort_bed(intv->a, intv->a + intv->n); // sort by st
|
||||
for (j = 1, j0 = 0; j <= intv->n; ++j) { // sort by st and then by end
|
||||
if (j == intv->n || intv->a[j].st != intv->a[j0].st) {
|
||||
radix_sort_end(intv->a + j0, intv->a + j);
|
||||
j0 = j;
|
||||
}
|
||||
}
|
||||
for (j = 1, j0 = 0, k = 0; j <= intv->n; ++j) { // merge intervals with the same (st, en)
|
||||
if (j == intv->n || intv->a[j].st != intv->a[j0].st || intv->a[j].en != intv->a[j0].en) {
|
||||
intv->a[k] = intv->a[j0];
|
||||
intv->a[k++].cnt = j - j0;
|
||||
j0 = j;
|
||||
}
|
||||
}
|
||||
intv->a = REALLOC(mm_idx_intv1_t, intv->a, k);
|
||||
intv->n = intv->m = k;
|
||||
n += k;
|
||||
}
|
||||
if (mm_verbose >= 3)
|
||||
fprintf(stderr, "[%s] read %ld introns, %ld of which are non-redundant\n", __func__, n0, n);
|
||||
return I;
|
||||
}
|
||||
|
||||
int mm_idx_bed_read(mm_idx_t *mi, const char *fn, int read_junc)
|
||||
{
|
||||
int32_t i;
|
||||
if (mi->h == 0) mm_idx_index_name(mi);
|
||||
mi->I = mm_idx_read_bed(mi, fn, read_junc);
|
||||
if (mi->I == 0) return -1;
|
||||
for (i = 0; i < mi->n_seq; ++i) // TODO: eliminate redundant intervals
|
||||
radix_sort_bed(mi->I[i].a, mi->I[i].a + mi->I[i].n);
|
||||
mi->I = mm_idx_bed_read_merge(mi, fn, read_junc, -1);
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -773,3 +824,244 @@ int mm_idx_bed_junc(const mm_idx_t *mi, int32_t ctg, int32_t st, int32_t en, uin
|
||||
}
|
||||
return left;
|
||||
}
|
||||
|
||||
/*********************************
|
||||
* Reading junctions for jumping *
|
||||
*********************************/
|
||||
|
||||
#define sort_key_jj(a) ((a).off)
|
||||
KRADIX_SORT_INIT(jj, mm_idx_jjump1_t, sort_key_jj, 4)
|
||||
|
||||
#define sort_key_jj2(a) ((a).off2)
|
||||
KRADIX_SORT_INIT(jj2, mm_idx_jjump1_t, sort_key_jj2, 4)
|
||||
|
||||
static void sort_jjump(mm_idx_jjump_t *jj2)
|
||||
{
|
||||
int32_t j0, j, k;
|
||||
if (jj2 == 0 || jj2->n == 0) return;
|
||||
radix_sort_jj(jj2->a, jj2->a + jj2->n);
|
||||
for (j0 = 0, j = 1; j <= jj2->n; ++j) {
|
||||
if (j == jj2->n || jj2->a[j0].off != jj2->a[j].off) {
|
||||
radix_sort_jj2(jj2->a + j0, jj2->a + j);
|
||||
j0 = j;
|
||||
}
|
||||
}
|
||||
// the actual merge
|
||||
for (j0 = 0, j = 1, k = 0; j <= jj2->n; ++j) {
|
||||
if (j == jj2->n || jj2->a[j0].off != jj2->a[j].off || jj2->a[j0].off2 != jj2->a[j].off2) {
|
||||
int32_t t, cnt = 0;
|
||||
uint16_t flag = 0;
|
||||
for (t = j0; t < j; ++t) cnt += jj2->a[t].cnt, flag |= jj2->a[t].flag;
|
||||
jj2->a[k] = jj2->a[j0];
|
||||
jj2->a[k].cnt = cnt;
|
||||
jj2->a[k++].flag = flag;
|
||||
j0 = j;
|
||||
}
|
||||
}
|
||||
jj2->n = k;
|
||||
jj2->a = REALLOC(mm_idx_jjump1_t, jj2->a, k);
|
||||
}
|
||||
|
||||
static mm_idx_jjump_t *mm_idx_bed2jjump(const mm_idx_t *mi, const mm_idx_intv_t *I, uint16_t flag)
|
||||
{
|
||||
int32_t i;
|
||||
mm_idx_jjump_t *J;
|
||||
J = CALLOC(mm_idx_jjump_t, mi->n_seq);
|
||||
for (i = 0; i < mi->n_seq; ++i) {
|
||||
int32_t j, k;
|
||||
const mm_idx_intv_t *intv = &I[i];
|
||||
mm_idx_jjump_t *jj = &J[i];
|
||||
jj->n = intv->n * 2;
|
||||
jj->a = CALLOC(mm_idx_jjump1_t, jj->n);
|
||||
for (j = k = 0; j < intv->n; ++j) {
|
||||
jj->a[k].off = intv->a[j].st, jj->a[k].off2 = intv->a[j].en, jj->a[k].cnt = intv->a[j].cnt, jj->a[k].strand = intv->a[j].strand, jj->a[k++].flag = flag;
|
||||
jj->a[k].off = intv->a[j].en, jj->a[k].off2 = intv->a[j].st, jj->a[k].cnt = intv->a[j].cnt, jj->a[k].strand = intv->a[j].strand, jj->a[k++].flag = flag;
|
||||
}
|
||||
sort_jjump(jj);
|
||||
}
|
||||
return J;
|
||||
}
|
||||
|
||||
static mm_idx_jjump_t *mm_idx_jjump_merge(const mm_idx_t *mi, const mm_idx_jjump_t *J0, const mm_idx_jjump_t *J1)
|
||||
{
|
||||
int32_t i;
|
||||
mm_idx_jjump_t *J2;
|
||||
J2 = CALLOC(mm_idx_jjump_t, mi->n_seq);
|
||||
for (i = 0; i < mi->n_seq; ++i) {
|
||||
int32_t j, k;
|
||||
const mm_idx_jjump_t *jj0 = &J0[i], *jj1 = &J1[i];
|
||||
mm_idx_jjump_t *jj2 = &J2[i];
|
||||
jj2->n = jj0->n + jj1->n;
|
||||
jj2->a = CALLOC(mm_idx_jjump1_t, jj2->n);
|
||||
for (j = k = 0; j < jj0->n; ++j) jj2->a[k++] = jj0->a[j];
|
||||
for (j = 0; j < jj1->n; ++j) jj2->a[k++] = jj1->a[j];
|
||||
sort_jjump(jj2);
|
||||
}
|
||||
return J2;
|
||||
}
|
||||
|
||||
int mm_idx_jjump_read(mm_idx_t *mi, const char *fn, int flag, int min_sc)
|
||||
{
|
||||
int32_t i, j, n_anno = 0, n_misc = 0;
|
||||
mm_idx_intv_t *I;
|
||||
mm_idx_jjump_t *J;
|
||||
if (mi->h == 0) mm_idx_index_name(mi);
|
||||
I = mm_idx_bed_read_merge(mi, fn, 1, min_sc);
|
||||
J = mm_idx_bed2jjump(mi, I, flag);
|
||||
for (i = 0; i < mi->n_seq; ++i) free(I[i].a);
|
||||
free(I);
|
||||
if (mi->J) {
|
||||
mm_idx_jjump_t *J2;
|
||||
J2 = mm_idx_jjump_merge(mi, mi->J, J);
|
||||
for (i = 0; i < mi->n_seq; ++i) {
|
||||
free(mi->J[i].a); free(J[i].a);
|
||||
}
|
||||
free(mi->J); free(J);
|
||||
mi->J = J2;
|
||||
} else mi->J = J;
|
||||
for (i = 0; i < mi->n_seq; ++i) {
|
||||
for (j = 0; j < mi->J[i].n; ++j)
|
||||
if (mi->J[i].a[j].flag & MM_JUNC_ANNO) ++n_anno;
|
||||
else ++n_misc;
|
||||
}
|
||||
if (mm_verbose >= 3)
|
||||
fprintf(stderr, "[%s] there are %d annotated and %d other splice positions in the index\n", __func__, n_anno, n_misc);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int32_t mm_idx_jump_get_core(int32_t n, const mm_idx_jjump1_t *a, int32_t x) // similar to mm_idx_find_intv()
|
||||
{
|
||||
int32_t s = 0, e = n;
|
||||
if (n == 0) return -1;
|
||||
if (x < a[0].off) return -1;
|
||||
while (s < e) {
|
||||
int32_t mid = s + (e - s) / 2;
|
||||
if (x >= a[mid].off && (mid + 1 >= n || x < a[mid+1].off)) return mid;
|
||||
else if (x < a[mid].off) e = mid;
|
||||
else s = mid + 1;
|
||||
}
|
||||
assert(0);
|
||||
}
|
||||
|
||||
const mm_idx_jjump1_t *mm_idx_jump_get(const mm_idx_t *db, int32_t cid, int32_t st, int32_t en, int32_t *n)
|
||||
{
|
||||
mm_idx_jjump_t *s;
|
||||
int32_t l, r;
|
||||
*n = 0;
|
||||
if (cid >= db->n_seq || cid < 0 || db->J == 0) return 0;
|
||||
if (en < 0 || en > db->seq[cid].len) en = db->seq[cid].len;
|
||||
s = &db->J[cid];
|
||||
if (s->n == 0) return 0;
|
||||
l = mm_idx_jump_get_core(s->n, s->a, st);
|
||||
r = mm_idx_jump_get_core(s->n, s->a, en);
|
||||
*n = r - l;
|
||||
return &s->a[l + 1];
|
||||
}
|
||||
|
||||
/****************
|
||||
* splice score *
|
||||
****************/
|
||||
|
||||
typedef struct mm_idx_spsc_s {
|
||||
uint32_t n, m;
|
||||
uint64_t *a; // pos<<56 | score<<1 | acceptor
|
||||
} mm_idx_spsc_t;
|
||||
|
||||
int32_t mm_idx_spsc_read(mm_idx_t *idx, const char *fn, int32_t max_sc)
|
||||
{
|
||||
gzFile fp;
|
||||
kstring_t str = {0,0,0};
|
||||
kstream_t *ks;
|
||||
int32_t dret, j;
|
||||
int64_t n_read = 0;
|
||||
|
||||
fp = fn && strcmp(fn, "-") != 0? gzopen(fn, "rb") : gzdopen(0, "rb");
|
||||
if (fp == 0) return -1;
|
||||
if (idx->h == 0) mm_idx_index_name(idx);
|
||||
if (max_sc > 63) max_sc = 63;
|
||||
idx->spsc = Kcalloc(0, mm_idx_spsc_t, idx->n_seq * 2);
|
||||
ks = ks_init(fp);
|
||||
while (ks_getuntil(ks, KS_SEP_LINE, &str, &dret) >= 0) {
|
||||
mm_idx_spsc_t *s;
|
||||
char *p, *q, *name = 0;
|
||||
int32_t i, type = -1, strand = 0, cid = -1, score = -1;
|
||||
int64_t pos = -1;
|
||||
for (i = 0, p = q = str.s;; ++p) {
|
||||
if (*p == '\t' || *p == 0) {
|
||||
int c = *p;
|
||||
*p = 0;
|
||||
if (i == 0) {
|
||||
name = q;
|
||||
} else if (i == 1) {
|
||||
pos = atol(q);
|
||||
} else if (i == 2) {
|
||||
strand = *q == '+'? 1 : '-'? -1 : 0;
|
||||
} else if (i == 3) {
|
||||
type = *q == 'D'? 0 : *q == 'A'? 1 : -1;
|
||||
} else if (i == 4) {
|
||||
score = atoi(q);
|
||||
break;
|
||||
}
|
||||
if (c == 0) break;
|
||||
q = p + 1, ++i;
|
||||
}
|
||||
}
|
||||
if (i < 4) continue; // not enough fields
|
||||
if (score > max_sc) score = max_sc;
|
||||
if (score < -max_sc) score = -max_sc;
|
||||
cid = mm_idx_name2id(idx, name);
|
||||
if (cid < 0 || type < 0 || strand == 0 || pos < 0) continue; // FIXME: give a warning!
|
||||
s = &idx->spsc[cid << 1 | (strand > 0? 0 : 1)];
|
||||
Kgrow(0, uint64_t, s->a, s->n, s->m);
|
||||
if (pos > 0 && pos < idx->seq[cid].len) { // ignore scores at the ends
|
||||
s->a[s->n++] = (uint64_t)pos << 8 | (score + KSW_SPSC_OFFSET) << 1 | type;
|
||||
++n_read;
|
||||
}
|
||||
}
|
||||
ks_destroy(ks);
|
||||
gzclose(fp);
|
||||
for (j = 0; j < idx->n_seq * 2; ++j) {
|
||||
mm_idx_spsc_t *s = &idx->spsc[j];
|
||||
if (s->n > 0)
|
||||
radix_sort_64(s->a, s->a + s->n);
|
||||
}
|
||||
if (mm_verbose >= 3)
|
||||
fprintf(stderr, "[M::%s] read %ld splice scores\n", __func__, (long)n_read);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int32_t mm_idx_find_intv(int32_t n, const uint64_t *a, int64_t x)
|
||||
{
|
||||
int32_t s = 0, e = n;
|
||||
if (n == 0) return -1;
|
||||
if (x < a[0]>>8) return -1;
|
||||
while (s < e) {
|
||||
int32_t mid = s + (e - s) / 2;
|
||||
if (x >= a[mid]>>8 && (mid + 1 >= n || x < a[mid+1]>>8)) return mid;
|
||||
else if (x < a[mid]>>8) e = mid;
|
||||
else s = mid + 1;
|
||||
}
|
||||
assert(0);
|
||||
}
|
||||
|
||||
int64_t mm_idx_spsc_get(const mm_idx_t *db, int32_t cid, int64_t st, int64_t en, int32_t rev, uint8_t *sc)
|
||||
{
|
||||
const mm_idx_spsc_t *s;
|
||||
if (cid >= db->n_seq || cid < 0 || db->spsc == 0) return -1;
|
||||
if (en < 0 || en > db->seq[cid].len) en = db->seq[cid].len;
|
||||
memset(sc, 0xff, en - st);
|
||||
s = &db->spsc[cid << 1 | (!!rev)];
|
||||
if (s->n > 0) {
|
||||
int32_t j, l, r;
|
||||
l = mm_idx_find_intv(s->n, s->a, st);
|
||||
r = mm_idx_find_intv(s->n, s->a, en);
|
||||
for (j = l + 1; j <= r; ++j) {
|
||||
int64_t x = (s->a[j]>>8) - st;
|
||||
uint8_t score = s->a[j] & 0xff;
|
||||
assert(x <= en - st);
|
||||
if (x == en - st) continue;
|
||||
if (sc[x] == 0xff || sc[x] < score) sc[x] = score;
|
||||
}
|
||||
}
|
||||
return en - st;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,201 @@
|
||||
#include <stdio.h>
|
||||
#include "mmpriv.h"
|
||||
#include "kalloc.h"
|
||||
|
||||
#define MM_MIN_EXON_LEN 20
|
||||
|
||||
static int32_t mm_jump_check(void *km, const mm_idx_t *mi, int32_t qlen, const uint8_t *qseq0, const mm_reg1_t *r, int32_t ext, int32_t is_left) // TODO: check close N
|
||||
{
|
||||
int32_t clip, clen, e = !r->rev ^ !is_left; // 0 for left of the alignment; 1 for right
|
||||
uint32_t cigar;
|
||||
if (!r->p || r->p->n_cigar <= 0) return -1; // only working with CIGAR
|
||||
clip = e == 0? r->qs : qlen - r->qe;
|
||||
cigar = r->p->cigar[is_left? 0 : r->p->n_cigar - 1];
|
||||
clen = (cigar&0xf) == MM_CIGAR_MATCH? cigar>>4 : 0;
|
||||
if (clen <= ext) return -1;
|
||||
if (is_left) {
|
||||
if (clip >= r->rs) return -1; // no space to jump
|
||||
} else {
|
||||
if (clip >= mi->seq[r->rid].len - r->re) return -1; // no space to jump
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static uint8_t *mm_jump_get_qseq_seq(void *km, int32_t qlen, const uint8_t *qseq0, const mm_reg1_t *r, int32_t is_left, int32_t ql0, uint8_t *qseq)
|
||||
{
|
||||
extern unsigned char seq_nt4_table[256];
|
||||
int32_t i, k = 0;
|
||||
if (!r->rev) {
|
||||
if (is_left)
|
||||
for (i = 0; i < ql0; ++i)
|
||||
qseq[k++] = seq_nt4_table[(uint8_t)qseq0[i]];
|
||||
else
|
||||
for (i = qlen - ql0; i < qlen; ++i)
|
||||
qseq[k++] = seq_nt4_table[(uint8_t)qseq0[i]];
|
||||
} else {
|
||||
if (is_left)
|
||||
for (i = qlen - 1; i >= qlen - ql0; --i) {
|
||||
uint8_t c = seq_nt4_table[(uint8_t)qseq0[i]];
|
||||
qseq[k++] = c >= 4? c : 3 - c;
|
||||
}
|
||||
else
|
||||
for (i = ql0 - 1; i >= 0; --i) {
|
||||
uint8_t c = seq_nt4_table[(uint8_t)qseq0[i]];
|
||||
qseq[k++] = c >= 4? c : 3 - c;
|
||||
}
|
||||
}
|
||||
return qseq;
|
||||
}
|
||||
|
||||
static void mm_jump_split_left(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq0, mm_reg1_t *r, int32_t ts_strand)
|
||||
{
|
||||
uint8_t *tseq = 0, *qseq = 0;
|
||||
int32_t i, n, l, i0, m, mm0;
|
||||
int32_t i0_anno = -1, n_anno = 0, mm0_anno = 0, i0_misc = -1, n_misc = 0, mm0_misc = 0;
|
||||
int32_t ext = 1 + (opt->b + opt->a - 1) / opt->a + 1;
|
||||
int32_t clip = !r->rev? r->qs : qlen - r->qe;
|
||||
int32_t extt = clip < ext? clip : ext;
|
||||
const mm_idx_jjump1_t *a;
|
||||
|
||||
if (mm_jump_check(km, mi, qlen, qseq0, r, ext + MM_MIN_EXON_LEN, 1) < 0) return;
|
||||
a = mm_idx_jump_get(mi, r->rid, r->rs - extt, r->rs + ext, &n);
|
||||
if (n == 0) return;
|
||||
|
||||
for (i = 0; i < n; ++i) { // traverse possible jumps
|
||||
const mm_idx_jjump1_t *ai = &a[i];
|
||||
int32_t tlen, tl1, j, mm1, mm2;
|
||||
assert(ai->off >= r->rs - extt && ai->off <= r->rs + ext);
|
||||
if (ts_strand * ai->strand < 0) continue; // wrong strand
|
||||
if (ai->off2 >= ai->off) continue; // wrong direction
|
||||
if (ai->off - ai->off2 < 6) continue; // intron too small
|
||||
if (ai->off2 < clip + ext) continue; // not long enough
|
||||
if (tseq == 0) {
|
||||
tseq = Kcalloc(km, uint8_t, (clip + ext) * 2); // tseq and qseq are allocated together
|
||||
qseq = tseq + clip + ext;
|
||||
mm_jump_get_qseq_seq(km, qlen, qseq0, r, 1, clip + ext, qseq);
|
||||
}
|
||||
tl1 = clip + (ai->off - r->rs);
|
||||
tlen = mm_idx_getseq2(mi, 0, r->rid, ai->off, r->rs + ext, &tseq[tl1]);
|
||||
assert(tlen == r->rs + ext - ai->off);
|
||||
tlen = mm_idx_getseq2(mi, 0, r->rid, ai->off2 - tl1, ai->off2, tseq);
|
||||
assert(tlen == tl1);
|
||||
for (j = 0, mm1 = 0; j < tl1; ++j)
|
||||
if (qseq[j] != tseq[j] || qseq[j] > 3 || tseq[j] > 3)
|
||||
++mm1;
|
||||
for (mm2 = 0; j < clip + ext; ++j)
|
||||
if (qseq[j] != tseq[j] || qseq[j] > 3 || tseq[j] > 3)
|
||||
++mm2;
|
||||
if (mm1 == 0 && mm2 <= 1) {
|
||||
if (ai->flag & MM_JUNC_ANNO)
|
||||
i0_anno = i, mm0_anno = mm1 + mm2, ++n_anno; // i0 points to the rightmost i
|
||||
else
|
||||
i0_misc = i, mm0_misc = mm1 + mm2, ++n_misc;
|
||||
}
|
||||
}
|
||||
if (n_anno > 0) m = n_anno, i0 = i0_anno, mm0 = mm0_anno;
|
||||
else m = n_misc, i0 = i0_misc, mm0 = mm0_misc;
|
||||
kfree(km, tseq);
|
||||
|
||||
l = m > 0? a[i0].off - r->rs : 0; // may be negative
|
||||
if (m == 1 && clip + l >= opt->jump_min_match) { // add one more exon
|
||||
mm_enlarge_cigar(r, 2);
|
||||
memmove(r->p->cigar + 2, r->p->cigar, r->p->n_cigar * 4);
|
||||
r->p->cigar[0] = (clip + l) << 4 | MM_CIGAR_MATCH;
|
||||
r->p->cigar[1] = (a[i0].off - a[i0].off2) << 4 | MM_CIGAR_N_SKIP;
|
||||
r->p->cigar[2] = ((r->p->cigar[2]>>4) - l) << 4 | MM_CIGAR_MATCH;
|
||||
r->p->n_cigar += 2;
|
||||
r->rs = a[i0].off2 - (clip + l);
|
||||
if (!r->rev) r->qs = 0;
|
||||
else r->qe = qlen;
|
||||
r->blen += clip, r->mlen += clip - mm0;
|
||||
r->p->dp_max0 += (clip - mm0) * opt->a - mm0 * opt->b;
|
||||
r->p->dp_max += (clip - mm0) * opt->a - mm0 * opt->b;
|
||||
if (!r->is_spliced) r->is_spliced = 1, r->p->dp_max += (opt->a + opt->b) + ((opt->a + opt->b) >> 1);
|
||||
} else if (m > 0 && a[i0].off > r->rs) { // trim by l; l is always positive
|
||||
r->p->cigar[0] -= l << 4 | MM_CIGAR_MATCH;
|
||||
r->rs += l;
|
||||
if (!r->rev) r->qs += l;
|
||||
else r->qe -= l;
|
||||
}
|
||||
}
|
||||
|
||||
static void mm_jump_split_right(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq0, mm_reg1_t *r, int32_t ts_strand)
|
||||
{
|
||||
uint8_t *tseq = 0, *qseq = 0;
|
||||
int32_t i, n, l, i0, m, mm0;
|
||||
int32_t i0_anno = -1, n_anno = 0, mm0_anno = 0, i0_misc = -1, n_misc = 0, mm0_misc = 0;
|
||||
int32_t ext = 1 + (opt->b + opt->a - 1) / opt->a + 1;
|
||||
int32_t clip = !r->rev? qlen - r->qe : r->qs;
|
||||
int32_t extt = clip < ext? clip : ext;
|
||||
const mm_idx_jjump1_t *a;
|
||||
|
||||
if (mm_jump_check(km, mi, qlen, qseq0, r, ext + MM_MIN_EXON_LEN, 0) < 0) return;
|
||||
a = mm_idx_jump_get(mi, r->rid, r->re - ext, r->re + extt, &n);
|
||||
if (n == 0) return;
|
||||
|
||||
for (i = 0; i < n; ++i) { // traverse possible jumps
|
||||
const mm_idx_jjump1_t *ai = &a[i];
|
||||
int32_t tlen, tl1, j, mm1, mm2;
|
||||
assert(ai->off >= r->re - ext && ai->off <= r->re + extt);
|
||||
if (ts_strand * ai->strand < 0) continue; // wrong strand
|
||||
if (ai->off2 <= ai->off) continue; // wrong direction
|
||||
if (ai->off2 - ai->off < 6) continue; // intron too small
|
||||
if (ai->off2 + clip + ext > mi->seq[r->rid].len) continue; // not long enough
|
||||
if (tseq == 0) {
|
||||
tseq = Kcalloc(km, uint8_t, (clip + ext) * 2); // tseq and qseq are allocated together
|
||||
qseq = tseq + clip + ext;
|
||||
mm_jump_get_qseq_seq(km, qlen, qseq0, r, 0, clip + ext, qseq);
|
||||
}
|
||||
tl1 = clip + (r->re - ai->off);
|
||||
tlen = mm_idx_getseq2(mi, 0, r->rid, r->re - ext, ai->off, tseq);
|
||||
assert(tlen == ai->off - (r->re - ext));
|
||||
tlen = mm_idx_getseq2(mi, 0, r->rid, ai->off2, ai->off2 + tl1, &tseq[clip + ext - tl1]);
|
||||
assert(tlen == tl1);
|
||||
for (j = 0, mm2 = 0; j < clip + ext - tl1; ++j)
|
||||
if (qseq[j] != tseq[j] || qseq[j] > 3 || tseq[j] > 3)
|
||||
++mm2;
|
||||
for (mm1 = 0; j < clip + ext; ++j)
|
||||
if (qseq[j] != tseq[j] || qseq[j] > 3 || tseq[j] > 3)
|
||||
++mm1;
|
||||
if (mm1 == 0 && mm2 <= 1) {
|
||||
if (ai->flag & MM_JUNC_ANNO) {
|
||||
if (i0_anno < 0) i0_anno = i, mm0_anno = mm1 + mm2;
|
||||
++n_anno;
|
||||
} else {
|
||||
if (i0_misc < 0) i0_misc = i, mm0_misc = mm1 + mm2;
|
||||
++n_misc;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (n_anno > 0) m = n_anno, i0 = i0_anno, mm0 = mm0_anno;
|
||||
else m = n_misc, i0 = i0_misc, mm0 = mm0_misc;
|
||||
kfree(km, tseq);
|
||||
|
||||
l = m > 0? r->re - a[i0].off : 0; // may be negative
|
||||
if (m == 1 && clip + l >= opt->jump_min_match) { // add one more exon
|
||||
mm_enlarge_cigar(r, 2);
|
||||
r->p->cigar[r->p->n_cigar - 1] = ((r->p->cigar[r->p->n_cigar - 1]>>4) - l) << 4 | MM_CIGAR_MATCH;
|
||||
r->p->cigar[r->p->n_cigar] = (a[i0].off2 - a[i0].off) << 4 | MM_CIGAR_N_SKIP;
|
||||
r->p->cigar[r->p->n_cigar + 1] = (clip + l) << 4 | MM_CIGAR_MATCH;
|
||||
r->p->n_cigar += 2;
|
||||
r->re = a[i0].off2 + (clip + l);
|
||||
if (!r->rev) r->qe = qlen;
|
||||
else r->qs = 0;
|
||||
r->blen += clip, r->mlen += clip - mm0;
|
||||
r->p->dp_max0 += (clip - mm0) * opt->a - mm0 * opt->b;
|
||||
r->p->dp_max += (clip - mm0) * opt->a - mm0 * opt->b;
|
||||
if (!r->is_spliced) r->is_spliced = 1, r->p->dp_max += (opt->a + opt->b) + ((opt->a + opt->b) >> 1);
|
||||
} else if (m > 0 && r->re > a[i0].off) { // trim by l; l is always positive
|
||||
r->p->cigar[r->p->n_cigar - 1] -= l << 4 | MM_CIGAR_MATCH;
|
||||
r->re -= l;
|
||||
if (!r->rev) r->qe -= l;
|
||||
else r->qs += l;
|
||||
}
|
||||
}
|
||||
|
||||
void mm_jump_split(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq, mm_reg1_t *r, int32_t ts_strand)
|
||||
{
|
||||
assert((opt->flag & MM_F_EQX) == 0);
|
||||
mm_jump_split_left(km, mi, opt, qlen, qseq, r, ts_strand);
|
||||
mm_jump_split_right(km, mi, opt, qlen, qseq, r, ts_strand);
|
||||
}
|
||||
@@ -40,7 +40,8 @@ void *km_init2(void *km_par, size_t min_core_size)
|
||||
kmem_t *km;
|
||||
km = (kmem_t*)kcalloc(km_par, 1, sizeof(kmem_t));
|
||||
km->par = km_par;
|
||||
km->min_core_size = min_core_size > 0? min_core_size : 0x80000;
|
||||
if (km_par) km->min_core_size = min_core_size > 0? min_core_size : ((kmem_t*)km_par)->min_core_size - 2;
|
||||
else km->min_core_size = min_core_size > 0? min_core_size : 0x80000;
|
||||
return (void*)km;
|
||||
}
|
||||
|
||||
@@ -183,6 +184,16 @@ void *krealloc(void *_km, void *ap, size_t n_bytes) // TODO: this can be made mo
|
||||
return q;
|
||||
}
|
||||
|
||||
void *krelocate(void *km, void *ap, size_t n_bytes)
|
||||
{
|
||||
void *p;
|
||||
if (km == 0 || ap == 0) return ap;
|
||||
p = kmalloc(km, n_bytes);
|
||||
memcpy(p, ap, n_bytes);
|
||||
kfree(km, ap);
|
||||
return p;
|
||||
}
|
||||
|
||||
void km_stat(const void *_km, km_stat_t *s)
|
||||
{
|
||||
kmem_t *km = (kmem_t*)_km;
|
||||
@@ -203,3 +214,11 @@ void km_stat(const void *_km, km_stat_t *s)
|
||||
s->largest = s->largest > size? s->largest : size;
|
||||
}
|
||||
}
|
||||
|
||||
void km_stat_print(const void *km)
|
||||
{
|
||||
km_stat_t st;
|
||||
km_stat(km, &st);
|
||||
fprintf(stderr, "[km_stat] cap=%ld, avail=%ld, largest=%ld, n_core=%ld, n_block=%ld\n",
|
||||
st.capacity, st.available, st.largest, st.n_blocks, st.n_cores);
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ typedef struct {
|
||||
|
||||
void *kmalloc(void *km, size_t size);
|
||||
void *krealloc(void *km, void *ptr, size_t size);
|
||||
void *krelocate(void *km, void *ap, size_t n_bytes);
|
||||
void *kcalloc(void *km, size_t count, size_t size);
|
||||
void kfree(void *km, void *ptr);
|
||||
|
||||
@@ -20,11 +21,29 @@ void *km_init(void);
|
||||
void *km_init2(void *km_par, size_t min_core_size);
|
||||
void km_destroy(void *km);
|
||||
void km_stat(const void *_km, km_stat_t *s);
|
||||
void km_stat_print(const void *km);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#define Kmalloc(km, type, cnt) ((type*)kmalloc((km), (cnt) * sizeof(type)))
|
||||
#define Kcalloc(km, type, cnt) ((type*)kcalloc((km), (cnt), sizeof(type)))
|
||||
#define Krealloc(km, type, ptr, cnt) ((type*)krealloc((km), (ptr), (cnt) * sizeof(type)))
|
||||
|
||||
#define Kgrow(km, type, ptr, __i, __m) do { \
|
||||
if ((__i) >= (__m)) { \
|
||||
(__m) = (__i) + 1; \
|
||||
(__m) += ((__m)>>1) + 16; \
|
||||
(ptr) = Krealloc(km, type, ptr, (__m)); \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
#define Kexpand(km, type, a, m) do { \
|
||||
(m) = (m) >= 4? (m) + ((m)>>1) : 16; \
|
||||
(a) = Krealloc(km, type, (a), (m)); \
|
||||
} while (0)
|
||||
|
||||
#define KMALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kmalloc((km), (len) * sizeof(*(ptr))))
|
||||
#define KCALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kcalloc((km), (len), sizeof(*(ptr))))
|
||||
#define KREALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))krealloc((km), (ptr), (len) * sizeof(*(ptr))))
|
||||
@@ -50,7 +69,7 @@ void km_stat(const void *_km, km_stat_t *s);
|
||||
} kmp_##name##_t; \
|
||||
SCOPE kmp_##name##_t *kmp_init_##name(void *km) { \
|
||||
kmp_##name##_t *mp; \
|
||||
KCALLOC(km, mp, 1); \
|
||||
mp = Kcalloc(km, kmp_##name##_t, 1); \
|
||||
mp->km = km; \
|
||||
return mp; \
|
||||
} \
|
||||
@@ -66,7 +85,7 @@ void km_stat(const void *_km, km_stat_t *s);
|
||||
} \
|
||||
SCOPE void kmp_free_##name(kmp_##name##_t *mp, kmptype_t *p) { \
|
||||
--mp->cnt; \
|
||||
if (mp->n == mp->max) KEXPAND(mp->km, mp->buf, mp->max); \
|
||||
if (mp->n == mp->max) Kexpand(mp->km, kmptype_t*, mp->buf, mp->max); \
|
||||
mp->buf[mp->n++] = p; \
|
||||
}
|
||||
|
||||
|
||||
@@ -15,6 +15,8 @@
|
||||
#define KSW_EZ_SPLICE_FOR 0x100
|
||||
#define KSW_EZ_SPLICE_REV 0x200
|
||||
#define KSW_EZ_SPLICE_FLANK 0x400
|
||||
#define KSW_EZ_SPLICE_CMPLX 0x800 // use the miniprot splice model
|
||||
#define KSW_EZ_SPLICE_SCORE 0x1000 // use splice score
|
||||
|
||||
// The subset of CIGAR operators used by ksw code.
|
||||
// Use MM_CIGAR_* from minimap.h if you need the full list.
|
||||
@@ -23,6 +25,8 @@
|
||||
#define KSW_CIGAR_DEL 2
|
||||
#define KSW_CIGAR_N_SKIP 3
|
||||
|
||||
#define KSW_SPSC_OFFSET 64
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
@@ -68,7 +72,7 @@ void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t gape2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||
|
||||
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
||||
|
||||
void ksw_extf2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t mch, int8_t mis, int8_t e, int w, int xdrop, ksw_extz_t *ez);
|
||||
|
||||
|
||||
+5
-5
@@ -80,17 +80,17 @@ void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
||||
}
|
||||
|
||||
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||
{
|
||||
extern void ksw_exts2_sse2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
||||
extern void ksw_exts2_sse41(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez);
|
||||
if (ksw_simd < 0) ksw_simd = x86_simd();
|
||||
if (ksw_simd & SIMD_SSE4_1)
|
||||
ksw_exts2_sse41(km, qlen, query, tlen, target, m, mat, q, e, q2, noncan, zdrop, junc_bonus, flag, junc, ez);
|
||||
ksw_exts2_sse41(km, qlen, query, tlen, target, m, mat, q, e, q2, noncan, zdrop, end_bonus, junc_bonus, junc_pen, flag, junc, ez);
|
||||
else if (ksw_simd & SIMD_SSE2)
|
||||
ksw_exts2_sse2(km, qlen, query, tlen, target, m, mat, q, e, q2, noncan, zdrop, junc_bonus, flag, junc, ez);
|
||||
ksw_exts2_sse2(km, qlen, query, tlen, target, m, mat, q, e, q2, noncan, zdrop, end_bonus, junc_bonus, junc_pen, flag, junc, ez);
|
||||
else abort();
|
||||
}
|
||||
#endif
|
||||
|
||||
+1
-1
@@ -358,7 +358,7 @@ void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
||||
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||
// update ez
|
||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||
ez->mte = H[en0], ez->mte_q = r - en;
|
||||
ez->mte = H[en0], ez->mte_q = r - en0;
|
||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e2)) break;
|
||||
|
||||
+95
-45
@@ -24,14 +24,14 @@
|
||||
#ifdef KSW_CPU_DISPATCH
|
||||
#ifdef __SSE4_1__
|
||||
void ksw_exts2_sse41(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||
#else
|
||||
void ksw_exts2_sse2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||
#endif
|
||||
#else
|
||||
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int8_t junc_bonus, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||
int8_t q, int8_t e, int8_t q2, int8_t noncan, int zdrop, int end_bonus, int8_t junc_bonus, int8_t junc_pen, int flag, const uint8_t *junc, ksw_extz_t *ez)
|
||||
#endif // ~KSW_CPU_DISPATCH
|
||||
{
|
||||
#define __dp_code_block1 \
|
||||
@@ -71,6 +71,7 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
||||
|
||||
ksw_reset_extz(ez);
|
||||
if (m <= 1 || qlen <= 0 || tlen <= 0 || q2 <= q + e) return;
|
||||
assert((flag & KSW_EZ_SPLICE_FOR) == 0 || (flag & KSW_EZ_SPLICE_REV) == 0); // can't be both set
|
||||
|
||||
zero_ = _mm_set1_epi8(0);
|
||||
q_ = _mm_set1_epi8(q);
|
||||
@@ -118,55 +119,100 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
||||
|
||||
// set the donor and acceptor arrays. TODO: this assumes 0/1/2/3 encoding!
|
||||
if (flag & (KSW_EZ_SPLICE_FOR|KSW_EZ_SPLICE_REV)) {
|
||||
int semi_cost = flag&KSW_EZ_SPLICE_FLANK? -noncan/2 : 0; // GTr or yAG is worth 0.5 bit; see PMID:18688272
|
||||
memset(donor, -noncan, tlen_ * 16);
|
||||
memset(acceptor, -noncan, tlen_ * 16);
|
||||
const int sp0[4] = { 8, 15, 21, 30 };
|
||||
int sp[4];
|
||||
if (flag & KSW_EZ_SPLICE_CMPLX) {
|
||||
for (t = 0; t < 4; ++t)
|
||||
sp[t] = (int)((double)sp0[t] / 3. + .499);
|
||||
} else {
|
||||
sp[0] = flag&KSW_EZ_SPLICE_FLANK? noncan / 2 : 0;
|
||||
sp[1] = sp[2] = sp[3] = noncan;
|
||||
}
|
||||
memset(donor, -sp[3], tlen_ * 16);
|
||||
memset(acceptor, -sp[3], tlen_ * 16);
|
||||
if (!(flag & KSW_EZ_REV_CIGAR)) {
|
||||
for (t = 0; t < tlen - 4; ++t) {
|
||||
int can_type = 0; // type of canonical site: 0=none, 1=GT/AG only, 2=GTr/yAG
|
||||
if ((flag & KSW_EZ_SPLICE_FOR) && target[t+1] == 2 && target[t+2] == 3) can_type = 1; // GTr...
|
||||
if ((flag & KSW_EZ_SPLICE_REV) && target[t+1] == 1 && target[t+2] == 3) can_type = 1; // CTr...
|
||||
if (can_type && (target[t+3] == 0 || target[t+3] == 2)) can_type = 2;
|
||||
if (can_type) ((int8_t*)donor)[t] = can_type == 2? 0 : semi_cost;
|
||||
int z = 3;
|
||||
if (flag & KSW_EZ_SPLICE_FOR) {
|
||||
if (target[t+1] == 2 && target[t+2] == 3) // |GT.
|
||||
z = target[t+3] == 0 || target[t+3] == 2? -1 : 0; // |GTr or not
|
||||
else if (target[t+1] == 2 && target[t+2] == 1) z = 1; // |GC.
|
||||
else if (target[t+1] == 0 && target[t+2] == 3) z = 2; // |AT.
|
||||
} else if (flag & KSW_EZ_SPLICE_REV) {
|
||||
if (target[t+1] == 1 && target[t+2] == 3) // |CT. (revcomp of .AG|)
|
||||
z = target[t+3] == 0 || target[t+3] == 2? -1 : 0;
|
||||
else if (target[t+1] == 2 && target[t+2] == 3) z = 2; // |GT. (revcomp of .AC|)
|
||||
}
|
||||
((int8_t*)donor)[t] = z < 0? 0 : -sp[z];
|
||||
}
|
||||
if (junc)
|
||||
for (t = 0; t < tlen - 1; ++t)
|
||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&8)))
|
||||
((int8_t*)donor)[t] += junc_bonus;
|
||||
for (t = 2; t < tlen; ++t) {
|
||||
int can_type = 0;
|
||||
if ((flag & KSW_EZ_SPLICE_FOR) && target[t-1] == 0 && target[t] == 2) can_type = 1; // ...yAG
|
||||
if ((flag & KSW_EZ_SPLICE_REV) && target[t-1] == 0 && target[t] == 1) can_type = 1; // ...yAC
|
||||
if (can_type && (target[t-2] == 1 || target[t-2] == 3)) can_type = 2;
|
||||
if (can_type) ((int8_t*)acceptor)[t] = can_type == 2? 0 : semi_cost;
|
||||
int z = 3;
|
||||
if (flag & KSW_EZ_SPLICE_FOR) {
|
||||
if (target[t-1] == 0 && target[t] == 2) // .AG|
|
||||
z = target[t-2] == 1 || target[t-2] == 3? -1 : 0; // yAG| or not
|
||||
else if (target[t-1] == 0 && target[t] == 1) z = 2; // .AC|
|
||||
} else if (flag & KSW_EZ_SPLICE_REV) {
|
||||
if (target[t-1] == 0 && target[t] == 1) // .AC| (revcomp of |GT.)
|
||||
z = target[t-2] == 1 || target[t-2] == 3? -1 : 0; // yAC| or not
|
||||
else if (target[t-1] == 2 && target[t] == 1) z = 1; // .GC| (revcomp of |GC.)
|
||||
else if (target[t-1] == 0 && target[t] == 3) z = 2; // .AT| (revcomp of |AT.)
|
||||
}
|
||||
((int8_t*)acceptor)[t] = z < 0? 0 : -sp[z];
|
||||
}
|
||||
if (junc)
|
||||
for (t = 0; t < tlen; ++t)
|
||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&4)))
|
||||
((int8_t*)acceptor)[t] += junc_bonus;
|
||||
} else {
|
||||
for (t = 0; t < tlen - 4; ++t) {
|
||||
int can_type = 0; // type of canonical site: 0=none, 1=GT/AG only, 2=GTr/yAG
|
||||
if ((flag & KSW_EZ_SPLICE_FOR) && target[t+1] == 2 && target[t+2] == 0) can_type = 1; // GAy...
|
||||
if ((flag & KSW_EZ_SPLICE_REV) && target[t+1] == 1 && target[t+2] == 0) can_type = 1; // CAy...
|
||||
if (can_type && (target[t+3] == 1 || target[t+3] == 3)) can_type = 2;
|
||||
if (can_type) ((int8_t*)donor)[t] = can_type == 2? 0 : semi_cost;
|
||||
int z = 3;
|
||||
if (flag & KSW_EZ_SPLICE_FOR) {
|
||||
if (target[t+1] == 2 && target[t+2] == 0) // |GA. (rev of .AG|)
|
||||
z = target[t+3] == 1 || target[t+3] == 3? -1 : 0;
|
||||
else if (target[t+1] == 1 && target[t+2] == 0) z = 2; // |CA. (rev of .AC|)
|
||||
} else if (flag & KSW_EZ_SPLICE_REV) {
|
||||
if (target[t+1] == 1 && target[t+2] == 0) // |CA. (comp of |GT.)
|
||||
z = target[t+3] == 1 || target[t+3] == 3? -1 : 0;
|
||||
else if (target[t+1] == 1 && target[t+2] == 2) z = 1; // |CG. (comp of |GC.)
|
||||
else if (target[t+1] == 3 && target[t+2] == 0) z = 2; // |TA. (comp of |AT.)
|
||||
}
|
||||
((int8_t*)donor)[t] = z < 0? 0 : -sp[z];
|
||||
}
|
||||
if (junc)
|
||||
for (t = 0; t < tlen - 1; ++t)
|
||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&4)))
|
||||
((int8_t*)donor)[t] += junc_bonus;
|
||||
for (t = 2; t < tlen; ++t) {
|
||||
int can_type = 0;
|
||||
if ((flag & KSW_EZ_SPLICE_FOR) && target[t-1] == 3 && target[t] == 2) can_type = 1; // ...rTG
|
||||
if ((flag & KSW_EZ_SPLICE_REV) && target[t-1] == 3 && target[t] == 1) can_type = 1; // ...rTC
|
||||
if (can_type && (target[t-2] == 0 || target[t-2] == 2)) can_type = 2;
|
||||
if (can_type) ((int8_t*)acceptor)[t] = can_type == 2? 0 : semi_cost;
|
||||
int z = 3;
|
||||
if (flag & KSW_EZ_SPLICE_FOR) {
|
||||
if (target[t-1] == 3 && target[t] == 2) // .TG| (rev of |GT.)
|
||||
z = target[t-2] == 0 || target[t-2] == 2? -1 : 0;
|
||||
else if (target[t-1] == 1 && target[t] == 2) z = 1; // .CG| (rev of |GC.)
|
||||
else if (target[t-1] == 3 && target[t] == 0) z = 2; // .TA| (rev of |AT.)
|
||||
} else if (flag & KSW_EZ_SPLICE_REV) {
|
||||
if (target[t-1] == 3 && target[t] == 1) // .TC| (comp of .AG|)
|
||||
z = target[t-2] == 0 || target[t-2] == 2? -1 : 0;
|
||||
else if (target[t-1] == 3 && target[t] == 2) z = 2; // .TG| (comp of .AC|)
|
||||
}
|
||||
((int8_t*)acceptor)[t] = z < 0? 0 : -sp[z];
|
||||
}
|
||||
if (junc)
|
||||
for (t = 0; t < tlen; ++t)
|
||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&8)))
|
||||
((int8_t*)acceptor)[t] += junc_bonus;
|
||||
}
|
||||
}
|
||||
|
||||
if (junc && (flag & KSW_EZ_SPLICE_SCORE)) { // junc[] keeps the donor score
|
||||
uint8_t donor_val = !!(flag & KSW_EZ_SPLICE_FOR) == !(flag & KSW_EZ_REV_CIGAR)? 0 : 1;
|
||||
for (t = 0; t < tlen - 1; ++t)
|
||||
((int8_t*)donor)[t] += junc[t+1] == 0xff || (junc[t+1]&1) != donor_val? -junc_pen : (int8_t)(junc[t+1]>>1) - (int8_t)KSW_SPSC_OFFSET;
|
||||
for (t = 0; t < tlen - 1; ++t)
|
||||
((int8_t*)acceptor)[t] += junc[t+1] == 0xff || (junc[t+1]&1) != !donor_val? -junc_pen : (int8_t)(junc[t+1]>>1) - (int8_t)KSW_SPSC_OFFSET;
|
||||
//for (t = 0; t < tlen - 1; ++t) if (junc[t+1] != 0xff) fprintf(stderr, "Y2\t%d\t%d\t%c\t%d\n", ((int8_t*)donor)[t], ((int8_t*)acceptor)[t], "DA"[junc[t+1]&1], (int8_t)(junc[t+1]>>1) - (int8_t)KSW_SPSC_OFFSET);
|
||||
} else if (junc) { // junc[] keeps the splice sites
|
||||
if (!(flag & KSW_EZ_REV_CIGAR)) {
|
||||
for (t = 0; t < tlen - 1; ++t)
|
||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&8)))
|
||||
((int8_t*)donor)[t] += junc_bonus;
|
||||
for (t = 0; t < tlen; ++t)
|
||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&4)))
|
||||
((int8_t*)acceptor)[t] += junc_bonus;
|
||||
} else {
|
||||
for (t = 0; t < tlen - 1; ++t)
|
||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t+1]&2)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t+1]&4)))
|
||||
((int8_t*)donor)[t] += junc_bonus;
|
||||
for (t = 0; t < tlen; ++t)
|
||||
if (((flag & KSW_EZ_SPLICE_FOR) && (junc[t]&1)) || ((flag & KSW_EZ_SPLICE_REV) && (junc[t]&8)))
|
||||
((int8_t*)acceptor)[t] += junc_bonus;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -376,7 +422,7 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
||||
} else H[0] = v8[0] - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||
// update ez
|
||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||
ez->mte = H[en0], ez->mte_q = r - en;
|
||||
ez->mte = H[en0], ez->mte_q = r - en0;
|
||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, 0)) break;
|
||||
@@ -406,10 +452,14 @@ void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
||||
if (!approx_max) kfree(km, H);
|
||||
if (with_cigar) { // backtrack
|
||||
int rev_cigar = !!(flag & KSW_EZ_REV_CIGAR);
|
||||
if (!ez->zdropped && !(flag&KSW_EZ_EXTZ_ONLY))
|
||||
if (!ez->zdropped && !(flag&KSW_EZ_EXTZ_ONLY)) {
|
||||
ksw_backtrack(km, 1, rev_cigar, long_thres, (uint8_t*)p, off, off_end, n_col_*16, tlen-1, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||
else if (ez->max_t >= 0 && ez->max_q >= 0)
|
||||
} else if (!ez->zdropped && (flag&KSW_EZ_EXTZ_ONLY) && ez->mqe + end_bonus > (int)ez->max) {
|
||||
ez->reach_end = 1;
|
||||
ksw_backtrack(km, 1, rev_cigar, long_thres, (uint8_t*)p, off, off_end, n_col_*16, ez->mqe_t, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||
} else if (ez->max_t >= 0 && ez->max_q >= 0) {
|
||||
ksw_backtrack(km, 1, rev_cigar, long_thres, (uint8_t*)p, off, off_end, n_col_*16, ez->max_t, ez->max_q, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||
}
|
||||
kfree(km, mem2); kfree(km, off);
|
||||
}
|
||||
}
|
||||
|
||||
+1
-1
@@ -269,7 +269,7 @@ void ksw_extz2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uin
|
||||
} else H[0] = v8[0] - qe - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||
// update ez
|
||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||
ez->mte = H[en0], ez->mte_q = r - en;
|
||||
ez->mte = H[en0], ez->mte_q = r - en0;
|
||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e)) break;
|
||||
|
||||
@@ -6,7 +6,25 @@
|
||||
#include "kalloc.h"
|
||||
#include "krmq.h"
|
||||
|
||||
uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_t *p, int32_t *v, int32_t *t, int32_t min_cnt, int32_t min_sc, int32_t *n_u_, int32_t *n_v_)
|
||||
static int64_t mg_chain_bk_end(int32_t max_drop, const mm128_t *z, const int32_t *f, const int64_t *p, int32_t *t, int64_t k)
|
||||
{
|
||||
int64_t i = z[k].y, end_i = -1, max_i = i;
|
||||
int32_t max_s = 0;
|
||||
if (i < 0 || t[i] != 0) return i;
|
||||
do {
|
||||
int32_t s;
|
||||
t[i] = 2;
|
||||
end_i = i = p[i];
|
||||
s = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (s > max_s) max_s = s, max_i = i;
|
||||
else if (max_s - s > max_drop) break;
|
||||
} while (i >= 0 && t[i] == 0);
|
||||
for (i = z[k].y; i >= 0 && i != end_i; i = p[i]) // reset modified t[]
|
||||
t[i] = 0;
|
||||
return max_i;
|
||||
}
|
||||
|
||||
uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_t *p, int32_t *v, int32_t *t, int32_t min_cnt, int32_t min_sc, int32_t max_drop, int32_t *n_u_, int32_t *n_v_)
|
||||
{
|
||||
mm128_t *z;
|
||||
uint64_t *u;
|
||||
@@ -17,33 +35,39 @@ uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_
|
||||
for (i = 0, n_z = 0; i < n; ++i) // precompute n_z
|
||||
if (f[i] >= min_sc) ++n_z;
|
||||
if (n_z == 0) return 0;
|
||||
KMALLOC(km, z, n_z);
|
||||
z = Kmalloc(km, mm128_t, n_z);
|
||||
for (i = 0, k = 0; i < n; ++i) // populate z[]
|
||||
if (f[i] >= min_sc) z[k].x = f[i], z[k++].y = i;
|
||||
radix_sort_128x(z, z + n_z);
|
||||
|
||||
memset(t, 0, n * 4);
|
||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // precompute n_u
|
||||
int64_t n_v0 = n_v;
|
||||
int32_t sc;
|
||||
for (i = z[k].y; i >= 0 && t[i] == 0; i = p[i])
|
||||
++n_v, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
++n_u;
|
||||
else n_v = n_v0;
|
||||
if (t[z[k].y] == 0) {
|
||||
int64_t n_v0 = n_v, end_i;
|
||||
int32_t sc;
|
||||
end_i = mg_chain_bk_end(max_drop, z, f, p, t, k);
|
||||
for (i = z[k].y; i != end_i; i = p[i])
|
||||
++n_v, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
++n_u;
|
||||
else n_v = n_v0;
|
||||
}
|
||||
}
|
||||
KMALLOC(km, u, n_u);
|
||||
u = Kmalloc(km, uint64_t, n_u);
|
||||
memset(t, 0, n * 4);
|
||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // populate u[]
|
||||
int64_t n_v0 = n_v;
|
||||
int32_t sc;
|
||||
for (i = z[k].y; i >= 0 && t[i] == 0; i = p[i])
|
||||
v[n_v++] = i, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
u[n_u++] = (uint64_t)sc << 32 | (n_v - n_v0);
|
||||
else n_v = n_v0;
|
||||
if (t[z[k].y] == 0) {
|
||||
int64_t n_v0 = n_v, end_i;
|
||||
int32_t sc;
|
||||
end_i = mg_chain_bk_end(max_drop, z, f, p, t, k);
|
||||
for (i = z[k].y; i != end_i; i = p[i])
|
||||
v[n_v++] = i, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
u[n_u++] = (uint64_t)sc << 32 | (n_v - n_v0);
|
||||
else n_v = n_v0;
|
||||
}
|
||||
}
|
||||
kfree(km, z);
|
||||
assert(n_v < INT32_MAX);
|
||||
@@ -58,7 +82,7 @@ static mm128_t *compact_a(void *km, int32_t n_u, uint64_t *u, int32_t n_v, int32
|
||||
int64_t i, j, k;
|
||||
|
||||
// write the result to b[]
|
||||
KMALLOC(km, b, n_v);
|
||||
b = Kmalloc(km, mm128_t, n_v);
|
||||
for (i = 0, k = 0; i < n_u; ++i) {
|
||||
int32_t k0 = k, ni = (int32_t)u[i];
|
||||
for (j = 0; j < ni; ++j)
|
||||
@@ -67,13 +91,13 @@ static mm128_t *compact_a(void *km, int32_t n_u, uint64_t *u, int32_t n_v, int32
|
||||
kfree(km, v);
|
||||
|
||||
// sort u[] and a[] by the target position, such that adjacent chains may be joined
|
||||
KMALLOC(km, w, n_u);
|
||||
w = Kmalloc(km, mm128_t, n_u);
|
||||
for (i = k = 0; i < n_u; ++i) {
|
||||
w[i].x = b[k].x, w[i].y = (uint64_t)k<<32|i;
|
||||
k += (int32_t)u[i];
|
||||
}
|
||||
radix_sort_128x(w, w + n_u);
|
||||
KMALLOC(km, u2, n_u);
|
||||
u2 = Kmalloc(km, uint64_t, n_u);
|
||||
for (i = k = 0; i < n_u; ++i) {
|
||||
int32_t j = (int32_t)w[i].y, n = (int32_t)u[j];
|
||||
u2[i] = u[j];
|
||||
@@ -114,7 +138,7 @@ static inline int32_t comput_sc(const mm128_t *ai, const mm128_t *aj, int32_t ma
|
||||
}
|
||||
|
||||
/* Input:
|
||||
* a[].x: tid<<33 | rev<<32 | tpos
|
||||
* a[].x: rev<<63 | tid<<32 | tpos
|
||||
* a[].y: flags<<40 | q_span<<32 | q_pos
|
||||
* Output:
|
||||
* n_u: #chains
|
||||
@@ -124,8 +148,8 @@ static inline int32_t comput_sc(const mm128_t *ai, const mm128_t *aj, int32_t ma
|
||||
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
int is_cdna, int n_seg, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||
{ // TODO: make sure this works when n has more than 32 bits
|
||||
int32_t *f, *t, *v, n_u, n_v, mmax_f = 0;
|
||||
int64_t *p, i, j, max_ii, st = 0, n_iter = 0;
|
||||
int32_t *f, *t, *v, n_u, n_v, mmax_f = 0, max_drop = bw;
|
||||
int64_t *p, i, j, max_ii, st = 0;
|
||||
uint64_t *u;
|
||||
|
||||
if (_u) *_u = 0, *n_u_ = 0;
|
||||
@@ -135,10 +159,11 @@ mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int
|
||||
}
|
||||
if (max_dist_x < bw) max_dist_x = bw;
|
||||
if (max_dist_y < bw && !is_cdna) max_dist_y = bw;
|
||||
KMALLOC(km, p, n);
|
||||
KMALLOC(km, f, n);
|
||||
KMALLOC(km, v, n);
|
||||
KCALLOC(km, t, n);
|
||||
if (is_cdna) max_drop = INT32_MAX;
|
||||
p = Kmalloc(km, int64_t, n);
|
||||
f = Kmalloc(km, int32_t, n);
|
||||
v = Kmalloc(km, int32_t, n);
|
||||
t = Kcalloc(km, int32_t, n);
|
||||
|
||||
// fill the score and backtrack arrays
|
||||
for (i = 0, max_ii = -1; i < n; ++i) {
|
||||
@@ -149,7 +174,6 @@ mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int
|
||||
for (j = i - 1; j >= st; --j) {
|
||||
int32_t sc;
|
||||
sc = comput_sc(&a[i], &a[j], max_dist_x, max_dist_y, bw, chn_pen_gap, chn_pen_skip, is_cdna, n_seg);
|
||||
++n_iter;
|
||||
if (sc == INT32_MIN) continue;
|
||||
sc += f[j];
|
||||
if (sc > max_f) {
|
||||
@@ -179,9 +203,10 @@ mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int
|
||||
if (max_ii < 0 || (a[i].x - a[max_ii].x <= (int64_t)max_dist_x && f[max_ii] < f[i]))
|
||||
max_ii = i;
|
||||
if (mmax_f < max_f) mmax_f = max_f;
|
||||
//fprintf(stderr, "X1\t%ld\t%ld:%d\t%ld\t%ld:%d\t%ld\t%ld\n", (long)i, (long)(a[i].x>>32), (int32_t)a[i].x, (long)max_j, max_j<0?-1L:(long)(a[max_j].x>>32), max_j<0?-1:(int32_t)a[max_j].x, (long)max_f, (long)v[i]);
|
||||
}
|
||||
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, &n_u, &n_v);
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, max_drop, &n_u, &n_v);
|
||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||
kfree(km, p); kfree(km, f); kfree(km, t);
|
||||
if (n_u == 0) {
|
||||
@@ -225,8 +250,8 @@ static inline int32_t comput_sc_simple(const mm128_t *ai, const mm128_t *aj, flo
|
||||
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||
{
|
||||
int32_t *f,*t, *v, n_u, n_v, mmax_f = 0, max_rmq_size = 0;
|
||||
int64_t *p, i, i0, st = 0, st_inner = 0, n_iter = 0;
|
||||
int32_t *f,*t, *v, n_u, n_v, mmax_f = 0, max_rmq_size = 0, max_drop = bw;
|
||||
int64_t *p, i, i0, st = 0, st_inner = 0;
|
||||
uint64_t *u;
|
||||
lc_elem_t *root = 0, *root_inner = 0;
|
||||
void *mem_mp = 0;
|
||||
@@ -238,11 +263,12 @@ mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_ski
|
||||
return 0;
|
||||
}
|
||||
if (max_dist < bw) max_dist = bw;
|
||||
if (max_dist_inner <= 0 || max_dist_inner >= max_dist) max_dist_inner = 0;
|
||||
KMALLOC(km, p, n);
|
||||
KMALLOC(km, f, n);
|
||||
KCALLOC(km, t, n);
|
||||
KMALLOC(km, v, n);
|
||||
if (max_dist_inner < 0) max_dist_inner = 0;
|
||||
if (max_dist_inner > max_dist) max_dist_inner = max_dist;
|
||||
p = Kmalloc(km, int64_t, n);
|
||||
f = Kmalloc(km, int32_t, n);
|
||||
t = Kcalloc(km, int32_t, n);
|
||||
v = Kmalloc(km, int32_t, n);
|
||||
mem_mp = km_init2(km, 0x10000);
|
||||
mp = kmp_init_rmq(mem_mp);
|
||||
|
||||
@@ -300,12 +326,11 @@ mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_ski
|
||||
krmq_interval(lc_elem, root_inner, &s, &lo, &hi);
|
||||
if (lo) {
|
||||
const lc_elem_t *q;
|
||||
int32_t width, n_rmq_iter = 0;
|
||||
int32_t width;
|
||||
krmq_itr_t(lc_elem) itr;
|
||||
krmq_itr_find(lc_elem, root_inner, lo, &itr);
|
||||
while ((q = krmq_at(&itr)) != 0) {
|
||||
if (q->y < (int32_t)a[i].y - max_dist_inner) break;
|
||||
++n_rmq_iter;
|
||||
j = q->i;
|
||||
sc = f[j] + comput_sc_simple(&a[i], &a[j], chn_pen_gap, chn_pen_skip, 0, &width);
|
||||
if (width <= bw) {
|
||||
@@ -320,7 +345,6 @@ mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_ski
|
||||
}
|
||||
if (!krmq_itr_prev(lc_elem, &itr)) break;
|
||||
}
|
||||
n_iter += n_rmq_iter;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -333,7 +357,7 @@ mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_ski
|
||||
}
|
||||
km_destroy(mem_mp);
|
||||
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, &n_u, &n_v);
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, max_drop, &n_u, &n_v);
|
||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||
kfree(km, p); kfree(km, f); kfree(km, t);
|
||||
if (n_u == 0) {
|
||||
|
||||
@@ -7,8 +7,6 @@
|
||||
#include "mmpriv.h"
|
||||
#include "ketopt.h"
|
||||
|
||||
#define MM_VERSION "2.23-r1111"
|
||||
|
||||
#ifdef __linux__
|
||||
#include <sys/resource.h>
|
||||
#include <sys/time.h>
|
||||
@@ -37,12 +35,12 @@ static ko_longopt_t long_options[] = {
|
||||
{ "splice", ko_no_argument, 310 },
|
||||
{ "cost-non-gt-ag", ko_required_argument, 'C' },
|
||||
{ "no-long-join", ko_no_argument, 312 },
|
||||
{ "sr", ko_no_argument, 313 },
|
||||
{ "sr", ko_optional_argument, 313 },
|
||||
{ "frag", ko_required_argument, 314 },
|
||||
{ "secondary", ko_required_argument, 315 },
|
||||
{ "cs", ko_optional_argument, 316 },
|
||||
{ "end-bonus", ko_required_argument, 317 },
|
||||
{ "no-pairing", ko_no_argument, 318 },
|
||||
{ "no-pairing", ko_no_argument, 318 }, // deprecated but reserved for backward compatibility
|
||||
{ "splice-flank", ko_required_argument, 319 },
|
||||
{ "idx-no-seq", ko_no_argument, 320 },
|
||||
{ "end-seed-pen", ko_required_argument, 321 },
|
||||
@@ -76,6 +74,18 @@ static ko_longopt_t long_options[] = {
|
||||
{ "cap-kalloc", ko_required_argument, 349 },
|
||||
{ "q-occ-frac", ko_required_argument, 350 },
|
||||
{ "chain-skip-scale",ko_required_argument,351 },
|
||||
{ "print-chains", ko_no_argument, 352 },
|
||||
{ "no-hash-name", ko_no_argument, 353 },
|
||||
{ "secondary-seq", ko_no_argument, 354 },
|
||||
{ "ds", ko_no_argument, 355 },
|
||||
{ "rmq-inner", ko_required_argument, 356 },
|
||||
{ "spsc", ko_required_argument, 357 },
|
||||
{ "junc-pen", ko_required_argument, 358 },
|
||||
{ "pairing", ko_required_argument, 359 },
|
||||
{ "jump-min-match", ko_required_argument, 360 },
|
||||
{ "write-junc", ko_no_argument, 361 },
|
||||
{ "pass1", ko_required_argument, 362 },
|
||||
{ "dbg-seed-occ", ko_no_argument, 501 },
|
||||
{ "help", ko_no_argument, 'h' },
|
||||
{ "max-intron-len", ko_required_argument, 'G' },
|
||||
{ "version", ko_no_argument, 'V' },
|
||||
@@ -119,12 +129,12 @@ static inline void yes_or_no(mm_mapopt_t *opt, int64_t flag, int long_idx, const
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
const char *opt_str = "2aSDw:k:K:t:r:f:Vv:g:G:I:d:XT:s:x:Hcp:M:n:z:A:B:O:E:m:N:Qu:R:hF:LC:yYPo:e:U:";
|
||||
const char *opt_str = "2aSDw:k:K:t:r:f:Vv:g:G:I:d:XT:s:x:Hcp:M:n:z:A:B:b:O:E:m:N:Qu:R:hF:LC:yYPo:e:U:J:j:";
|
||||
ketopt_t o = KETOPT_INIT;
|
||||
mm_mapopt_t opt;
|
||||
mm_idxopt_t ipt;
|
||||
int i, c, n_threads = 3, n_parts, old_best_n = -1;
|
||||
char *fnw = 0, *rg = 0, *junc_bed = 0, *s, *alt_list = 0;
|
||||
char *fnw = 0, *rg = 0, *fn_bed_junc = 0, *fn_bed_jump = 0, *fn_bed_pass1 = 0, *fn_spsc = 0, *s, *alt_list = 0;
|
||||
FILE *fp_help = stderr;
|
||||
mm_idx_reader_t *idx_rdr;
|
||||
mm_idx_t *mi;
|
||||
@@ -177,6 +187,7 @@ int main(int argc, char *argv[])
|
||||
else if (c == 'm') opt.min_chain_score = atoi(o.arg);
|
||||
else if (c == 'A') opt.a = atoi(o.arg);
|
||||
else if (c == 'B') opt.b = atoi(o.arg);
|
||||
else if (c == 'b') opt.transition = atoi(o.arg);
|
||||
else if (c == 's') opt.min_dp_max = atoi(o.arg);
|
||||
else if (c == 'C') opt.noncan = atoi(o.arg);
|
||||
else if (c == 'I') ipt.batch_size = mm_parse_num(o.arg);
|
||||
@@ -185,7 +196,13 @@ int main(int argc, char *argv[])
|
||||
else if (c == 'R') rg = o.arg;
|
||||
else if (c == 'h') fp_help = stdout;
|
||||
else if (c == '2') opt.flag |= MM_F_2_IO_THREADS;
|
||||
else if (c == 'o') {
|
||||
else if (c == 'j') fn_bed_jump = o.arg;
|
||||
else if (c == 'J') {
|
||||
int t;
|
||||
t = atoi(o.arg);
|
||||
if (t == 0) opt.flag |= MM_F_SPLICE_OLD;
|
||||
else if (t == 1) opt.flag &= ~MM_F_SPLICE_OLD;
|
||||
} else if (c == 'o') {
|
||||
if (strcmp(o.arg, "-") != 0) {
|
||||
if (freopen(o.arg, "wb", stdout) == NULL) {
|
||||
fprintf(stderr, "[ERROR]\033[1;31m failed to write the output to file '%s'\033[0m: %s\n", o.arg, strerror(errno));
|
||||
@@ -204,9 +221,8 @@ int main(int argc, char *argv[])
|
||||
else if (c == 309) mm_dbg_flag |= MM_DBG_PRINT_QNAME | MM_DBG_PRINT_ALN_SEQ, n_threads = 1; // --print-aln-seq
|
||||
else if (c == 310) opt.flag |= MM_F_SPLICE; // --splice
|
||||
else if (c == 312) opt.flag |= MM_F_NO_LJOIN; // --no-long-join
|
||||
else if (c == 313) opt.flag |= MM_F_SR; // --sr
|
||||
else if (c == 317) opt.end_bonus = atoi(o.arg); // --end-bonus
|
||||
else if (c == 318) opt.flag |= MM_F_INDEPEND_SEG; // --no-pairing
|
||||
else if (c == 318) opt.flag |= MM_F_INDEPEND_SEG; // --no-pairing (deprecated)
|
||||
else if (c == 320) ipt.flag |= MM_I_NO_SEQ; // --idx-no-seq
|
||||
else if (c == 321) opt.anchor_ext_shift = atoi(o.arg); // --end-seed-pen
|
||||
else if (c == 322) opt.flag |= MM_F_FOR_ONLY; // --for-only
|
||||
@@ -222,8 +238,9 @@ int main(int argc, char *argv[])
|
||||
else if (c == 336) opt.flag |= MM_F_HARD_MLEVEL; // --hard-mask-level
|
||||
else if (c == 337) opt.max_sw_mat = mm_parse_num(o.arg); // --cap-sw-mat
|
||||
else if (c == 338) opt.max_qlen = mm_parse_num(o.arg); // --max-qlen
|
||||
else if (c == 340) junc_bed = o.arg; // --junc-bed
|
||||
else if (c == 340) fn_bed_junc = o.arg; // --junc-bed
|
||||
else if (c == 341) opt.junc_bonus = atoi(o.arg); // --junc-bonus
|
||||
else if (c == 358) opt.junc_pen = atoi(o.arg); // --junc-pen
|
||||
else if (c == 342) opt.flag |= MM_F_SAM_HIT_ONLY; // --sam-hit-only
|
||||
else if (c == 343) opt.chain_gap_scale = atof(o.arg); // --chain-gap-scale
|
||||
else if (c == 351) opt.chain_skip_scale = atof(o.arg); // --chain-skip-scale
|
||||
@@ -233,8 +250,29 @@ int main(int argc, char *argv[])
|
||||
else if (c == 348) opt.flag |= MM_F_QSTRAND | MM_F_NO_INV; // --qstrand
|
||||
else if (c == 349) opt.cap_kalloc = mm_parse_num(o.arg); // --cap-kalloc
|
||||
else if (c == 350) opt.q_occ_frac = atof(o.arg); // --q-occ-frac
|
||||
else if (c == 352) mm_dbg_flag |= MM_DBG_PRINT_CHAIN; // --print-chains
|
||||
else if (c == 353) opt.flag |= MM_F_NO_HASH_NAME; // --no-hash-name
|
||||
else if (c == 354) opt.flag |= MM_F_SECONDARY_SEQ; // --secondary-seq
|
||||
else if (c == 355) opt.flag |= MM_F_OUT_DS; // --ds
|
||||
else if (c == 356) opt.rmq_inner_dist = mm_parse_num(o.arg); // --rmq-inner
|
||||
else if (c == 357) fn_spsc = o.arg; // --spsc
|
||||
else if (c == 360) opt.jump_min_match = mm_parse_num(o.arg); // --jump-min-match
|
||||
else if (c == 361) opt.flag |= MM_F_OUT_JUNC | MM_F_CIGAR; // --write-junc
|
||||
else if (c == 362) fn_bed_pass1 = o.arg; // --jump-pass1
|
||||
else if (c == 501) mm_dbg_flag |= MM_DBG_SEED_FREQ; // --dbg-seed-occ
|
||||
else if (c == 330) {
|
||||
fprintf(stderr, "[WARNING] \033[1;31m --lj-min-ratio has been deprecated.\033[0m\n");
|
||||
} else if (c == 313) { // --sr
|
||||
if (o.arg == 0 || strcmp(o.arg, "dna") == 0) {
|
||||
opt.flag |= MM_F_SR;
|
||||
} else if (strcmp(o.arg, "rna") == 0) {
|
||||
opt.flag |= MM_F_SR_RNA;
|
||||
} else if (strcmp(o.arg, "no") == 0) {
|
||||
opt.flag &= ~(uint64_t)(MM_F_SR|MM_F_SR_RNA);
|
||||
} else if (mm_verbose >= 2) {
|
||||
opt.flag |= MM_F_SR;
|
||||
fprintf(stderr, "[WARNING]\033[1;31m --sr only takes 'dna' or 'rna'. Invalid values are assumed to be 'dna'.\033[0m\n");
|
||||
}
|
||||
} else if (c == 314) { // --frag
|
||||
yes_or_no(&opt, MM_F_FRAG_MODE, o.longidx, o.arg, 1);
|
||||
} else if (c == 315) { // --secondary
|
||||
@@ -257,7 +295,16 @@ int main(int argc, char *argv[])
|
||||
} else if (c == 326) { // --dual
|
||||
yes_or_no(&opt, MM_F_NO_DUAL, o.longidx, o.arg, 0);
|
||||
} else if (c == 347) { // --rmq
|
||||
yes_or_no(&opt, MM_F_RMQ, o.longidx, o.arg, 1);
|
||||
if (o.arg) yes_or_no(&opt, MM_F_RMQ, o.longidx, o.arg, 1);
|
||||
else opt.flag |= MM_F_RMQ;
|
||||
} else if (c == 359) { // --pairing
|
||||
if (strcmp(o.arg, "no") == 0) opt.flag |= MM_F_INDEPEND_SEG;
|
||||
else if (strcmp(o.arg, "weak") == 0) opt.flag |= MM_F_WEAK_PAIRING, opt.flag &= ~(uint64_t)MM_F_INDEPEND_SEG;
|
||||
else {
|
||||
if (strcmp(o.arg, "strong") != 0 && mm_verbose >= 2)
|
||||
fprintf(stderr, "[WARNING]\033[1;31m unrecognized argument for --pairing; assuming 'strong'.\033[0m\n");
|
||||
opt.flag &= ~(uint64_t)(MM_F_INDEPEND_SEG|MM_F_WEAK_PAIRING);
|
||||
}
|
||||
} else if (c == 'S') {
|
||||
opt.flag |= MM_F_OUT_CS | MM_F_CIGAR | MM_F_OUT_CS_LONG;
|
||||
if (mm_verbose >= 2)
|
||||
@@ -298,10 +345,6 @@ int main(int argc, char *argv[])
|
||||
if (*s == ',') opt.e2 = strtol(s + 1, &s, 10);
|
||||
}
|
||||
}
|
||||
if ((opt.flag & MM_F_SPLICE) && (opt.flag & MM_F_FRAG_MODE)) {
|
||||
fprintf(stderr, "[ERROR]\033[1;31m --splice and --frag should not be specified at the same time.\033[0m\n");
|
||||
return 1;
|
||||
}
|
||||
if (!fnw && !(opt.flag&MM_F_CIGAR))
|
||||
ipt.flag |= MM_I_NO_SEQ;
|
||||
if (mm_check_opt(&ipt, &opt) < 0)
|
||||
@@ -318,7 +361,7 @@ int main(int argc, char *argv[])
|
||||
fprintf(fp_help, " -H use homopolymer-compressed k-mer (preferrable for PacBio)\n");
|
||||
fprintf(fp_help, " -k INT k-mer size (no larger than 28) [%d]\n", ipt.k);
|
||||
fprintf(fp_help, " -w INT minimizer window size [%d]\n", ipt.w);
|
||||
fprintf(fp_help, " -I NUM split index for every ~NUM input bases [4G]\n");
|
||||
fprintf(fp_help, " -I NUM split index for every ~NUM input bases [8G]\n");
|
||||
fprintf(fp_help, " -d FILE dump index to FILE []\n");
|
||||
fprintf(fp_help, " Mapping:\n");
|
||||
fprintf(fp_help, " -f FLOAT filter out top FLOAT fraction of repetitive minimizers [%g]\n", opt.mid_occ_frac);
|
||||
@@ -340,6 +383,8 @@ int main(int argc, char *argv[])
|
||||
fprintf(fp_help, " -z INT[,INT] Z-drop score and inversion Z-drop score [%d,%d]\n", opt.zdrop, opt.zdrop_inv);
|
||||
fprintf(fp_help, " -s INT minimal peak DP alignment score [%d]\n", opt.min_dp_max);
|
||||
fprintf(fp_help, " -u CHAR how to find GT-AG. f:transcript strand, b:both strands, n:don't match GT-AG [n]\n");
|
||||
fprintf(fp_help, " -J INT splice mode. 0: original minimap2 model; 1: miniprot model [1]\n");
|
||||
fprintf(fp_help, " -j FILE junctions in BED12 to extend *short* RNA-seq alignment []\n");
|
||||
fprintf(fp_help, " Input/Output:\n");
|
||||
fprintf(fp_help, " -a output in the SAM format (PAF by default)\n");
|
||||
fprintf(fp_help, " -o FILE output alignments to FILE [stdout]\n");
|
||||
@@ -347,21 +392,24 @@ int main(int argc, char *argv[])
|
||||
fprintf(fp_help, " -R STR SAM read group line in a format like '@RG\\tID:foo\\tSM:bar' []\n");
|
||||
fprintf(fp_help, " -c output CIGAR in PAF\n");
|
||||
fprintf(fp_help, " --cs[=STR] output the cs tag; STR is 'short' (if absent) or 'long' [none]\n");
|
||||
fprintf(fp_help, " --ds output the ds tag, which is an extension to cs\n");
|
||||
fprintf(fp_help, " --MD output the MD tag\n");
|
||||
fprintf(fp_help, " --eqx write =/X CIGAR operators\n");
|
||||
fprintf(fp_help, " -Y use soft clipping for supplementary alignments\n");
|
||||
fprintf(fp_help, " -y copy FASTA/Q comments to output SAM\n");
|
||||
fprintf(fp_help, " -t INT number of threads [%d]\n", n_threads);
|
||||
fprintf(fp_help, " -K NUM minibatch size for mapping [500M]\n");
|
||||
// fprintf(fp_help, " -v INT verbose level [%d]\n", mm_verbose);
|
||||
fprintf(fp_help, " --version show version number\n");
|
||||
fprintf(fp_help, " Preset:\n");
|
||||
fprintf(fp_help, " -x STR preset (always applied before other options; see minimap2.1 for details) []\n");
|
||||
fprintf(fp_help, " - map-pb/map-ont - PacBio CLR/Nanopore vs reference mapping\n");
|
||||
fprintf(fp_help, " - map-hifi - PacBio HiFi reads vs reference mapping\n");
|
||||
fprintf(fp_help, " - ava-pb/ava-ont - PacBio/Nanopore read overlap\n");
|
||||
fprintf(fp_help, " - lr:hq - accurate long reads (error rate <1%%) against a reference genome\n");
|
||||
fprintf(fp_help, " - splice/splice:hq - spliced alignment for long reads/accurate long reads\n");
|
||||
fprintf(fp_help, " - splice:sr - spliced alignment for short RNA-seq reads\n");
|
||||
fprintf(fp_help, " - asm5/asm10/asm20 - asm-to-ref mapping, for ~0.1/1/5%% sequence divergence\n");
|
||||
fprintf(fp_help, " - splice/splice:hq - long-read/Pacbio-CCS spliced alignment\n");
|
||||
fprintf(fp_help, " - sr - genomic short-read mapping\n");
|
||||
fprintf(fp_help, " - sr - short reads against a reference\n");
|
||||
fprintf(fp_help, " - map-pb/map-hifi/map-ont/map-iclr - CLR/HiFi/Nanopore/ICLR vs reference mapping\n");
|
||||
fprintf(fp_help, " - ava-pb/ava-ont - PacBio CLR/Nanopore read overlap\n");
|
||||
fprintf(fp_help, "\nSee `man ./minimap2.1' for detailed description of these and other advanced command-line options.\n");
|
||||
return fp_help == stdout? 0 : 1;
|
||||
}
|
||||
@@ -412,7 +460,26 @@ int main(int argc, char *argv[])
|
||||
__func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), mi->n_seq);
|
||||
if (argc != o.ind + 1) mm_mapopt_update(&opt, mi);
|
||||
if (mm_verbose >= 3) mm_idx_stat(mi);
|
||||
if (junc_bed) mm_idx_bed_read(mi, junc_bed, 1);
|
||||
if (fn_bed_junc) {
|
||||
mm_idx_bed_read(mi, fn_bed_junc, 1);
|
||||
if (mi->I == 0 && mm_verbose >= 2)
|
||||
fprintf(stderr, "[WARNING] failed to load the junction BED file\n");
|
||||
}
|
||||
if (fn_bed_jump) {
|
||||
mm_idx_jjump_read(mi, fn_bed_jump, MM_JUNC_ANNO, -1);
|
||||
if (mi->J == 0 && mm_verbose >= 2)
|
||||
fprintf(stderr, "[WARNING] failed to load the jump BED file\n");
|
||||
}
|
||||
if (fn_bed_pass1) {
|
||||
mm_idx_jjump_read(mi, fn_bed_pass1, MM_JUNC_MISC, 5);
|
||||
if (mi->J == 0 && mm_verbose >= 2)
|
||||
fprintf(stderr, "[WARNING] failed to load the pass-1 jump BED file\n");
|
||||
}
|
||||
if (fn_spsc) {
|
||||
mm_idx_spsc_read(mi, fn_spsc, mm_max_spsc_bonus(&opt));
|
||||
if (mi->spsc == 0 && mm_verbose >= 2)
|
||||
fprintf(stderr, "[WARNING] failed to load the splice score file\n");
|
||||
}
|
||||
if (alt_list) mm_idx_alt_read(mi, alt_list);
|
||||
if (argc - (o.ind + 1) == 0) {
|
||||
mm_idx_destroy(mi);
|
||||
|
||||
@@ -10,11 +10,6 @@
|
||||
#include "bseq.h"
|
||||
#include "khash.h"
|
||||
|
||||
struct mm_tbuf_s {
|
||||
void *km;
|
||||
int rep_len, frag_gap;
|
||||
};
|
||||
|
||||
mm_tbuf_t *mm_tbuf_init(void)
|
||||
{
|
||||
mm_tbuf_t *b;
|
||||
@@ -229,10 +224,10 @@ static mm_reg1_t *align_regs(const mm_mapopt_t *opt, const mm_idx_t *mi, void *k
|
||||
return regs;
|
||||
}
|
||||
|
||||
void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **seqs, int *n_regs, mm_reg1_t **regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
||||
void mm_map_frag_core(const mm_idx_t *mi, int n_segs, const int *qlens, const char **seqs, int *n_regs, mm_reg1_t **regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
||||
{
|
||||
int i, j, rep_len, qlen_sum, n_regs0, n_mini_pos;
|
||||
int max_chain_gap_qry, max_chain_gap_ref, is_splice = !!(opt->flag & MM_F_SPLICE), is_sr = !!(opt->flag & MM_F_SR);
|
||||
int max_chain_gap_qry, max_chain_gap_ref, is_splice = !!(opt->flag & MM_F_SPLICE), is_sr = !!(opt->flag & MM_F_SR), is_sr_rna = !!(opt->flag & MM_F_SR_RNA);
|
||||
uint32_t hash;
|
||||
int64_t n_a;
|
||||
uint64_t *u, *mini_pos;
|
||||
@@ -248,7 +243,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
if (qlen_sum == 0 || n_segs <= 0 || n_segs > MM_MAX_SEG) return;
|
||||
if (opt->max_qlen > 0 && qlen_sum > opt->max_qlen) return;
|
||||
|
||||
hash = qname? __ac_X31_hash_string(qname) : 0;
|
||||
hash = qname && !(opt->flag & MM_F_NO_HASH_NAME)? __ac_X31_hash_string(qname) : 0;
|
||||
hash ^= __ac_Wang_hash(qlen_sum) + __ac_Wang_hash(opt->seed);
|
||||
hash = __ac_Wang_hash(hash);
|
||||
|
||||
@@ -328,7 +323,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
mm_hit_sort(b->km, &n_regs0, regs0, opt->alt_drop); // this step can be merged into mm_gen_regs(); will do if this shows up in profile
|
||||
}
|
||||
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_SEED)
|
||||
if (mm_dbg_flag & (MM_DBG_PRINT_SEED|MM_DBG_PRINT_CHAIN))
|
||||
for (j = 0; j < n_regs0; ++j)
|
||||
for (i = regs0[j].as; i < regs0[j].as + regs0[j].cnt; ++i)
|
||||
fprintf(stderr, "CN\t%d\t%s\t%d\t%c\t%d\t%d\t%d\n", j, mi->seq[a[i].x<<1>>33].name, (int32_t)a[i].x, "+-"[a[i].x>>63], (int32_t)a[i].y, (int32_t)(a[i].y>>32&0xff),
|
||||
@@ -343,7 +338,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
if (n_segs == 1) { // uni-segment
|
||||
regs0 = align_regs(opt, mi, b->km, qlens[0], seqs[0], &n_regs0, regs0, a);
|
||||
regs0 = (mm_reg1_t*)realloc(regs0, sizeof(*regs0) * n_regs0);
|
||||
mm_set_mapq(b->km, n_regs0, regs0, opt->min_chain_score, opt->a, rep_len, is_sr);
|
||||
mm_set_mapq2(b->km, n_regs0, regs0, opt->min_chain_score, opt->a, rep_len, is_sr || is_sr_rna, is_splice);
|
||||
n_regs[0] = n_regs0, regs[0] = regs0;
|
||||
} else { // multi-segment
|
||||
mm_seg_t *seg;
|
||||
@@ -352,7 +347,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
for (i = 0; i < n_segs; ++i) {
|
||||
mm_set_parent(b->km, opt->mask_level, opt->mask_len, n_regs[i], regs[i], opt->a * 2 + opt->b, opt->flag&MM_F_HARD_MLEVEL, opt->alt_drop); // update mm_reg1_t::parent
|
||||
regs[i] = align_regs(opt, mi, b->km, qlens[i], seqs[i], &n_regs[i], regs[i], seg[i].a);
|
||||
mm_set_mapq(b->km, n_regs[i], regs[i], opt->min_chain_score, opt->a, rep_len, is_sr);
|
||||
mm_set_mapq2(b->km, n_regs[i], regs[i], opt->min_chain_score, opt->a, rep_len, is_sr || is_sr_rna, is_splice);
|
||||
}
|
||||
mm_seg_free(b->km, n_segs, seg);
|
||||
if (n_segs == 2 && opt->pe_ori >= 0 && (opt->flag&MM_F_CIGAR))
|
||||
@@ -364,6 +359,10 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
kfree(b->km, u);
|
||||
kfree(b->km, mini_pos);
|
||||
|
||||
if (mi->J && n_segs == 1 && is_splice)
|
||||
for (i = 0; i < n_regs0; ++i)
|
||||
mm_jump_split(b->km, mi, opt, qlens[0], (const uint8_t*)seqs[0], ®s0[i], 0);
|
||||
|
||||
if (b->km) {
|
||||
km_stat(b->km, &kmst);
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||
@@ -378,6 +377,18 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
}
|
||||
}
|
||||
|
||||
void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **seqs, int *n_regs, mm_reg1_t **regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
||||
{
|
||||
if ((opt->flag & MM_F_WEAK_PAIRING) && n_segs == 2 && opt->pe_ori >= 0 && (opt->flag&MM_F_CIGAR)) {
|
||||
int i;
|
||||
for (i = 0; i < n_segs; ++i)
|
||||
mm_map_frag_core(mi, 1, &qlens[i], &seqs[i], &n_regs[i], ®s[i], b, opt, qname);
|
||||
mm_pair(b->km, opt->max_gap_ref, opt->pe_bonus, opt->a * 2 + opt->b, opt->a, qlens, n_regs, regs);
|
||||
} else {
|
||||
mm_map_frag_core(mi, n_segs, qlens, seqs, n_regs, regs, b, opt, qname);
|
||||
}
|
||||
}
|
||||
|
||||
mm_reg1_t *mm_map(const mm_idx_t *mi, int qlen, const char *seq, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt, const char *qname)
|
||||
{
|
||||
mm_reg1_t *regs;
|
||||
@@ -452,6 +463,10 @@ static void worker_for(void *_data, long i, int tid) // kt_for() callback
|
||||
r->qs = qlens[j] - r->qe;
|
||||
r->qe = qlens[j] - t;
|
||||
r->rev = !r->rev;
|
||||
if (r->p) {
|
||||
if (r->p->trans_strand == 1) r->p->trans_strand = 2;
|
||||
else if (r->p->trans_strand == 2) r->p->trans_strand = 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_QNAME)
|
||||
@@ -514,7 +529,7 @@ static void merge_hits(step_t *s)
|
||||
mm_select_sub(km, opt->pri_ratio, s->p->mi->k*2, opt->best_n, 0, opt->max_gap * 0.8, &s->n_reg[k], s->reg[k]);
|
||||
mm_set_sam_pri(s->n_reg[k], s->reg[k]);
|
||||
}
|
||||
mm_set_mapq(km, s->n_reg[k], s->reg[k], opt->min_chain_score, opt->a, rep_len, !!(opt->flag & MM_F_SR));
|
||||
mm_set_mapq2(km, s->n_reg[k], s->reg[k], opt->min_chain_score, opt->a, rep_len, !!(opt->flag & (MM_F_SR|MM_F_SR_RNA)), !!(opt->flag & MM_F_SPLICE));
|
||||
}
|
||||
if (s->n_seg[f] == 2 && opt->pe_ori >= 0 && (opt->flag&MM_F_CIGAR))
|
||||
mm_pair(km, frag_gap_part[0], opt->pe_bonus, opt->a * 2 + opt->b, opt->a, qlens, &s->n_reg[k0], &s->reg[k0]);
|
||||
@@ -583,23 +598,30 @@ static void *worker_pipeline(void *shared, int step, void *in)
|
||||
mm_err_fwrite(r->p, r->p->capacity, 4, p->fp_split);
|
||||
}
|
||||
}
|
||||
} else if (p->opt->flag & MM_F_OUT_JUNC) { // extra logic for --write-junc
|
||||
for (j = 0; j < s->n_reg[i]; ++j) {
|
||||
const mm_reg1_t *r = &s->reg[i][j];
|
||||
if (r->id != r->parent || r->mapq < 10) continue;
|
||||
mm_write_junc(&p->str, mi, t, r);
|
||||
if (p->str.l > 0) mm_err_puts(p->str.s);
|
||||
}
|
||||
} else if (s->n_reg[i] > 0) { // the query has at least one hit
|
||||
for (j = 0; j < s->n_reg[i]; ++j) {
|
||||
mm_reg1_t *r = &s->reg[i][j];
|
||||
const mm_reg1_t *r = &s->reg[i][j];
|
||||
assert(!r->sam_pri || r->id == r->parent);
|
||||
if ((p->opt->flag & MM_F_NO_PRINT_2ND) && r->id != r->parent)
|
||||
continue;
|
||||
if (p->opt->flag & MM_F_OUT_SAM)
|
||||
mm_write_sam3(&p->str, mi, t, i - seg_st, j, s->n_seg[k], &s->n_reg[seg_st], (const mm_reg1_t*const*)&s->reg[seg_st], km, p->opt->flag, s->rep_len[i]);
|
||||
else
|
||||
mm_write_paf3(&p->str, mi, t, r, km, p->opt->flag, s->rep_len[i]);
|
||||
mm_write_paf4(&p->str, mi, t, r, km, p->opt->flag, s->rep_len[i], s->n_seg[k], i - seg_st);
|
||||
mm_err_puts(p->str.s);
|
||||
}
|
||||
} else if ((p->opt->flag & MM_F_PAF_NO_HIT) || ((p->opt->flag & MM_F_OUT_SAM) && !(p->opt->flag & MM_F_SAM_HIT_ONLY))) { // output an empty hit, if requested
|
||||
if (p->opt->flag & MM_F_OUT_SAM)
|
||||
mm_write_sam3(&p->str, mi, t, i - seg_st, -1, s->n_seg[k], &s->n_reg[seg_st], (const mm_reg1_t*const*)&s->reg[seg_st], km, p->opt->flag, s->rep_len[i]);
|
||||
else
|
||||
mm_write_paf3(&p->str, mi, t, 0, 0, p->opt->flag, s->rep_len[i]);
|
||||
mm_write_paf4(&p->str, mi, t, 0, 0, p->opt->flag, s->rep_len[i], s->n_seg[k], i - seg_st);
|
||||
mm_err_puts(p->str.s);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,40 +5,49 @@
|
||||
#include <stdio.h>
|
||||
#include <sys/types.h>
|
||||
|
||||
#define MM_F_NO_DIAG 0x001 // no exact diagonal hit
|
||||
#define MM_F_NO_DUAL 0x002 // skip pairs where query name is lexicographically larger than target name
|
||||
#define MM_F_CIGAR 0x004
|
||||
#define MM_F_OUT_SAM 0x008
|
||||
#define MM_F_NO_QUAL 0x010
|
||||
#define MM_F_OUT_CG 0x020
|
||||
#define MM_F_OUT_CS 0x040
|
||||
#define MM_F_SPLICE 0x080 // splice mode
|
||||
#define MM_F_SPLICE_FOR 0x100 // match GT-AG
|
||||
#define MM_F_SPLICE_REV 0x200 // match CT-AC, the reverse complement of GT-AG
|
||||
#define MM_F_NO_LJOIN 0x400
|
||||
#define MM_F_OUT_CS_LONG 0x800
|
||||
#define MM_F_SR 0x1000
|
||||
#define MM_F_FRAG_MODE 0x2000
|
||||
#define MM_F_NO_PRINT_2ND 0x4000
|
||||
#define MM_F_2_IO_THREADS 0x8000
|
||||
#define MM_F_LONG_CIGAR 0x10000
|
||||
#define MM_F_INDEPEND_SEG 0x20000
|
||||
#define MM_F_SPLICE_FLANK 0x40000
|
||||
#define MM_F_SOFTCLIP 0x80000
|
||||
#define MM_F_FOR_ONLY 0x100000
|
||||
#define MM_F_REV_ONLY 0x200000
|
||||
#define MM_F_HEAP_SORT 0x400000
|
||||
#define MM_F_ALL_CHAINS 0x800000
|
||||
#define MM_F_OUT_MD 0x1000000
|
||||
#define MM_F_COPY_COMMENT 0x2000000
|
||||
#define MM_F_EQX 0x4000000 // use =/X instead of M
|
||||
#define MM_F_PAF_NO_HIT 0x8000000 // output unmapped reads to PAF
|
||||
#define MM_F_NO_END_FLT 0x10000000
|
||||
#define MM_F_HARD_MLEVEL 0x20000000
|
||||
#define MM_F_SAM_HIT_ONLY 0x40000000
|
||||
#define MM_VERSION "2.29-r1283"
|
||||
|
||||
#define MM_F_NO_DIAG (0x001LL) // no exact diagonal hit
|
||||
#define MM_F_NO_DUAL (0x002LL) // skip pairs where query name is lexicographically larger than target name
|
||||
#define MM_F_CIGAR (0x004LL)
|
||||
#define MM_F_OUT_SAM (0x008LL)
|
||||
#define MM_F_NO_QUAL (0x010LL)
|
||||
#define MM_F_OUT_CG (0x020LL)
|
||||
#define MM_F_OUT_CS (0x040LL)
|
||||
#define MM_F_SPLICE (0x080LL) // splice mode
|
||||
#define MM_F_SPLICE_FOR (0x100LL) // match GT-AG
|
||||
#define MM_F_SPLICE_REV (0x200LL) // match CT-AC, the reverse complement of GT-AG
|
||||
#define MM_F_NO_LJOIN (0x400LL)
|
||||
#define MM_F_OUT_CS_LONG (0x800LL)
|
||||
#define MM_F_SR (0x1000LL)
|
||||
#define MM_F_FRAG_MODE (0x2000LL)
|
||||
#define MM_F_NO_PRINT_2ND (0x4000LL)
|
||||
#define MM_F_2_IO_THREADS (0x8000LL)
|
||||
#define MM_F_LONG_CIGAR (0x10000LL)
|
||||
#define MM_F_INDEPEND_SEG (0x20000LL)
|
||||
#define MM_F_SPLICE_FLANK (0x40000LL)
|
||||
#define MM_F_SOFTCLIP (0x80000LL)
|
||||
#define MM_F_FOR_ONLY (0x100000LL)
|
||||
#define MM_F_REV_ONLY (0x200000LL)
|
||||
#define MM_F_HEAP_SORT (0x400000LL)
|
||||
#define MM_F_ALL_CHAINS (0x800000LL)
|
||||
#define MM_F_OUT_MD (0x1000000LL)
|
||||
#define MM_F_COPY_COMMENT (0x2000000LL)
|
||||
#define MM_F_EQX (0x4000000LL) // use =/X instead of M
|
||||
#define MM_F_PAF_NO_HIT (0x8000000LL) // output unmapped reads to PAF
|
||||
#define MM_F_NO_END_FLT (0x10000000LL)
|
||||
#define MM_F_HARD_MLEVEL (0x20000000LL)
|
||||
#define MM_F_SAM_HIT_ONLY (0x40000000LL)
|
||||
#define MM_F_RMQ (0x80000000LL)
|
||||
#define MM_F_QSTRAND (0x100000000LL)
|
||||
#define MM_F_NO_INV (0x200000000LL)
|
||||
#define MM_F_NO_HASH_NAME (0x400000000LL)
|
||||
#define MM_F_SPLICE_OLD (0x800000000LL)
|
||||
#define MM_F_SECONDARY_SEQ (0x1000000000LL) //output SEQ field for seqondary alignments using hard clipping
|
||||
#define MM_F_OUT_DS (0x2000000000LL)
|
||||
#define MM_F_WEAK_PAIRING (0x4000000000LL)
|
||||
#define MM_F_SR_RNA (0x8000000000LL)
|
||||
#define MM_F_OUT_JUNC (0x10000000000LL)
|
||||
|
||||
#define MM_I_HPC 0x1
|
||||
#define MM_I_NO_SEQ 0x2
|
||||
@@ -85,6 +94,8 @@ typedef struct {
|
||||
uint32_t *S; // 4-bit packed sequence
|
||||
struct mm_idx_bucket_s *B; // index (hidden)
|
||||
struct mm_idx_intv_s *I; // intervals (hidden)
|
||||
struct mm_idx_spsc_s *spsc;// splice score (hidden)
|
||||
struct mm_idx_jjump_s *J; // junctions to create jumps (hidden)
|
||||
void *km, *h;
|
||||
} mm_idx_t;
|
||||
|
||||
@@ -92,6 +103,7 @@ typedef struct {
|
||||
typedef struct {
|
||||
uint32_t capacity; // the capacity of cigar[]
|
||||
int32_t dp_score, dp_max, dp_max2; // DP score; score of the max-scoring segment; score of the best alternate mappings
|
||||
int32_t dp_max0; // DP score before mm_update_dp_max() adjustment
|
||||
uint32_t n_ambi:30, trans_strand:2; // number of ambiguous bases; transcript strand: 0 for unknown, 1 for +, 2 for -
|
||||
uint32_t n_cigar; // number of cigar operations in cigar[]
|
||||
uint32_t cigar[];
|
||||
@@ -108,7 +120,7 @@ typedef struct {
|
||||
int32_t mlen, blen; // seeded exact match length; seeded alignment block length
|
||||
int32_t n_sub; // number of suboptimal mappings
|
||||
int32_t score0; // initial chaining score (before chain merging/spliting)
|
||||
uint32_t mapq:8, split:2, rev:1, inv:1, sam_pri:1, proper_frag:1, pe_thru:1, seg_split:1, seg_id:8, split_inv:1, is_alt:1, strand_retained:1, dummy:5;
|
||||
uint32_t mapq:8, split:2, rev:1, inv:1, sam_pri:1, proper_frag:1, pe_thru:1, seg_split:1, seg_id:8, split_inv:1, is_alt:1, strand_retained:1, is_spliced:1, dummy:4;
|
||||
uint32_t hash;
|
||||
float div;
|
||||
mm_extra_t *p;
|
||||
@@ -148,9 +160,10 @@ typedef struct {
|
||||
float alt_drop;
|
||||
|
||||
int a, b, q, e, q2, e2; // matching score, mismatch, gap-open and gap-ext penalties
|
||||
int transition; // transition mismatch score (A:G, C:T)
|
||||
int sc_ambi; // score when one or both bases are "N"
|
||||
int noncan; // cost of non-canonical splicing sites
|
||||
int junc_bonus;
|
||||
int junc_bonus, junc_pen;
|
||||
int zdrop, zdrop_inv; // break alignment if alignment score drops too fast along the diagonal
|
||||
int end_bonus;
|
||||
int min_dp_max; // drop an alignment if the score of the max scoring segment is below this threshold
|
||||
@@ -163,6 +176,8 @@ typedef struct {
|
||||
|
||||
int pe_ori, pe_bonus;
|
||||
|
||||
int32_t jump_min_match;
|
||||
|
||||
float mid_occ_frac; // only used by mm_mapopt_update(); see below
|
||||
float q_occ_frac;
|
||||
int32_t min_mid_occ, max_mid_occ;
|
||||
@@ -188,6 +203,11 @@ typedef struct {
|
||||
} mm_idx_reader_t;
|
||||
|
||||
// memory buffer for thread-local storage during mapping
|
||||
struct mm_tbuf_s {
|
||||
void *km;
|
||||
int rep_len, frag_gap;
|
||||
};
|
||||
|
||||
typedef struct mm_tbuf_s mm_tbuf_t;
|
||||
|
||||
// global variables
|
||||
@@ -398,6 +418,10 @@ int mm_idx_alt_read(mm_idx_t *mi, const char *fn);
|
||||
int mm_idx_bed_read(mm_idx_t *mi, const char *fn, int read_junc);
|
||||
int mm_idx_bed_junc(const mm_idx_t *mi, int32_t ctg, int32_t st, int32_t en, uint8_t *s);
|
||||
|
||||
int mm_max_spsc_bonus(const mm_mapopt_t *mo);
|
||||
int32_t mm_idx_spsc_read(mm_idx_t *idx, const char *fn, int32_t max_sc);
|
||||
int64_t mm_idx_spsc_get(const mm_idx_t *db, int32_t cid, int64_t st0, int64_t en0, int32_t rev, uint8_t *sc);
|
||||
|
||||
// deprecated APIs for backward compatibility
|
||||
void mm_mapopt_init(mm_mapopt_t *opt);
|
||||
mm_idx_t *mm_idx_build(const char *fn, int w, int k, int flag, int n_threads);
|
||||
|
||||
+161
-48
@@ -1,4 +1,4 @@
|
||||
.TH minimap2 1 "18 November 2021" "minimap2-2.23 (r1111)" "Bioinformatics tools"
|
||||
.TH minimap2 1 "18 April 2025" "minimap2-2.29 (r1283)" "Bioinformatics tools"
|
||||
.SH NAME
|
||||
.PP
|
||||
minimap2 - mapping and alignment between collections of DNA sequences
|
||||
@@ -77,7 +77,7 @@ SAM format.
|
||||
Minimizer k-mer length [15]
|
||||
.TP
|
||||
.BI -w \ INT
|
||||
Minimizer window size [2/3 of k-mer length]. A minimizer is the smallest k-mer
|
||||
Minimizer window size [10]. A minimizer is the smallest k-mer
|
||||
in a window of w consecutive k-mers.
|
||||
.TP
|
||||
.B -H
|
||||
@@ -88,16 +88,17 @@ on the HPC sequence.
|
||||
.BI -I \ NUM
|
||||
Load at most
|
||||
.I NUM
|
||||
target bases into RAM for indexing [4G]. If there are more than
|
||||
target bases into RAM for indexing [8G]. If there are more than
|
||||
.I NUM
|
||||
bases in
|
||||
.IR target.fa ,
|
||||
minimap2 needs to read
|
||||
.I query.fa
|
||||
multiple times to map it against each batch of target sequences.
|
||||
multiple times to map it against each batch of target sequences. This would create a multi-part index.
|
||||
.I NUM
|
||||
may be ending with k/K/m/M/g/G. NB: mapping quality is incorrect given a
|
||||
multi-part index.
|
||||
multi-part index. See also option
|
||||
.BR --split-prefix .
|
||||
.TP
|
||||
.B --idx-no-seq
|
||||
Don't store target sequences in the index. It saves disk space and memory but
|
||||
@@ -254,6 +255,11 @@ or more of the shorter chain [0.5]
|
||||
Use the minigraph chaining algorithm [no]. The minigraph algorithm is better
|
||||
for aligning contigs through long INDELs.
|
||||
.TP
|
||||
.BI --rmq-inner \ NUM
|
||||
Apply full dynamic programming for anchors within distance
|
||||
.I NUM
|
||||
[1000].
|
||||
.TP
|
||||
.B --hard-mask-level
|
||||
Honor option
|
||||
.B -M
|
||||
@@ -291,11 +297,13 @@ maximum alignment gap is mostly controlled by
|
||||
.B --splice
|
||||
Enable the splice alignment mode.
|
||||
.TP
|
||||
.B --sr
|
||||
Enable short-read alignment heuristics. In the short-read mode, minimap2
|
||||
applies a second round of chaining with a higher minimizer occurrence threshold
|
||||
if no good chain is found. In addition, minimap2 attempts to patch gaps between
|
||||
seeds with ungapped alignment.
|
||||
.BR --sr [= no | dna | rna ]
|
||||
Enable short-read alignment heuristics [no]. If this option is used with no argument,
|
||||
.RB ` dna '
|
||||
is set. In the DNA short-read mode, minimap2 applies a second round of chaining
|
||||
with a higher minimizer occurrence threshold if no good chain is found. In
|
||||
addition, minimap2 attempts to patch gaps between seeds with ungapped
|
||||
alignment.
|
||||
.TP
|
||||
.BI --split-prefix \ STR
|
||||
Prefix to create temporary files. Typically used for a multi-part index.
|
||||
@@ -315,9 +323,8 @@ Only map to the reverse complement strand of the reference sequences.
|
||||
If yes, sort anchors with heap merge, instead of radix sort. Heap merge is
|
||||
faster for short reads, but slower for long reads. [no]
|
||||
.TP
|
||||
.B --no-pairing
|
||||
Treat two reads in a pair as independent reads. The mate related fields in SAM
|
||||
are still properly populated.
|
||||
.B --no-hash-name
|
||||
Produce the same alignment for identical sequences regardless of their sequence names.
|
||||
.SS Alignment options
|
||||
.TP 10
|
||||
.BI -A \ INT
|
||||
@@ -326,6 +333,10 @@ Matching score [2]
|
||||
.BI -B \ INT
|
||||
Mismatching penalty [4]
|
||||
.TP
|
||||
.BI -b \ INT
|
||||
Mismatching penalty for transitions [same as
|
||||
.BR -B ].
|
||||
.TP
|
||||
.BI -O \ INT1[,INT2]
|
||||
Gap open penalty [4,24]. If
|
||||
.I INT2
|
||||
@@ -339,10 +350,28 @@ costs
|
||||
.RI min{ O1 + k * E1 , O2 + k * E2 }.
|
||||
In the splice mode, the second gap penalties are not used.
|
||||
.TP
|
||||
.BI -J \ INT
|
||||
Splice model [1]. 0 for the original minimap2 splice model that always penalizes non-GT-AG splicing;
|
||||
1 for the miniprot model that considers non-GT-AG. Option
|
||||
.B -C
|
||||
has no effect with the default
|
||||
.BR -J1 .
|
||||
.TP
|
||||
.BR -j \ FILE
|
||||
Junctions used to extend alignment towards ends of reads [].
|
||||
.I FILE
|
||||
can be gene annotations in the BED12 format (aka 12-column BED), or intron
|
||||
positions in 5-column BED with the strand column required. BED12 file can be
|
||||
converted from GTF/GFF3 with `paftools.js gff2bed anno.gtf'. This option is
|
||||
intended for short RNA-seq reads, while
|
||||
.B --junc-bed
|
||||
for long noisy RNA-seq reads.
|
||||
.TP
|
||||
.BI -C \ INT
|
||||
Cost for a non-canonical GT-AG splicing (effective with
|
||||
.BR --splice )
|
||||
[0]
|
||||
.B --splice
|
||||
.BR -J0 )
|
||||
[0].
|
||||
.TP
|
||||
.BI -z \ INT1[,INT2]
|
||||
Truncate an alignment if the running alignment score drops too quickly along
|
||||
@@ -379,7 +408,16 @@ no attempt to match GT-AG [n]
|
||||
Score bonus when alignment extends to the end of the query sequence [0].
|
||||
.TP
|
||||
.BI --score-N \ INT
|
||||
Score of a mismatch involving ambiguous bases [1].
|
||||
Penalty of a mismatch involving ambiguous bases [1].
|
||||
.TP
|
||||
.BR --pairing = strong | weak | no
|
||||
How to pair paired-end reads [strong].
|
||||
.RB ` no '
|
||||
for aligning the two ends in a pair independently with no `properly paired' set.
|
||||
.RB ` weak '
|
||||
for aligning the two ends independently and then pairing the hits.
|
||||
.RB ` strong '
|
||||
for jointly aligning and pairing the two ends.
|
||||
.TP
|
||||
.BR --splice-flank = yes | no
|
||||
Assume the next base to a
|
||||
@@ -398,16 +436,40 @@ on SIRV data, please add
|
||||
.B --splice-flank=no
|
||||
to the command line.
|
||||
.TP
|
||||
.BR --spsc \ FILE
|
||||
Splice scores []. Each line consists of five fields: 1) contig, 2) offset, 3) `+' or `-', 4) `D' or `A', and 5) score,
|
||||
where offset is the number of bases before a splice junction, `D' indicates the
|
||||
line corresponds to a donor site and `A' for an acceptor site.
|
||||
A positive score suggests the junction is preferred and a negative score
|
||||
suggests the junction is not preferred.
|
||||
.TP
|
||||
.BR --junc-pen \ INT
|
||||
Penalty for a position not in FILE specified by
|
||||
.B --spsc
|
||||
[5]. Effective with
|
||||
.B --spsc
|
||||
but not
|
||||
.BR --junc-bed .
|
||||
.TP
|
||||
.BR --junc-bed \ FILE
|
||||
Gene annotations in the BED12 format (aka 12-column BED), or intron positions
|
||||
in 5-column BED. With this option, minimap2 prefers splicing in annotations.
|
||||
BED12 file can be converted from GTF/GFF3 with `paftools.js gff2bed anno.gtf'
|
||||
[].
|
||||
Junctions to prefer during base alignment [].
|
||||
Same format as
|
||||
.BR -j .
|
||||
It is
|
||||
.I NOT
|
||||
recommended to apply this option to short RNA-seq reads. This would increase
|
||||
run time with little improvement to junction accuracy.
|
||||
.TP
|
||||
.BR --junc-bonus \ INT
|
||||
Score bonus for a splice donor or acceptor found in annotation (effective with
|
||||
.BR --junc-bed )
|
||||
[9].
|
||||
Score bonus for a splice donor or acceptor found in annotation [9]. Effective with
|
||||
.B --junc-bed
|
||||
but not
|
||||
.BR --spsc .
|
||||
.TP
|
||||
.BR --jump-min-match \ INT
|
||||
Minimum matching length to create a jump [3]. Equivalent to
|
||||
.B STAR
|
||||
.BR --alignSJDBoverhangMin .
|
||||
.TP
|
||||
.BI --end-seed-pen \ INT
|
||||
Drop a terminal anchor if
|
||||
@@ -433,7 +495,7 @@ Set 0 to disable [100m].
|
||||
.BI --cap-kalloc \ NUM
|
||||
Free thread-local kalloc memory reservoir if after the alignment the size of the reservoir above
|
||||
.IR NUM .
|
||||
Set 0 to disable [0].
|
||||
Set 0 to disable [500m].
|
||||
.SS Input/output options
|
||||
.TP 10
|
||||
.B -a
|
||||
@@ -465,20 +527,13 @@ Copy input FASTA/Q comments to output.
|
||||
.B -c
|
||||
Generate CIGAR. In PAF, the CIGAR is written to the `cg' custom tag.
|
||||
.TP
|
||||
.BI --cs[= STR ]
|
||||
.BR --cs [= short | long ]
|
||||
Output the
|
||||
.B cs
|
||||
tag.
|
||||
.I STR
|
||||
can be either
|
||||
.I short
|
||||
or
|
||||
.IR long .
|
||||
If no
|
||||
.I STR
|
||||
is given,
|
||||
.I short
|
||||
is assumed. [none]
|
||||
If no argument is given,
|
||||
.RB ` short '
|
||||
is set. [none]
|
||||
.TP
|
||||
.B --MD
|
||||
Output the MD tag (see the SAM spec).
|
||||
@@ -489,6 +544,29 @@ Output =/X CIGAR operators for sequence match/mismatch.
|
||||
.B -Y
|
||||
In SAM output, use soft clipping for supplementary alignments.
|
||||
.TP
|
||||
.B --secondary-seq
|
||||
In SAM output, show query sequences for secondary alignments.
|
||||
.TP
|
||||
.B --write-junc
|
||||
Output splice junctions in 6-column BED: contig name, start, end,
|
||||
read name, score and strand. Score is the sum of donor and acceptor scores,
|
||||
where GT gets 3, GC gets 2 and AT gets 1 at donor sites,
|
||||
while AG gets 3 and AC gets 1 at acceptor sites.
|
||||
Alignments with mapping quality below 10 are ignored.
|
||||
.TP
|
||||
.BI --pass1 \ FILE
|
||||
Junctions BED file outputted by
|
||||
.B --write-junc
|
||||
[]. Rows with scores lower than 5 are ignored. When both
|
||||
.B -j
|
||||
and
|
||||
.B --pass1
|
||||
are present, junctions in
|
||||
.B -j
|
||||
are preferred over in
|
||||
.BR --pass1
|
||||
when there is ambiguity.
|
||||
.TP
|
||||
.BI --seed \ INT
|
||||
Integer seed for randomizing equally best hits. Minimap2 hashes
|
||||
.I INT
|
||||
@@ -549,42 +627,70 @@ are:
|
||||
Align noisy long reads of ~10% error rate to a reference genome. This is the
|
||||
default mode.
|
||||
.TP
|
||||
.B lr:hq
|
||||
Align accurate long reads (error rate <1%) to a reference genome
|
||||
.RB ( -k19
|
||||
.B -w19 -U50,500
|
||||
.BR -g10k ).
|
||||
This was recommended by ONT developers for recent Nanopore reads
|
||||
produced with chemistry v14 that can reach ~99% in accuracy.
|
||||
It was shown to work better for accurate Nanopore reads
|
||||
than
|
||||
.BR map-hifi .
|
||||
.TP
|
||||
.B map-hifi
|
||||
Align PacBio high-fidelity (HiFi) reads to a reference genome
|
||||
.RB ( -k19
|
||||
.B -w19 -U50,500 -g10k -A1 -B4 -O6,26 -E2,1
|
||||
.RB ( -xlr:hq
|
||||
.B -A1 -B4 -O6,26 -E2,1
|
||||
.BR -s200 ).
|
||||
It differs from
|
||||
.B lr:hq
|
||||
only in scoring. It has not been tested whether
|
||||
.B lr:hq
|
||||
would work better for PacBio HiFi reads.
|
||||
.TP
|
||||
.B map-pb
|
||||
Align older PacBio continuous long (CLR) reads to a reference genome
|
||||
.RB ( -Hk19 ).
|
||||
Note that this data type is effectively deprecated by HiFi.
|
||||
Unless you work on very old data, you probably want to use
|
||||
.B map-hifi
|
||||
or
|
||||
.BR lr:hq .
|
||||
.TP
|
||||
.B map-iclr
|
||||
Align Illumina Complete Long Reads (ICLR) to a reference genome
|
||||
.RB ( -k19
|
||||
.B -B6 -b4
|
||||
.BR -O10,50 ).
|
||||
This was recommended by Illumina developers.
|
||||
.TP
|
||||
.B asm5
|
||||
Long assembly to reference mapping
|
||||
.RB ( -k19
|
||||
.B -w19 -U50,500 --rmq -r100k -g10k -A1 -B19 -O39,81 -E3,1 -s200 -z200
|
||||
.B -w19 -U50,500 --rmq -r1k,100k -g10k -A1 -B19 -O39,81 -E3,1 -s200 -z200
|
||||
.BR -N50 ).
|
||||
Typically, the alignment will not extend to regions with 5% or higher sequence
|
||||
divergence. Only use this preset if the average divergence is far below 5%.
|
||||
divergence. Use this preset if the average divergence is not much higher than 0.1%.
|
||||
.TP
|
||||
.B asm10
|
||||
Long assembly to reference mapping
|
||||
.RB ( -k19
|
||||
.B -w19 -U50,500 --rmq -r100k -g10k -A1 -B9 -O16,41 -E2,1 -s200 -z200
|
||||
.B -w19 -U50,500 --rmq -r1k,100k -g10k -A1 -B9 -O16,41 -E2,1 -s200 -z200
|
||||
.BR -N50 ).
|
||||
Up to 10% sequence divergence.
|
||||
Use this if the average divergence is around 1%.
|
||||
.TP
|
||||
.B asm20
|
||||
Long assembly to reference mapping
|
||||
.RB ( -k19
|
||||
.B -w10 -U50,500 --rmq -r100k -g10k -A1 -B4 -O6,26 -E2,1 -s200 -z200
|
||||
.B -w10 -U50,500 --rmq -r1k,100k -g10k -A1 -B4 -O6,26 -E2,1 -s200 -z200
|
||||
.BR -N50 ).
|
||||
Up to 20% sequence divergence.
|
||||
Use this if the average divergence is around several percent.
|
||||
.TP
|
||||
.B splice
|
||||
Long-read spliced alignment
|
||||
.RB ( -k15
|
||||
.B -w5 --splice -g2k -G200k -A1 -B2 -O2,32 -E1,0 -b0 -C9 -z200 -ub --junc-bonus=9 --cap-sw-mem=0
|
||||
.B -w5 --splice -g2k -G200k -A1 -B2 -O2,32 -E1,0 -C9 -z200 -ub --junc-bonus=9 --cap-sw-mem=0
|
||||
.BR --splice-flank=yes ).
|
||||
In the splice mode, 1) long deletions are taken as introns and represented as
|
||||
the
|
||||
@@ -595,15 +701,21 @@ costs are different during chaining; 4) the computation of the
|
||||
tag ignores introns to demote hits to pseudogenes.
|
||||
.TP
|
||||
.B splice:hq
|
||||
Long-read splice alignment for PacBio CCS reads
|
||||
Spliced alignment for accurate long RNA-seq reads such as PacBio iso-seq
|
||||
.RB ( -xsplice
|
||||
.B -C5 -O6,24
|
||||
.BR -B4 ).
|
||||
.TP
|
||||
.B splice:sr
|
||||
Spliced alignment for short RNA-seq reads
|
||||
.RB ( -xsplice:hq
|
||||
.B --frag=yes -m25 -s40 -2K100m --heap-sort=yes --pairing=weak --sr=rna --min-dp-len=20
|
||||
.BR --secondary=no ).
|
||||
.TP
|
||||
.B sr
|
||||
Short single-end reads without splicing
|
||||
Short-read alignment without splicing
|
||||
.RB ( -k21
|
||||
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -b0 -r100 -p.5 -N20 -f1000,5000 -n2 -m20
|
||||
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -r100 -p.5 -N20 -f1000,5000 -n2 -m25
|
||||
.B -s40 -g100 -2K50m --heap-sort=yes
|
||||
.BR --secondary=no ).
|
||||
.TP
|
||||
@@ -676,7 +788,7 @@ s2 i Chaining score of the best secondary chain
|
||||
NM i Total number of mismatches and gaps in the alignment
|
||||
MD Z To generate the ref sequence in the alignment
|
||||
AS i DP alignment score
|
||||
SA Z List of other supplementary alignments
|
||||
SA Z List of other supplementary alignments (with approximate CIGAR strings)
|
||||
ms i DP score of the max scoring segment in the alignment
|
||||
nn i Number of ambiguous bases in the alignment
|
||||
ts A Transcript strand (splice mode only)
|
||||
@@ -685,6 +797,7 @@ cs Z Difference string
|
||||
dv f Approximate per-base sequence divergence
|
||||
de f Gap-compressed per-base sequence divergence
|
||||
rl i Length of query regions harboring repetitive seeds
|
||||
zd i Alignment broken due to Z-drop; bit 1: left broken; bit 2: right broken
|
||||
.TE
|
||||
|
||||
.PP
|
||||
|
||||
+2
-1
@@ -16,7 +16,8 @@ minimap2 -c test/MT-human.fa test/MT-orang.fa \
|
||||
| paftools.js liftover -l10000 - <(echo -e "MT_orang\t2000\t5000") # liftOver
|
||||
# no test data for the following examples
|
||||
paftools.js junceval -e anno.gtf splice.sam > out.txt # compare splice junctions to annotations
|
||||
paftools.js splice2bed anno.gtf > anno.bed # convert GTF/GFF3 to BED12
|
||||
paftools.js splice2bed splice.sam > splice.bed # convert PAF/SAM to BED12
|
||||
paftools.js gff2bed anno.gtf > anno.bed # convert GTF/GFF3 to BED12
|
||||
```
|
||||
|
||||
## Table of Contents
|
||||
|
||||
-335
@@ -1,335 +0,0 @@
|
||||
#!/usr/bin/env k8
|
||||
|
||||
var getopt = function(args, ostr) {
|
||||
var oli; // option letter list index
|
||||
if (typeof(getopt.place) == 'undefined')
|
||||
getopt.ind = 0, getopt.arg = null, getopt.place = -1;
|
||||
if (getopt.place == -1) { // update scanning pointer
|
||||
if (getopt.ind >= args.length || args[getopt.ind].charAt(getopt.place = 0) != '-') {
|
||||
getopt.place = -1;
|
||||
return null;
|
||||
}
|
||||
if (getopt.place + 1 < args[getopt.ind].length && args[getopt.ind].charAt(++getopt.place) == '-') { // found "--"
|
||||
++getopt.ind;
|
||||
getopt.place = -1;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
var optopt = args[getopt.ind].charAt(getopt.place++); // character checked for validity
|
||||
if (optopt == ':' || (oli = ostr.indexOf(optopt)) < 0) {
|
||||
if (optopt == '-') return null; // if the user didn't specify '-' as an option, assume it means null.
|
||||
if (getopt.place < 0) ++getopt.ind;
|
||||
return '?';
|
||||
}
|
||||
if (oli+1 >= ostr.length || ostr.charAt(++oli) != ':') { // don't need argument
|
||||
getopt.arg = null;
|
||||
if (getopt.place < 0 || getopt.place >= args[getopt.ind].length) ++getopt.ind, getopt.place = -1;
|
||||
} else { // need an argument
|
||||
if (getopt.place >= 0 && getopt.place < args[getopt.ind].length)
|
||||
getopt.arg = args[getopt.ind].substr(getopt.place);
|
||||
else if (args.length <= ++getopt.ind) { // no arg
|
||||
getopt.place = -1;
|
||||
if (ostr.length > 0 && ostr.charAt(0) == ':') return ':';
|
||||
return '?';
|
||||
} else getopt.arg = args[getopt.ind]; // white space
|
||||
getopt.place = -1;
|
||||
++getopt.ind;
|
||||
}
|
||||
return optopt;
|
||||
}
|
||||
|
||||
function read_fastx(file, buf)
|
||||
{
|
||||
if (file.readline(buf) < 0) return null;
|
||||
var m, line = buf.toString();
|
||||
if ((m = /^([>@])(\S+)/.exec(line)) == null)
|
||||
throw Error("wrong fastx format");
|
||||
var is_fq = (m[1] == '@');
|
||||
var name = m[2];
|
||||
if (file.readline(buf) < 0)
|
||||
throw Error("missing sequence line");
|
||||
var seq = buf.toString();
|
||||
if (is_fq) { // skip quality
|
||||
file.readline(buf);
|
||||
file.readline(buf);
|
||||
}
|
||||
return [name, seq];
|
||||
}
|
||||
|
||||
function filter_paf(a, opt)
|
||||
{
|
||||
if (a.length == 0) return;
|
||||
var k = 0;
|
||||
for (var i = 0; i < a.length; ++i) {
|
||||
var ai = a[i];
|
||||
if (ai[10] < opt.min_blen) continue;
|
||||
if (ai[9] < ai[10] * opt.min_iden) continue;
|
||||
var clip = [0, 0];
|
||||
if (ai[4] == '+') {
|
||||
clip[0] = ai[2] < ai[7]? ai[2] : ai[7];
|
||||
clip[1] = ai[1] - ai[3] < ai[6] - ai[8]? ai[1] - ai[3] : ai[6] - ai[8];
|
||||
} else {
|
||||
clip[0] = ai[2] < ai[6] - ai[8]? ai[2] : ai[6] - ai[8];
|
||||
clip[1] = ai[1] - ai[3] < ai[7]? ai[1] - ai[3] : ai[7];
|
||||
}
|
||||
if (clip[0] > opt.max_clip_len || clip[1] > opt.max_clip_len) continue;
|
||||
a[k++] = ai;
|
||||
}
|
||||
a.length = k;
|
||||
}
|
||||
|
||||
function parse_events(t, ev, id, buf)
|
||||
{
|
||||
var re = /(:(\d+))|(([\+\-\*])([a-z]+))/g;
|
||||
var m, cs = null;
|
||||
for (var j = 12; j < t.length; ++j) {
|
||||
if ((m = /^cs:Z:(\S+)/.exec(t[j])) != null) {
|
||||
cs = m[1].toLowerCase();
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (cs == null) {
|
||||
warn("Warning: no cs tag for read '" + t[0] + "'");
|
||||
return;
|
||||
}
|
||||
var st = t[2], en = t[3];
|
||||
var x = st;
|
||||
while ((m = re.exec(cs)) != null) {
|
||||
var l;
|
||||
if (m[2] != null) { // an identitcal match ":\d+"
|
||||
l = parseInt(m[2]);
|
||||
// [start, end, type, index, changed_base]
|
||||
ev.push([x, x + l, 0, id]);
|
||||
} else {
|
||||
if (m[4] == '*') {
|
||||
l = 1;
|
||||
ev.push([x, x + 1, 1, id, m[5][0]]);
|
||||
} else if (m[4] == '+') {
|
||||
l = m[5].length;
|
||||
ev.push([x, x + l, 2, id]);
|
||||
} else if (m[4] == '-') {
|
||||
l = 0;
|
||||
ev.push([x, x, -1, id, m[5]]);
|
||||
}
|
||||
}
|
||||
x += l;
|
||||
}
|
||||
if (x != en)
|
||||
throw Error("inconsistent cs for read '" + t[0] + "'");
|
||||
}
|
||||
|
||||
function find_het_sub(ev, a, opt)
|
||||
{
|
||||
var n = a.length, last0_i = -1, h = [], d = [];
|
||||
for (var i = 0; i < n; ++i) h[i] = [], d[i] = [];
|
||||
for (var i = 0; i < ev.length; ++i) {
|
||||
if (ev[i][2] == 0) {
|
||||
if (last0_i < 0 || ev[i][0] != ev[last0_i][0]) last0_i = i;
|
||||
else if (ev[i][1] > ev[last0_i][1])
|
||||
last0_i = i;
|
||||
} else if (ev[i][2] == 1 && last0_i >= 0 && ev[i][0] < ev[last0_i][1]) {
|
||||
if (ev[last0_i][1] - ev[last0_i][0] >= opt.min_mlen) {
|
||||
if (opt.dbg_ev) print("EV", ev[last0_i].join("\t"), "|", ev[i].join("\t"));
|
||||
var e0 = ev[last0_i], hl = h[e0[3]];
|
||||
if (hl.length == 0 || hl[hl.length-1][0] != e0[0])
|
||||
hl.push([e0[0], e0[1]]);
|
||||
d[ev[i][3]].push([ev[i][0], e0[1] - e0[0]]);
|
||||
}
|
||||
}
|
||||
}
|
||||
var b = [];
|
||||
for (var i = 0; i < n; ++i) {
|
||||
var sh = 0, dh = 0;
|
||||
for (var j = 0; j < h[i].length; ++j)
|
||||
sh += h[i][j][1] - h[i][j][0];
|
||||
for (var j = 0; j < d[i].length; ++j)
|
||||
dh += d[i][j][1];
|
||||
// [start, end, index, #consistent, lenConsistent, #conflictive, lenConflictive, identity, mlen]
|
||||
b[i] = [a[i][2], a[i][3], i, h[i].length, sh, d[i].length, dh, a[i][9] / a[i][10], a[i][9]];
|
||||
}
|
||||
return b;
|
||||
}
|
||||
|
||||
function flt_utg_for_ec(b, opt)
|
||||
{
|
||||
var k = 0;
|
||||
for (var i = 0; i < b.length; ++i) {
|
||||
var bi = b[i];
|
||||
if (bi[4] == 0 && bi[6] == 0) b[k++] = bi; // entirely ambiguous
|
||||
else if (bi[6] < (bi[4] + bi[6]) * opt.max_ratio0) b[k++] = bi;
|
||||
}
|
||||
b.length = k;
|
||||
if (b.length == 0) return;
|
||||
// find the longest contiguous segment
|
||||
b.sort(function(x,y) { return x[0]-y[0] });
|
||||
var st = b[0][0], en = b[0][1], max_st = 0, max_en = 0, max_max_en = en;
|
||||
for (var i = 1; i < b.length; ++i) {
|
||||
if (b[i][0] > en) {
|
||||
if (en - st > max_en - max_st)
|
||||
max_st = st, max_en = en;
|
||||
st = b[i][0], en = b[i][1];
|
||||
} else {
|
||||
en = en > b[i][1]? en : b[i][1];
|
||||
}
|
||||
max_max_en = max_max_en > b[i][1]? max_max_en : b[i][1];
|
||||
}
|
||||
if (en - st > max_en - max_st)
|
||||
max_st = st, max_en = en;
|
||||
if (max_max_en != en || st != b[0][0]) {
|
||||
var k = 0;
|
||||
for (var i = 0; i < b.length; ++i)
|
||||
if (b[i][0] < max_en && b[i][1] > max_st)
|
||||
b[k++] = b[i];
|
||||
b.length = k;
|
||||
}
|
||||
}
|
||||
|
||||
function flt_utg_for_bin(b, opt) // filter out alignments clearly on the wrong phase
|
||||
{
|
||||
var k = 0;
|
||||
for (var i = 0; i < b.length; ++i) {
|
||||
var bi = b[i];
|
||||
if (bi[4] + bi[6] == 0 || bi[4] >= (bi[4] + bi[6]) * opt.max_ratio0) b[k++] = bi;
|
||||
}
|
||||
b.length = k;
|
||||
}
|
||||
|
||||
function ec_core(b, n_a, ev, buf, ecb) // error correction
|
||||
{
|
||||
var intv = [];
|
||||
for (var i = 0; i < n_a; ++i)
|
||||
intv[i] = null;
|
||||
intv[b[0][2]] = [b[0][0], b[0][1]];
|
||||
var en = b[0][1];
|
||||
for (var i = 1; i < b.length; ++i) {
|
||||
if (b[i][1] <= en) continue;
|
||||
intv[b[i][2]] = [en, b[i][1]];
|
||||
en = b[i][1];
|
||||
}
|
||||
var k = 0;
|
||||
ecb.capacity = buf.capacity;
|
||||
ecb.length = 0;
|
||||
for (var i = 0; i < ev.length; ++i) {
|
||||
var e = ev[i], I = intv[e[3]];
|
||||
if (I == null) continue;
|
||||
if (e[0] >= I[0] && e[0] < I[1]) { // this is to reduce duplicated events around junctions
|
||||
//print("X", e.join("\t"));
|
||||
if (e[2] == 0) {
|
||||
ecb.length += e[1] - e[0];
|
||||
for (var j = e[0]; j < e[1]; ++j)
|
||||
ecb[k++] = buf[j];
|
||||
} else if (e[2] == 1) {
|
||||
++ecb.length;
|
||||
ecb[k++] = e[4].charCodeAt(0);
|
||||
} else if (e[2] < 0) {
|
||||
ecb.length += e[4].length;
|
||||
for (var j = 0; j < e[4].length; ++j)
|
||||
ecb[k++] = e[4].charCodeAt(j);
|
||||
} // else, skip e[2] == 2
|
||||
}
|
||||
}
|
||||
if (ecb.length != k) throw Error("BUG!");
|
||||
}
|
||||
|
||||
function process_paf(a, opt, fp_seq, buf, ecb)
|
||||
{
|
||||
if (a.length == 0) return;
|
||||
var len = a[0][1], name = a[0][0], seq = null;
|
||||
if (len < opt.min_rlen) return;
|
||||
if (fp_seq) {
|
||||
var ret;
|
||||
while ((ret = read_fastx(fp_seq, buf)) != null)
|
||||
if (ret[0] == a[0][0])
|
||||
break;
|
||||
if (ret == null)
|
||||
throw Error("failed to find sequence for read '" + a[0][0] + "'");
|
||||
name = ret[0], seq = ret[1];
|
||||
if (seq.length != len)
|
||||
throw Error("inconsistent length for read '" + name + "'");
|
||||
}
|
||||
filter_paf(a, opt);
|
||||
if (a.length == 0) return;
|
||||
var ev = [];
|
||||
for (var i = 0; i < a.length; ++i)
|
||||
parse_events(a[i], ev, i, buf);
|
||||
ev.sort(function(x,y) { return x[0]!=y[0]? x[0]-y[0] : x[2]-y[2] });
|
||||
if (seq == null) print("SQ", name, a[0][1], a.length);
|
||||
var b = find_het_sub(ev, a, opt);
|
||||
if (opt.ec) flt_utg_for_ec(b, opt);
|
||||
else flt_utg_for_bin(b, opt);
|
||||
if (seq == null) {
|
||||
for (var i = 0; i < b.length; ++i) {
|
||||
var m, ai = a[b[i][2]], score = 0;
|
||||
for (var j = 10; j < ai.length; ++j)
|
||||
if ((m = /^AS:i:(\d+)/.exec(ai[j])) != null)
|
||||
score = m[1];
|
||||
print("TS", b[i][2], b[i][0], b[i][1], ai.slice(5, 9).join("\t"), b[i].slice(3, 7).join("\t"), score);
|
||||
}
|
||||
print("//");
|
||||
} else { // error correction
|
||||
if (b.length == 0) return;
|
||||
buf.set(seq, 0);
|
||||
ec_core(b, a.length, ev, buf, ecb);
|
||||
print(">" + name);
|
||||
print(ecb);
|
||||
}
|
||||
}
|
||||
|
||||
function main(args)
|
||||
{
|
||||
var c, opt = { min_rlen:5000, min_blen:5000, min_iden:0.8, min_mlen:5, max_clip_len:500, max_ratio0:0.25, dbg_ev:false };
|
||||
while ((c = getopt(args, "l:b:d:m:c:r:E")) != null) {
|
||||
if (c == 'l') opt.min_rlen = parseInt(getopt.arg);
|
||||
else if (c == 'b') opt.min_blen = parseInt(getopt.arg);
|
||||
else if (c == 'd') opt.min_iden = parseFloat(getopt.arg);
|
||||
else if (c == 'm') opt.min_slen = parseInt(getopt.arg);
|
||||
else if (c == 'c') opt.max_clip_len = parseInt(getopt.arg);
|
||||
else if (c == 'r') opt.max_ratio0 = parseFloat(getopt.arg);
|
||||
else if (c == 'E') opt.dbg_ev = true;
|
||||
}
|
||||
if (args.length - getopt.ind < 1) {
|
||||
print("Usage: mmphase.js [options] <map-with-cs.paf> [reads.fa]");
|
||||
print("Options:");
|
||||
print(" -l INT min read length [" + opt.min_rlen + "]");
|
||||
print(" -b INT min alignment length [" + opt.min_blen + "]");
|
||||
print(" -d FLOAT min identity [" + opt.min_iden + "]");
|
||||
print(" -s INT min match length [" + opt.min_mlen + "]");
|
||||
print(" -c INT max clip length [" + opt.max_clip_len + "]");
|
||||
print(" -r FLOAT initial ratio for haplotype filtering [" + opt.max_ratio0 + "]");
|
||||
return 0;
|
||||
}
|
||||
|
||||
opt.ec = args.length - getopt.ind < 2? false : true;
|
||||
if (!opt.ec) {
|
||||
print("CC");
|
||||
print("CC", "SQ qName qLen nHits");
|
||||
print("CC", "TS index qStart qEnd tName tLen tStart tEnd nConsistent lCons nConflictive lConf score");
|
||||
print("CC");
|
||||
}
|
||||
|
||||
var buf = new Bytes(), ecb = new Bytes();
|
||||
var fp_paf = new File(args[getopt.ind]);
|
||||
var fp_seq = args.length - getopt.ind >= 2? new File(args[getopt.ind+1]) : null;
|
||||
var a = [];
|
||||
while (fp_paf.readline(buf) >= 0) {
|
||||
var t = buf.toString().split("\t");
|
||||
if (a.length > 0 && a[0][0] != t[0]) {
|
||||
process_paf(a, opt, fp_seq, buf, ecb);
|
||||
a.length = 0;
|
||||
}
|
||||
for (var i = 1; i <= 3; ++i) t[i] = parseInt(t[i]);
|
||||
if (t[1] < opt.min_rlen) continue;
|
||||
for (var i = 6; i <= 10; ++i) t[i] = parseInt(t[i]);
|
||||
if (t[10] < opt.min_blen) continue;
|
||||
a.push(t);
|
||||
}
|
||||
if (a.length >= 0)
|
||||
process_paf(a, opt, fp_seq, buf, ecb);
|
||||
if (fp_seq) fp_seq.close();
|
||||
fp_paf.close();
|
||||
ecb.destroy();
|
||||
buf.destroy();
|
||||
}
|
||||
|
||||
var ret = main(arguments)
|
||||
exit(ret)
|
||||
Executable
+241
@@ -0,0 +1,241 @@
|
||||
#!/usr/bin/env k8
|
||||
|
||||
"use strict";
|
||||
|
||||
Array.prototype.delete_at = function(i) {
|
||||
for (let j = i; j < this.length - 1; ++j)
|
||||
this[j] = this[j + 1];
|
||||
--this.length;
|
||||
}
|
||||
|
||||
function* getopt(argv, ostr, longopts) {
|
||||
if (argv.length == 0) return;
|
||||
let pos = 0, cur = 0;
|
||||
while (cur < argv.length) {
|
||||
let lopt = "", opt = "?", arg = "";
|
||||
while (cur < argv.length) { // skip non-option arguments
|
||||
if (argv[cur][0] == "-" && argv[cur].length > 1) {
|
||||
if (argv[cur] == "--") cur = argv.length;
|
||||
break;
|
||||
} else ++cur;
|
||||
}
|
||||
if (cur == argv.length) break;
|
||||
let a = argv[cur];
|
||||
if (a[0] == "-" && a[1] == "-") { // a long option
|
||||
pos = -1;
|
||||
let c = 0, k = -1, tmp = "", o;
|
||||
const pos_eq = a.indexOf("=");
|
||||
if (pos_eq > 0) {
|
||||
o = a.substring(2, pos_eq);
|
||||
arg = a.substring(pos_eq + 1);
|
||||
} else o = a.substring(2);
|
||||
for (let i = 0; i < longopts.length; ++i) {
|
||||
let y = longopts[i];
|
||||
if (y[y.length - 1] == "=") y = y.substring(0, y.length - 1);
|
||||
if (o.length <= y.length && o == y.substring(0, o.length)) {
|
||||
k = i, tmp = y;
|
||||
++c; // c is the number of matches
|
||||
if (o == y) { // exact match
|
||||
c = 1;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (c == 1) { // find a unique match
|
||||
lopt = tmp;
|
||||
if (pos_eq < 0 && longopts[k][longopts[k].length-1] == "=" && cur + 1 < argv.length) {
|
||||
arg = argv[cur+1];
|
||||
argv.delete_at(cur + 1);
|
||||
}
|
||||
}
|
||||
} else { // a short option
|
||||
if (pos == 0) pos = 1;
|
||||
opt = a[pos++];
|
||||
let k = ostr.indexOf(opt);
|
||||
if (k < 0) {
|
||||
opt = "?";
|
||||
} else if (k + 1 < ostr.length && ostr[k+1] == ":") { // requiring an argument
|
||||
if (pos >= a.length) {
|
||||
arg = argv[cur+1];
|
||||
argv.delete_at(cur + 1);
|
||||
} else arg = a.substring(pos);
|
||||
pos = -1;
|
||||
}
|
||||
}
|
||||
if (pos < 0 || pos >= argv[cur].length) {
|
||||
argv.delete_at(cur);
|
||||
pos = 0;
|
||||
}
|
||||
if (lopt != "") yield { opt: `--${lopt}`, arg: arg };
|
||||
else if (opt != "?") yield { opt: `-${opt}`, arg: arg };
|
||||
else yield { opt: "?", arg: "" };
|
||||
}
|
||||
}
|
||||
|
||||
function* k8_readline(fn) {
|
||||
let buf = new Bytes();
|
||||
let file = new File(fn);
|
||||
while (file.readline(buf) >= 0) {
|
||||
yield buf.toString();
|
||||
}
|
||||
file.close();
|
||||
buf.destroy();
|
||||
}
|
||||
|
||||
function merge_hits(b) {
|
||||
if (b.length == 1)
|
||||
return { name1:b[0].name1, name2:b[0].name2, len1:b[0].len1, len2:b[0].len2, min_cov:b[0].min_cov, max_cov:b[0].max_cov, cov1:b[0].cov1, cov2:b[0].cov2, s1:b[0].s1, dv:b[0].dv };
|
||||
b.sort(function(x, y) { return x.st1 - y.st1 });
|
||||
let f = [], bt = [];
|
||||
for (let i = 0; i < b.length; ++i)
|
||||
f[i] = b[i].s1, bt[i] = -1;
|
||||
for (let i = 0; i < b.length; ++i) {
|
||||
for (let j = 0; j < i; ++j) {
|
||||
if (b[j].st2 < b[i].st2) {
|
||||
if (b[j].en1 >= b[i].en1) continue;
|
||||
if (b[j].en2 >= b[i].en2) continue;
|
||||
const ov1 = b[j].en1 <= b[i].st1? 0 : b[i].st1 - b[j].en1;
|
||||
const li1 = b[i].en1 - b[i].st1;
|
||||
const s11 = b[i].s1 / li1 * (li1 - ov1);
|
||||
const ov2 = b[j].en2 <= b[i].st2? 0 : b[i].st2 - b[j].en2;
|
||||
const li2 = b[i].en2 - b[i].st2;
|
||||
const s12 = b[i].s1 / li2 * (li2 - ov2);
|
||||
const s1 = s11 < s12? s11 : s12;
|
||||
if (f[i] < f[j] + s1)
|
||||
f[i] = f[j] + s1, bt[i] = j;
|
||||
}
|
||||
}
|
||||
}
|
||||
let max_i = -1, max_f = 0, d = [];
|
||||
for (let i = 0; i < b.length; ++i)
|
||||
if (max_f < f[i])
|
||||
max_f = f[i], max_i = i;
|
||||
for (let k = max_i; k >= 0; k = bt[k])
|
||||
d.push(k);
|
||||
d = d.reverse();
|
||||
let dv = 0, tot = 0, cov1 = 0, cov2 = 0, st1 = 0, en1 = 0, st2 = 0, en2 = 0;
|
||||
for (let k = 0; k < d.length; ++k) {
|
||||
const i = d[k];
|
||||
tot += b[i].blen;
|
||||
dv += b[i].dv * b[i].blen;
|
||||
if (b[i].st1 > en1) {
|
||||
cov1 += en1 - st1;
|
||||
st1 = b[i].st1, en1 = b[i].en1;
|
||||
} else en1 = en1 > b[i].en1? en1 : b[i].en1;
|
||||
if (b[i].st2 > en2) {
|
||||
cov2 += en2 - st2;
|
||||
st2 = b[i].st2, en2 = b[i].en2;
|
||||
} else en2 = en2 > b[i].en2? en2 : b[i].en2;
|
||||
}
|
||||
dv /= tot;
|
||||
cov1 = (cov1 + (en1 - st1)) / b[0].len1;
|
||||
cov2 = (cov2 + (en2 - st2)) / b[0].len2;
|
||||
const min_cov = cov1 < cov2? cov1 : cov2;
|
||||
const max_cov = cov1 > cov2? cov1 : cov2;
|
||||
//warn(d.length, b[0].name1, b[0].name2, min_cov, max_cov);
|
||||
return { name1:b[0].name1, name2:b[0].name2, len1:b[0].len1, len2:b[0].len2, min_cov:min_cov, max_cov:max_cov, cov1:cov1, cov2:cov2, s1:max_f, dv:dv };
|
||||
}
|
||||
|
||||
function main(args) {
|
||||
let opt = { min_cov:.9, max_dv:.015, max_diff:20000 };
|
||||
for (const o of getopt(args, "c:d:e:", [])) {
|
||||
if (o.opt == '-c') opt.min_cov = parseFloat(o.arg);
|
||||
else if (o.opt == '-d') opt.max_dv = parseFloat(o.arg);
|
||||
else if (o.opt == '-e') opt.max_diff = parseFloat(o.arg);
|
||||
}
|
||||
if (args.length == 0) {
|
||||
print("Usage: pafcluster.js [options] <ava.paf>");
|
||||
print("Options:");
|
||||
print(` -c FLOAT min coverage [${opt.min_cov}]`);
|
||||
print(` -d FLOAT max divergence [${opt.max_dv}]`);
|
||||
print(` -e FLOAT max difference [${opt.max_diff}]`);
|
||||
return;
|
||||
}
|
||||
|
||||
// read
|
||||
let a = [], len = {}, name2len = {};
|
||||
for (const line of k8_readline(args[0])) {
|
||||
let m, t = line.split("\t");
|
||||
if (t[4] != "+") continue;
|
||||
for (let i = 1; i < 4; ++i) t[i] = parseInt(t[i]);
|
||||
for (let i = 6; i < 11; ++i) t[i] = parseInt(t[i]);
|
||||
const len1 = t[1], len2 = t[6];
|
||||
let s1 = -1, dv = -1.0;
|
||||
for (let i = 12; i < t.length; ++i) {
|
||||
if ((m = /^(s1|dv):\S:(\S+)/.exec(t[i])) != null) {
|
||||
if (m[1] == "s1") s1 = parseInt(m[2]);
|
||||
else if (m[1] == "dv") dv = parseFloat(m[2]);
|
||||
}
|
||||
}
|
||||
if (s1 < 0 || dv < 0) continue;
|
||||
const cov1 = (parseInt(t[3]) - parseInt(t[2])) / len1;
|
||||
const cov2 = (parseInt(t[8]) - parseInt(t[7])) / len2;
|
||||
const min_cov = cov1 < cov2? cov1 : cov2;
|
||||
const max_cov = cov1 > cov2? cov1 : cov2;
|
||||
name2len[t[0]] = len1;
|
||||
name2len[t[5]] = len2;
|
||||
a.push({ name1:t[0], name2:t[5], len1:len1, len2:len2, min_cov:min_cov, max_cov:max_cov, s1:s1, dv:dv, cov1:cov1, cov2:cov2, st1:t[2], en1:t[3], st2:t[7], en2:t[8], blen:t[10] });
|
||||
len[t[0]] = len1, len[t[5]] = len2;
|
||||
}
|
||||
warn(`Read ${a.length} hits`);
|
||||
|
||||
// merge duplicated hits
|
||||
let h = {};
|
||||
for (let i = 0; i < a.length; ++i) {
|
||||
const key = `${a[i].name1}\t${a[i].name2}`;
|
||||
if (h[key] == null) h[key] = [];
|
||||
h[key].push(a[i]);
|
||||
}
|
||||
a = [];
|
||||
for (const key in h)
|
||||
a.push(merge_hits(h[key]));
|
||||
|
||||
// core loop
|
||||
while (a.length > 1) {
|
||||
// select the sequence with the highest sum of s1
|
||||
let h = {};
|
||||
for (let i = 0; i < a.length; ++i) {
|
||||
if (h[a[i].name1] == null) h[a[i].name1] = 0;
|
||||
h[a[i].name1] += a[i].s1;
|
||||
}
|
||||
let max_s1 = 0, max_name = "";
|
||||
for (const name in h)
|
||||
if (max_s1 < h[name])
|
||||
max_s1 = h[name], max_name = name;
|
||||
// find contigs in the same group
|
||||
h = {};
|
||||
h[max_name] = 1;
|
||||
for (let i = 0; i < a.length; ++i) {
|
||||
if (a[i].name1 != max_name && a[i].name2 != max_name)
|
||||
continue;
|
||||
const diff1 = a[i].len1 * (1.0 - a[i].cov1);
|
||||
const diff2 = a[i].len2 * (1.0 - a[i].cov2);
|
||||
if (a[i].min_cov >= opt.min_cov && a[i].dv <= opt.max_dv && diff1 <= opt.max_diff && diff2 <= opt.max_diff)
|
||||
h[a[i].name1] = h[a[i].name2] = 1;
|
||||
}
|
||||
let n = 0;
|
||||
for (const key in h) {
|
||||
++n;
|
||||
delete name2len[key];
|
||||
}
|
||||
print(`SD\t${max_name}\t${n}`);
|
||||
for (const key in h) print(`CL\t${key}\t${len[key]}`);
|
||||
print("//");
|
||||
// filter out redundant hits
|
||||
let b = [];
|
||||
for (let i = 0; i < a.length; ++i)
|
||||
if (h[a[i].name1] == null && h[a[i].name2] == null)
|
||||
b.push(a[i]);
|
||||
warn(`Reduced the number of hits from ${a.length} to ${b.length}`);
|
||||
a = b;
|
||||
}
|
||||
|
||||
// output remaining singletons
|
||||
for (const key in name2len) {
|
||||
print(`SD\t${key}\t1`);
|
||||
print(`CL\t${key}\t${name2len[key]}`);
|
||||
print(`//`);
|
||||
}
|
||||
}
|
||||
|
||||
main(arguments);
|
||||
+639
-61
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env k8
|
||||
|
||||
var paftools_version = '2.23-r1111';
|
||||
var paftools_version = '2.29-r1283';
|
||||
|
||||
/*****************************
|
||||
***** Library functions *****
|
||||
@@ -133,26 +133,50 @@ Interval.find_ovlp = function(a, st, en)
|
||||
|
||||
function fasta_read(fn)
|
||||
{
|
||||
var h = {}, gt = '>'.charCodeAt(0);
|
||||
var h = {}, seqlen = [];
|
||||
var buf = new Bytes();
|
||||
var file = fn == '-'? new File() : new File(fn);
|
||||
var buf = new Bytes(), seq = null, name = null, seqlen = [];
|
||||
while (file.readline(buf) >= 0) {
|
||||
if (buf[0] == gt) {
|
||||
if (seq != null && name != null) {
|
||||
seqlen.push([name, seq.length]);
|
||||
h[name] = seq;
|
||||
name = seq = null;
|
||||
}
|
||||
var m, line = buf.toString();
|
||||
if ((m = /^>(\S+)/.exec(line)) != null) {
|
||||
name = m[1];
|
||||
seq = new Bytes();
|
||||
}
|
||||
} else seq.set(buf);
|
||||
}
|
||||
if (seq != null && name != null) {
|
||||
seqlen.push([name, seq.length]);
|
||||
h[name] = seq;
|
||||
if (typeof k8_version == "undefined") { // for k8-0.x
|
||||
var seq = null, name = null, gt = '>'.charCodeAt(0);
|
||||
while (file.readline(buf) >= 0) {
|
||||
if (buf[0] == gt) {
|
||||
if (seq != null && name != null) {
|
||||
seqlen.push([name, seq.length]);
|
||||
h[name] = seq;
|
||||
name = seq = null;
|
||||
}
|
||||
var m, line = buf.toString();
|
||||
if ((m = /^>(\S+)/.exec(line)) != null) {
|
||||
name = m[1];
|
||||
seq = new Bytes();
|
||||
}
|
||||
} else seq.set(buf);
|
||||
}
|
||||
if (seq != null && name != null) {
|
||||
seqlen.push([name, seq.length]);
|
||||
h[name] = seq;
|
||||
}
|
||||
} else { // for k8-1.x
|
||||
var seq = null, name = null;
|
||||
while (file.readline(buf) >= 0) {
|
||||
var line = buf.toString();
|
||||
if (line[0] == ">") {
|
||||
if (seq != null && name != null) {
|
||||
seqlen.push([name, seq.length]);
|
||||
h[name] = new Uint8Array(seq.buffer);
|
||||
name = seq = null;
|
||||
}
|
||||
var m;
|
||||
if ((m = /^>(\S+)/.exec(line)) != null) {
|
||||
name = m[1];
|
||||
seq = new Bytes();
|
||||
}
|
||||
} else seq.set(line);
|
||||
}
|
||||
if (seq != null && name != null) {
|
||||
seqlen.push([name, seq.length]);
|
||||
h[name] = new Uint8Array(seq.buffer);
|
||||
}
|
||||
}
|
||||
buf.destroy();
|
||||
file.close();
|
||||
@@ -161,16 +185,27 @@ function fasta_read(fn)
|
||||
|
||||
function fasta_free(fa)
|
||||
{
|
||||
for (var name in fa)
|
||||
fa[name].destroy();
|
||||
if (typeof k8_version == "undefined")
|
||||
for (var name in fa)
|
||||
fa[name].destroy();
|
||||
// FIXME: for k8-1.0, sequences are not freed. This is ok for now but not general.
|
||||
}
|
||||
|
||||
Bytes.prototype.reverse = function()
|
||||
{
|
||||
for (var i = 0; i < this.length>>1; ++i) {
|
||||
var tmp = this[i];
|
||||
this[i] = this[this.length - i - 1];
|
||||
this[this.length - i - 1] = tmp;
|
||||
if (typeof k8_version === "undefined") { // k8-0.x
|
||||
for (var i = 0; i < this.length>>1; ++i) {
|
||||
var tmp = this[i];
|
||||
this[i] = this[this.length - i - 1];
|
||||
this[this.length - i - 1] = tmp;
|
||||
}
|
||||
} else { // k8-1.x
|
||||
var buf = new Uint8Array(this.buffer);
|
||||
for (var i = 0; i < buf.length>>1; ++i) {
|
||||
var tmp = buf[i];
|
||||
buf[i] = buf[buf.length - i - 1];
|
||||
buf[buf.length - i - 1] = tmp;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -185,13 +220,24 @@ Bytes.prototype.revcomp = function()
|
||||
for (var i = 0; i < s1.length; ++i)
|
||||
Bytes.rctab[s1.charCodeAt(i)] = s2.charCodeAt(i);
|
||||
}
|
||||
for (var i = 0; i < this.length>>1; ++i) {
|
||||
var tmp = this[this.length - i - 1];
|
||||
this[this.length - i - 1] = Bytes.rctab[this[i]];
|
||||
this[i] = Bytes.rctab[tmp];
|
||||
if (typeof k8_version === "undefined") { // k8-0.x
|
||||
for (var i = 0; i < this.length>>1; ++i) {
|
||||
var tmp = this[this.length - i - 1];
|
||||
this[this.length - i - 1] = Bytes.rctab[this[i]];
|
||||
this[i] = Bytes.rctab[tmp];
|
||||
}
|
||||
if (this.length&1)
|
||||
this[this.length>>1] = Bytes.rctab[this[this.length>>1]];
|
||||
} else { // k8-1.x
|
||||
var buf = new Uint8Array(this.buffer);
|
||||
for (var i = 0; i < buf.length>>1; ++i) {
|
||||
var tmp = buf[buf.length - i - 1];
|
||||
buf[buf.length - i - 1] = Bytes.rctab[buf[i]];
|
||||
buf[i] = Bytes.rctab[tmp];
|
||||
}
|
||||
if (buf.length&1)
|
||||
buf[buf.length>>1] = Bytes.rctab[buf[buf.length>>1]];
|
||||
}
|
||||
if (this.length&1)
|
||||
this[this.length>>1] = Bytes.rctab[this[this.length>>1]];
|
||||
}
|
||||
|
||||
/********************
|
||||
@@ -1532,22 +1578,24 @@ function paf_view(args)
|
||||
|
||||
function paf_gff2bed(args)
|
||||
{
|
||||
var c, fn_ucsc_fai = null, is_short = false, keep_gff = false, print_junc = false, output_gene = false;
|
||||
while ((c = getopt(args, "u:sgjG")) != null) {
|
||||
var c, fn_ucsc_fai = null, is_short = false, keep_gff = false, print_junc = false, output_gene = false, ens_canon_only = false;
|
||||
while ((c = getopt(args, "u:sgjGe")) != null) {
|
||||
if (c == 'u') fn_ucsc_fai = getopt.arg;
|
||||
else if (c == 's') is_short = true;
|
||||
else if (c == 'g') keep_gff = true;
|
||||
else if (c == 'j') print_junc = true;
|
||||
else if (c == 'G') output_gene = true;
|
||||
else if (c == 'e') ens_canon_only = true;
|
||||
}
|
||||
|
||||
if (getopt.ind == args.length) {
|
||||
print("Usage: paftools.js gff2bed [options] <in.gff>");
|
||||
print("Options:");
|
||||
print(" -j Output junction BED");
|
||||
print(" -s Print names in the short form");
|
||||
print(" -j output junction BED");
|
||||
print(" -s print names in the short form");
|
||||
print(" -u FILE hg38.fa.fai for chr name conversion");
|
||||
print(" -g Output GFF (used with -u)");
|
||||
print(" -e only show transcript tagged with 'Ensembl_canonical'");
|
||||
print(" -g output GFF (used with -u)");
|
||||
exit(1);
|
||||
}
|
||||
|
||||
@@ -1606,7 +1654,7 @@ function paf_gff2bed(args)
|
||||
print(a[0][0], st, en, name, 1000, a[0][3], cds_st, cds_en, color, a.length, sizes.join(",") + ",", starts.join(",") + ",");
|
||||
}
|
||||
|
||||
var re_gtf = /\b(transcript_id|transcript_type|transcript_biotype|gene_name|gene_id|gbkey|transcript_name) "([^"]+)";/g;
|
||||
var re_gtf = /\b(transcript_id|transcript_type|transcript_biotype|gene_name|gene_id|gbkey|transcript_name|tag) "([^"]+)";/g;
|
||||
var re_gff3 = /\b(transcript_id|transcript_type|transcript_biotype|gene_name|gene_id|gbkey|transcript_name)=([^;]+)/g;
|
||||
var re_gtf_gene = /\b(gene_id|gene_type|gene_name) "([^;]+)";/g;
|
||||
var re_gff3_gene = /\b(gene_id|gene_type|source_gene|gene_biotype|gene_name)=([^;]+);/g;
|
||||
@@ -1646,13 +1694,14 @@ function paf_gff2bed(args)
|
||||
if (t[2] != "CDS" && t[2] != "exon") continue;
|
||||
t[3] = parseInt(t[3]) - 1;
|
||||
t[4] = parseInt(t[4]);
|
||||
var id = null, type = "", name = "N/A", biotype = "", m, tname = "N/A";
|
||||
var id = null, type = "", name = "N/A", biotype = "", m, tname = "N/A", ens_canonical = false;
|
||||
while ((m = re_gtf.exec(t[8])) != null) {
|
||||
if (m[1] == "transcript_id") id = m[2];
|
||||
else if (m[1] == "transcript_type") type = m[2];
|
||||
else if (m[1] == "transcript_biotype" || m[1] == "gbkey") biotype = m[2];
|
||||
else if (m[1] == "gene_name" || m[1] == "gene_id") name = m[2];
|
||||
else if (m[1] == "transcript_name") tname = m[2];
|
||||
else if (m[1] == "tag" && m[2] == "Ensembl_canonical") ens_canonical = true;
|
||||
}
|
||||
while ((m = re_gff3.exec(t[8])) != null) {
|
||||
if (m[1] == "transcript_id") id = m[2];
|
||||
@@ -1661,6 +1710,7 @@ function paf_gff2bed(args)
|
||||
else if (m[1] == "gene_name" || m[1] == "gene_id") name = m[2];
|
||||
else if (m[1] == "transcript_name") tname = m[2];
|
||||
}
|
||||
if (ens_canon_only && !ens_canonical) continue;
|
||||
if (type == "" && biotype != "") type = biotype;
|
||||
if (id == null) throw Error("No transcript_id");
|
||||
if (id != last_id) {
|
||||
@@ -1690,15 +1740,17 @@ function paf_gff2bed(args)
|
||||
|
||||
function paf_sam2paf(args)
|
||||
{
|
||||
var c, pri_only = false, long_cs = false;
|
||||
while ((c = getopt(args, "pL")) != null) {
|
||||
var c, pri_only = false, long_cs = false, pri_pri_only = false;
|
||||
while ((c = getopt(args, "pPL")) != null) {
|
||||
if (c == 'p') pri_only = true;
|
||||
else if (c == 'P') pri_pri_only = pri_only = true;
|
||||
else if (c == 'L') long_cs = true;
|
||||
}
|
||||
if (args.length == getopt.ind) {
|
||||
print("Usage: paftools.js sam2paf [options] <in.sam>");
|
||||
print("Options:");
|
||||
print(" -p convert primary or supplementary alignments only");
|
||||
print(" -P convert primary alignments only");
|
||||
print(" -L output the cs tag in the long form");
|
||||
exit(1);
|
||||
}
|
||||
@@ -1725,6 +1777,7 @@ function paf_sam2paf(args)
|
||||
throw Error("at line " + lineno + ": inconsistent SEQ and QUAL lengths - " + t[9].length + " != " + t[10].length);
|
||||
if (t[2] == '*' || (flag&4) || t[5] == '*') continue;
|
||||
if (pri_only && (flag&0x100)) continue;
|
||||
if (pri_pri_only && (flag&0x900)) continue;
|
||||
var tlen = ctg_len[t[2]];
|
||||
if (tlen == null) throw Error("at line " + lineno + ": can't find the length of contig " + t[2]);
|
||||
// find tags
|
||||
@@ -1837,7 +1890,10 @@ function paf_sam2paf(args)
|
||||
// optional tags
|
||||
var type = flag&0x100? 'S' : 'P';
|
||||
var tags = ["tp:A:" + type];
|
||||
if (NM != null) tags.push("mm:i:"+mm);
|
||||
if (NM != null) {
|
||||
tags.push("NM:i:"+NM);
|
||||
tags.push("mm:i:"+mm);
|
||||
}
|
||||
tags.push("gn:i:"+(I[1]+D[1]), "go:i:"+(I[0]+D[0]), "cg:Z:" + t[5].replace(/\d+[SH]/g, ''));
|
||||
if (cs_str != null) tags.push("cs:Z:" + cs_str);
|
||||
else if (cs.length > 0) tags.push("cs:Z:" + cs.join(""));
|
||||
@@ -2047,7 +2103,7 @@ function paf_mapeval(args)
|
||||
warn("Usage: paftools.js mapeval [options] <in.paf>|<in.sam>");
|
||||
warn("Options:");
|
||||
warn(" -r FLOAT mapping correct if overlap_length/union_length>FLOAT [" + ovlp_ratio + "]");
|
||||
warn(" -Q INT print wrong mappings with mapQ>INT [don't print]");
|
||||
warn(" -Q INT print wrong mappings with mapQ>=INT [don't print]");
|
||||
warn(" -m INT 0: eval the longest aln only; 1: first aln only; 2: all primary aln [0]");
|
||||
exit(1);
|
||||
}
|
||||
@@ -2131,7 +2187,7 @@ function paf_mapeval(args)
|
||||
}
|
||||
|
||||
var lineno = 0, last = null, a = [], n_unmapped = null;
|
||||
var re_cigar = /(\d+)([MIDSHN])/g;
|
||||
var re_cigar = /(\d+)([MIDSHN=X])/g;
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, line = buf.toString();
|
||||
++lineno;
|
||||
@@ -2169,7 +2225,7 @@ function paf_mapeval(args)
|
||||
var n_gap = 0, mlen = 0;
|
||||
while ((m = re_cigar.exec(t[5])) != null) {
|
||||
var len = parseInt(m[1]);
|
||||
if (m[2] == 'M') pos_end += len, mlen += len;
|
||||
if (m[2] == 'M' || m[2] == 'X' || m[2] == '=') pos_end += len, mlen += len;
|
||||
else if (m[2] == 'I') n_gap += len;
|
||||
else if (m[2] == 'D') n_gap += len, pos_end += len;
|
||||
}
|
||||
@@ -2341,12 +2397,15 @@ function paf_pbsim2fq(args)
|
||||
|
||||
function paf_junceval(args)
|
||||
{
|
||||
var c, l_fuzzy = 0, print_ovlp = false, print_err_only = false, first_only = false, chr_only = false;
|
||||
while ((c = getopt(args, "l:epc")) != null) {
|
||||
var c, l_fuzzy = 0, print_ovlp = false, print_err_only = false, first_only = false, chr_only = false, aa = false, is_bed = false;
|
||||
while ((c = getopt(args, "l:epcab1")) != null) {
|
||||
if (c == 'l') l_fuzzy = parseInt(getopt.arg);
|
||||
else if (c == 'e') print_err_only = print_ovlp = true;
|
||||
else if (c == 'p') print_ovlp = true;
|
||||
else if (c == 'c') chr_only = true;
|
||||
else if (c == 'a') aa = true;
|
||||
else if (c == 'b') is_bed = true;
|
||||
else if (c == '1') first_only = true;
|
||||
}
|
||||
|
||||
if (args.length - getopt.ind < 1) {
|
||||
@@ -2356,6 +2415,9 @@ function paf_junceval(args)
|
||||
print(" -p print overlapping introns");
|
||||
print(" -e print erroreous overlapping introns");
|
||||
print(" -c only consider alignments to /^(chr)?([0-9]+|X|Y)$/");
|
||||
print(" -a miniprot PAF as input");
|
||||
print(" -b BED as input");
|
||||
print(" -1 only process the first alignment of each query");
|
||||
exit(1);
|
||||
}
|
||||
|
||||
@@ -2409,13 +2471,17 @@ function paf_junceval(args)
|
||||
|
||||
file = getopt.ind+1 >= args.length || args[getopt.ind+1] == '-'? new File() : new File(args[getopt.ind+1]);
|
||||
var last_qname = null;
|
||||
var re_cigar = /(\d+)([MIDNSHP=X])/g;
|
||||
var re_cigar = /(\d+)([MIDNSHP=XFGUV])/g;
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, t = buf.toString().split("\t");
|
||||
var ctg_name = null, cigar = null, pos = null, qname = t[0];
|
||||
var ctg_name = null, cigar = null, pos = null, qname;
|
||||
|
||||
if (t[0].charAt(0) == '@') continue;
|
||||
if (t[4] == '+' || t[4] == '-' || t[4] == '*') { // PAF
|
||||
if (t[0] == "##PAF") t.shift();
|
||||
qname = t[0];
|
||||
if (is_bed) {
|
||||
ctg_name = t[0], pos = parseInt(t[1]), cigar == null;
|
||||
} else if (t[4] == '+' || t[4] == '-' || t[4] == '*') { // PAF
|
||||
ctg_name = t[5], pos = parseInt(t[7]);
|
||||
var type = 'P';
|
||||
for (i = 12; i < t.length; ++i) {
|
||||
@@ -2428,6 +2494,10 @@ function paf_junceval(args)
|
||||
} else { // SAM
|
||||
ctg_name = t[2], pos = parseInt(t[3]) - 1, cigar = t[5];
|
||||
var flag = parseInt(t[1]);
|
||||
if (flag & 1) {
|
||||
if (flag & 0x40) qname += '/1';
|
||||
else if (flag & 0x80) qname += '/2';
|
||||
}
|
||||
if (flag&0x100) continue; // secondary
|
||||
}
|
||||
|
||||
@@ -2445,12 +2515,43 @@ function paf_junceval(args)
|
||||
}
|
||||
|
||||
var intron = [];
|
||||
while ((m = re_cigar.exec(cigar)) != null) {
|
||||
var len = parseInt(m[1]), op = m[2];
|
||||
if (op == 'N') {
|
||||
intron.push([pos, pos + len]);
|
||||
pos += len;
|
||||
} else if (op == 'M' || op == 'X' || op == '=' || op == 'D') pos += len;
|
||||
if (is_bed) {
|
||||
intron.push([pos, parseInt(t[2])]);
|
||||
} else if (aa) {
|
||||
var tmp_junc = [], tmp = 0;
|
||||
while ((m = re_cigar.exec(cigar)) != null) {
|
||||
var len = parseInt(m[1]), op = m[2];
|
||||
if (op == 'N') {
|
||||
tmp_junc.push([tmp, tmp + len]);
|
||||
tmp += len;
|
||||
} else if (op == 'U') {
|
||||
tmp_junc.push([tmp + 1, tmp + len - 2]);
|
||||
tmp += len;
|
||||
} else if (op == 'V') {
|
||||
tmp_junc.push([tmp + 2, tmp + len - 1]);
|
||||
tmp += len;
|
||||
} else if (op == 'M' || op == 'X' || op == '=' || op == 'D') {
|
||||
tmp += len * 3;
|
||||
} else if (op == 'F' || op == 'G') {
|
||||
tmp += len;
|
||||
}
|
||||
}
|
||||
if (t[4] == '+') {
|
||||
for (var i = 0; i < tmp_junc.length; ++i)
|
||||
intron.push([pos + tmp_junc[i][0], pos + tmp_junc[i][1]]);
|
||||
} else if (t[4] == '-') {
|
||||
var glen = parseInt(t[8]) - parseInt(t[7]);
|
||||
for (var i = tmp_junc.length - 1; i >= 0; --i)
|
||||
intron.push([pos + (glen - tmp_junc[i][1]), pos + (glen - tmp_junc[i][0])]);
|
||||
}
|
||||
} else {
|
||||
while ((m = re_cigar.exec(cigar)) != null) {
|
||||
var len = parseInt(m[1]), op = m[2];
|
||||
if (op == 'N') {
|
||||
intron.push([pos, pos + len]);
|
||||
pos += len;
|
||||
} else if (op == 'M' || op == 'X' || op == '=' || op == 'D') pos += len;
|
||||
}
|
||||
}
|
||||
if (intron.length == 0) {
|
||||
++n_sgl;
|
||||
@@ -2509,6 +2610,276 @@ function paf_junceval(args)
|
||||
}
|
||||
}
|
||||
|
||||
function paf_exoneval(args) // adapted from paf_junceval()
|
||||
{
|
||||
var c, l_fuzzy = 0, print_ovlp = false, print_err_only = false, first_only = false, chr_only = false, aa = false, is_bed = false, use_cds = false, eval_base = false;
|
||||
while ((c = getopt(args, "l:epcab1ds")) != null) {
|
||||
if (c == 'l') l_fuzzy = parseInt(getopt.arg);
|
||||
else if (c == 'e') print_err_only = print_ovlp = true;
|
||||
else if (c == 'p') print_ovlp = true;
|
||||
else if (c == 'c') chr_only = true;
|
||||
else if (c == 'a') aa = true, use_cds = true;
|
||||
else if (c == 'b') is_bed = true;
|
||||
else if (c == '1') first_only = true;
|
||||
else if (c == 'd') use_cds = true;
|
||||
else if (c == 's') eval_base = true;
|
||||
}
|
||||
|
||||
if (args.length - getopt.ind < 1) {
|
||||
print("Usage: paftools.js exoneval [options] <gene.gtf> <aln.sam>");
|
||||
print("Options:");
|
||||
print(" -l INT tolerance of junction positions (0 for exact) [0]");
|
||||
print(" -d evaluate coding regions only (exon regions by default)");
|
||||
print(" -a miniprot PAF as input (force -d)");
|
||||
print(" -p print overlapping exons");
|
||||
print(" -e print erroreous overlapping exons");
|
||||
print(" -c only consider alignments to /^(chr)?([0-9]+|X|Y)$/");
|
||||
print(" -1 only process the first alignment of each query");
|
||||
print(" -b BED as input");
|
||||
print(" -s compute base Sn and Sp (more memory)");
|
||||
exit(1);
|
||||
}
|
||||
|
||||
var file, buf = new Bytes();
|
||||
|
||||
warn("Reading reference GTF...");
|
||||
var tr = {};
|
||||
file = args[getopt.ind] == '-'? new File() : new File(args[getopt.ind]);
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, t = buf.toString().split("\t");
|
||||
if (t[0].charAt(0) == '#') continue;
|
||||
if (use_cds) {
|
||||
if (t[2] != "cds" && t[2] != "CDS") continue;
|
||||
} else {
|
||||
if (t[2] != 'exon') continue;
|
||||
}
|
||||
var st = parseInt(t[3]) - 1;
|
||||
var en = parseInt(t[4]);
|
||||
if ((m = /transcript_id "(\S+)"/.exec(t[8])) == null) continue;
|
||||
var tid = m[1];
|
||||
if (tr[tid] == null) tr[tid] = [t[0], t[6], 0, 0, []];
|
||||
tr[tid][4].push([st, en]); // this keeps transcript
|
||||
}
|
||||
file.close();
|
||||
|
||||
var anno = {};
|
||||
for (var tid in tr) { // traverse each transcript
|
||||
var t = tr[tid];
|
||||
Interval.sort(t[4]);
|
||||
t[2] = t[4][0][0];
|
||||
t[3] = t[4][t[4].length - 1][1];
|
||||
if (anno[t[0]] == null) anno[t[0]] = [];
|
||||
var s = t[4];
|
||||
for (var i = 0; i < s.length; ++i) // traverse each exon
|
||||
anno[t[0]].push([s[i][0], s[i][1]]);
|
||||
}
|
||||
tr = null;
|
||||
|
||||
for (var chr in anno) { // index exons
|
||||
var e = anno[chr];
|
||||
if (e.length == 0) continue;
|
||||
Interval.sort(e);
|
||||
var k = 0;
|
||||
for (var i = 1; i < e.length; ++i) // dedup
|
||||
if (e[i][0] != e[k][0] || e[i][1] != e[k][1])
|
||||
e[++k] = e[i].slice(0);
|
||||
e.length = k + 1;
|
||||
Interval.index_end(e);
|
||||
}
|
||||
|
||||
var n_pri = 0, n_unmapped = 0, n_mapped = 0;
|
||||
var n_exon = 0, n_exon_hit = 0, n_exon_novel = 0;
|
||||
|
||||
file = getopt.ind+1 >= args.length || args[getopt.ind+1] == '-'? new File() : new File(args[getopt.ind+1]);
|
||||
var last_qname = null, qexon = {};
|
||||
var re_cigar = /(\d+)([MIDNSHP=XFGUV])/g;
|
||||
|
||||
warn("Evaluating alignments...");
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, t = buf.toString().split("\t");
|
||||
var ctg_name = null, cigar = null, pos = null, qname;
|
||||
|
||||
if (t[0].charAt(0) == '@') continue;
|
||||
if (t[0] == "##PAF") t.shift();
|
||||
qname = t[0];
|
||||
if (is_bed) {
|
||||
ctg_name = t[0], pos = parseInt(t[1]), cigar == null;
|
||||
} else if (t[4] == '+' || t[4] == '-' || t[4] == '*') { // PAF
|
||||
ctg_name = t[5], pos = parseInt(t[7]);
|
||||
var type = 'P';
|
||||
for (i = 12; i < t.length; ++i) {
|
||||
if ((m = /^(tp:A|cg:Z):(\S+)/.exec(t[i])) != null) {
|
||||
if (m[1] == 'tp:A') type = m[2];
|
||||
else cigar = m[2];
|
||||
}
|
||||
}
|
||||
if (type == 'S') continue; // secondary
|
||||
} else { // SAM
|
||||
ctg_name = t[2], pos = parseInt(t[3]) - 1, cigar = t[5];
|
||||
var flag = parseInt(t[1]);
|
||||
if (flag&0x100) continue; // secondary
|
||||
}
|
||||
|
||||
if (chr_only && !/^(chr)?([0-9]+|X|Y)$/.test(ctg_name)) continue;
|
||||
if (first_only && last_qname == qname) continue;
|
||||
if (ctg_name == '*') { // unmapped
|
||||
++n_unmapped;
|
||||
continue;
|
||||
} else {
|
||||
++n_pri;
|
||||
if (last_qname != qname) {
|
||||
++n_mapped;
|
||||
last_qname = qname;
|
||||
}
|
||||
}
|
||||
|
||||
var exon = [];
|
||||
if (is_bed) { // BED
|
||||
exon.push([pos, parseInt(t[2])]);
|
||||
} else if (aa) {
|
||||
var tmp_exon = [], tmp = 0, tmp_st = 0;
|
||||
while ((m = re_cigar.exec(cigar)) != null) {
|
||||
var len = parseInt(m[1]), op = m[2];
|
||||
if (op == 'N') {
|
||||
tmp_exon.push([tmp_st, tmp]);
|
||||
tmp_st = tmp + len, tmp += len;
|
||||
} else if (op == 'U') {
|
||||
tmp_exon.push([tmp_st, tmp + 1]);
|
||||
tmp_st = tmp + len - 2, tmp += len;
|
||||
} else if (op == 'V') {
|
||||
tmp_exon.push([tmp_st, tmp + 2]);
|
||||
tmp_st = tmp + len - 1, tmp += len;
|
||||
} else if (op == 'M' || op == 'X' || op == '=' || op == 'D') {
|
||||
tmp += len * 3;
|
||||
} else if (op == 'F' || op == 'G') {
|
||||
tmp += len;
|
||||
}
|
||||
}
|
||||
tmp_exon.push([tmp_st, tmp]);
|
||||
if (t[4] == '+') {
|
||||
for (var i = 0; i < tmp_exon.length; ++i)
|
||||
exon.push([pos + tmp_exon[i][0], pos + tmp_exon[i][1]]);
|
||||
} else if (t[4] == '-') { // For protein-to-genome alignment, the coordinates are on the query strand. Need to flip them.
|
||||
var glen = parseInt(t[8]) - parseInt(t[7]);
|
||||
for (var i = tmp_exon.length - 1; i >= 0; --i)
|
||||
exon.push([pos + (glen - tmp_exon[i][1]), pos + (glen - tmp_exon[i][0])]);
|
||||
}
|
||||
} else {
|
||||
var tmp_st = pos;
|
||||
while ((m = re_cigar.exec(cigar)) != null) {
|
||||
var len = parseInt(m[1]), op = m[2];
|
||||
if (op == 'N') {
|
||||
exon.push([tmp_st, pos]);
|
||||
tmp_st = pos + len, pos += len;
|
||||
} else if (op == 'M' || op == 'X' || op == '=' || op == 'D') pos += len;
|
||||
}
|
||||
exon.push([tmp_st, pos]);
|
||||
}
|
||||
n_exon += exon.length;
|
||||
|
||||
var chr = anno[ctg_name];
|
||||
if (chr != null) {
|
||||
for (var i = 0; i < exon.length; ++i) {
|
||||
if (eval_base) {
|
||||
if (qexon[ctg_name] == null) qexon[ctg_name] = [];
|
||||
qexon[ctg_name].push([exon[i][0], exon[i][1]]);
|
||||
}
|
||||
var o = Interval.find_ovlp(chr, exon[i][0], exon[i][1]);
|
||||
if (o.length > 0) {
|
||||
var hit = false;
|
||||
for (var j = 0; j < o.length; ++j) {
|
||||
var st_diff = exon[i][0] - o[j][0];
|
||||
var en_diff = exon[i][1] - o[j][1];
|
||||
if (st_diff < 0) st_diff = -st_diff;
|
||||
if (en_diff < 0) en_diff = -en_diff;
|
||||
if (st_diff <= l_fuzzy && en_diff <= l_fuzzy)
|
||||
++n_exon_hit, hit = true;
|
||||
if (hit) break;
|
||||
}
|
||||
if (print_ovlp) {
|
||||
var type = hit? 'C' : 'P';
|
||||
if (hit && print_err_only) continue;
|
||||
var x = '[';
|
||||
for (var j = 0; j < o.length; ++j) {
|
||||
if (j) x += ', ';
|
||||
x += '(' + o[j][0] + "," + o[j][1] + ')';
|
||||
}
|
||||
x += ']';
|
||||
print(type, qname, i+1, ctg_name, exon[i][0], exon[i][1], x);
|
||||
}
|
||||
} else {
|
||||
++n_exon_novel;
|
||||
if (print_ovlp)
|
||||
print('N', qname, i+1, ctg_name, exon[i][0], exon[i][1]);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
n_exon_novel += exon.length;
|
||||
}
|
||||
}
|
||||
file.close();
|
||||
|
||||
buf.destroy();
|
||||
|
||||
if (!print_ovlp) {
|
||||
print("# unmapped reads: " + n_unmapped);
|
||||
print("# mapped reads: " + n_mapped);
|
||||
print("# primary alignments: " + n_pri);
|
||||
print("# predicted exons: " + n_exon);
|
||||
print("# non-overlapping exons: " + n_exon_novel);
|
||||
print("# correct exons: " + n_exon_hit + " (" + (n_exon_hit / n_exon * 100).toFixed(2) + "%)");
|
||||
}
|
||||
|
||||
function merge_and_index(ex) {
|
||||
for (var chr in ex) {
|
||||
var a = [];
|
||||
e = ex[chr];
|
||||
Interval.sort(e);
|
||||
var st = e[0][0], en = e[0][1];
|
||||
for (var i = 1; i < e.length; ++i) { // merge
|
||||
if (e[i][0] > en) {
|
||||
a.push([st, en]);
|
||||
st = e[i][0], en = e[i][1];
|
||||
} else {
|
||||
en = en > e[i][1]? en : e[i][1];
|
||||
}
|
||||
}
|
||||
a.push([st, en]);
|
||||
Interval.index_end(a);
|
||||
ex[chr] = a;
|
||||
}
|
||||
}
|
||||
|
||||
function cal_sn(a0, a1) {
|
||||
var tot = 0, cov = 0;
|
||||
for (var chr in a1) {
|
||||
var e0 = a0[chr], e1 = a1[chr];
|
||||
for (var i = 0; i < e1.length; ++i)
|
||||
tot += e1[i][1] - e1[i][0];
|
||||
if (e0 == null) continue;
|
||||
for (var i = 0; i < e1.length; ++i) {
|
||||
var o = Interval.find_ovlp(e0, e1[i][0], e1[i][1]);
|
||||
for (var j = 0; j < o.length; ++j) { // this only works when there are no overlaps between intervals
|
||||
var st = e1[i][0] > o[j][0]? e1[i][0] : o[j][0];
|
||||
var en = e1[i][1] < o[j][1]? e1[i][1] : o[j][1];
|
||||
cov += en - st;
|
||||
}
|
||||
}
|
||||
}
|
||||
return [tot, cov];
|
||||
}
|
||||
|
||||
if (eval_base) {
|
||||
warn("Computing base Sn and Sp...");
|
||||
merge_and_index(qexon);
|
||||
merge_and_index(anno);
|
||||
var sn = cal_sn(qexon, anno);
|
||||
var sp = cal_sn(anno, qexon);
|
||||
print("Base Sn: " + sn[1] + " / " + sn[0] + " = " + (sn[1] / sn[0] * 100).toFixed(2) + "%");
|
||||
print("Base Sp: " + sp[1] + " / " + sp[0] + " = " + (sp[1] / sp[0] * 100).toFixed(2) + "%");
|
||||
}
|
||||
}
|
||||
|
||||
// evaluate overlap sensitivity
|
||||
function paf_ov_eval(args)
|
||||
{
|
||||
@@ -2704,6 +3075,23 @@ function paf_misjoin(args)
|
||||
return len < (en - st) * cen_ratio? false : true;
|
||||
}
|
||||
|
||||
function test_cen_point(cen, chr, x) {
|
||||
var b = cen[chr];
|
||||
if (b == null) return false;
|
||||
for (var j = 0; j < b.length; ++j)
|
||||
if (x >= b[j][0] && x < b[j][1])
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
if (show_err || show_long) {
|
||||
print("C\tJ inter-chromosomal misjoin");
|
||||
print("C\tj inter-chromosomal misjoin with both breakpoints ending in centromeres");
|
||||
print("C\tG long gap on the reference genome");
|
||||
print("C\tg long gap on the reference genome with both breakpoints ending in centromeres");
|
||||
print("C\tM closed inversion");
|
||||
print("C");
|
||||
}
|
||||
function process(a) {
|
||||
var k = 0;
|
||||
for (var i = 0; i < a.length; ++i) {
|
||||
@@ -2716,14 +3104,17 @@ function paf_misjoin(args)
|
||||
a = a.sort(function(x,y){return x[2]-y[2]});
|
||||
if (show_long) for (var i = 0; i < a.length; ++i) print(a[i].join("\t"));
|
||||
for (var i = 1; i < a.length; ++i) {
|
||||
var ov = [false, false];
|
||||
var ov = [false, false], end_cen = [false, false];
|
||||
ov[0] = test_cen(cen, a[i-1][5], a[i-1][7], a[i-1][8]);
|
||||
ov[1] = test_cen(cen, a[i][5], a[i][7], a[i][8]);
|
||||
end_cen[0] = test_cen_point(cen, a[i-1][5], a[i-1][4] == '+'? a[i-1][8] : a[i-1][7]);
|
||||
end_cen[1] = test_cen_point(cen, a[i][5], a[i][4] == '+'? a[i][7] : a[i][8]);
|
||||
if (a[i-1][5] != a[i][5]) { // different chr
|
||||
if (ov[0] || ov[1]) ++n_diff[1];
|
||||
else if (show_err) {
|
||||
print("J", a[i-1].slice(0, 12).join("\t"));
|
||||
print("J", a[i].slice(0, 12).join("\t"));
|
||||
var label = end_cen[0] && end_cen[1]? 'j' : 'J';
|
||||
print(label, a[i-1].slice(0, 12).join("\t"));
|
||||
print(label, a[i].slice(0, 12).join("\t"));
|
||||
}
|
||||
++n_diff[0];
|
||||
} else if (a[i-1][4] == a[i][4]) { // a gap
|
||||
@@ -2733,8 +3124,9 @@ function paf_misjoin(args)
|
||||
if (gap > max_gap) {
|
||||
if (ov[0] || ov[1]) ++n_gap[1];
|
||||
else if (show_err) {
|
||||
print("G", a[i-1].slice(0, 12).join("\t"));
|
||||
print("G", a[i].slice(0, 12).join("\t"));
|
||||
var label = end_cen[0] && end_cen[1]? 'g' : 'G';
|
||||
print(label, a[i-1].slice(0, 12).join("\t"));
|
||||
print(label, a[i].slice(0, 12).join("\t"));
|
||||
}
|
||||
++n_gap[0];
|
||||
}
|
||||
@@ -2852,6 +3244,7 @@ function paf_sveval(args)
|
||||
if (bed != null && bed[t[0]] == null) continue;
|
||||
if (t[4] == '<INV>' || t[4] == '<INVDUP>') continue; // no inversion
|
||||
if (/[\[\]]/.test(t[4])) continue; // no break points
|
||||
if (t[6] != "." && t[6] != "PASS") continue;
|
||||
var st = parseInt(t[1]) - 1, en = st + t[3].length;
|
||||
// parse svlen
|
||||
var b = _paf_get_alen(t), svlen = b[0];
|
||||
@@ -3084,6 +3477,183 @@ function paf_pafcmp(args)
|
||||
buf.destroy();
|
||||
}
|
||||
|
||||
function paf_longcs2seq(args) {
|
||||
var c, opt = { query:false };
|
||||
while ((c = getopt(args, "q")) != null)
|
||||
if (c == 'q') opt.query = true;
|
||||
if (args.length == getopt.ind) {
|
||||
print("Usage: paftools.js longcs2seq [-q] <long-cs.paf>");
|
||||
return;
|
||||
}
|
||||
var re_cs = /([:=*+-])(\d+|[A-Za-z]+)/g
|
||||
var buf = new Bytes();
|
||||
var file = args[getopt.ind] == "-"? new File() : new File(args[getopt.ind]);
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, cs = null, t = buf.toString().split("\t");
|
||||
for (var i = 12; i < t.length; ++i)
|
||||
if ((m = /^cs:Z:(\S+)/.exec(t[i])) != null) {
|
||||
cs = m[1];
|
||||
break;
|
||||
}
|
||||
if (cs == null) continue;
|
||||
var ts = "", qs = "";
|
||||
while ((m = re_cs.exec(cs)) != null) {
|
||||
if (m[1] == "=") ts += m[2], qs += m[2];
|
||||
else if (m[1] == "+") qs += m[2].toUpperCase();
|
||||
else if (m[1] == "-") ts += m[2].toUpperCase();
|
||||
else if (m[1] == "*") ts += m[2][0].toUpperCase(), qs += m[2][1].toUpperCase();
|
||||
else if (m[1] == ":") throw Error("Long cs is required");
|
||||
}
|
||||
if (opt.query) {
|
||||
print(">" + t[0] + "_" + t[2] + "_" + t[3]);
|
||||
print(qs);
|
||||
} else {
|
||||
print(">" + t[5] + "_" + t[7] + "_" + t[8]);
|
||||
print(ts);
|
||||
}
|
||||
}
|
||||
file.close();
|
||||
buf.destroy();
|
||||
}
|
||||
|
||||
function paf_paf2gff(args) {
|
||||
var c, opt = { aa:false };
|
||||
var re_cigar = /(\d+)([A-Z=])/g;
|
||||
while ((c = getopt(args, "a")) != null) {
|
||||
if (c == 'a') opt.aa = true;
|
||||
}
|
||||
if (args.length == getopt.ind) {
|
||||
print("Usage: paftools.js paf2gff [-a] <in.paf>");
|
||||
return;
|
||||
}
|
||||
var buf = new Bytes();
|
||||
var file = args[getopt.ind] == '-'? new File() : new File(args[getopt.ind]);
|
||||
var hid = 1, last_name = null;
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, t = buf.toString().split("\t");
|
||||
if (t[5] == '*') continue; // skip unmapped lines
|
||||
|
||||
if (t[0] != last_name) last_name = t[0], hid = 1;
|
||||
else ++hid;
|
||||
for (var i = 1; i <= 3; ++i) t[i] = parseInt(t[i]);
|
||||
for (var i = 6; i <= 11; ++i) t[i] = parseInt(t[i]);
|
||||
var cigar = null, score = null, np = null, dist_stop = null, dist_start = null;
|
||||
for (var i = 12; i < t.length; ++i) {
|
||||
if ((m = /^(cg:Z|AS:i|np:i|da:i|do:i):(\S+)/.exec(t[i])) != null) {
|
||||
if (m[1] == 'cg:Z') cigar = m[2];
|
||||
else if (m[1] == 'AS:i') score = parseInt(m[2]);
|
||||
else if (m[1] == 'np:i') np = parseInt(m[2]);
|
||||
else if (m[1] == 'do:i') dist_stop = parseInt(m[2]);
|
||||
else if (m[1] == 'da:i') dist_start = parseInt(m[2]);
|
||||
}
|
||||
}
|
||||
if (cigar == null) throw Error("failed to find the cg:Z tag");
|
||||
if (score == null) throw Error("failed to find the AS:i tag");
|
||||
|
||||
var st = 0, en = 0, phase = 0, pseudo = false, fs = 0, a = [];
|
||||
if (dist_start != null && dist_start == 0)
|
||||
a.push([t[5], 'paf2gff', 'start_codon', 0, 3, 0, t[4], '.', 0]);
|
||||
while ((m = re_cigar.exec(cigar)) != null) {
|
||||
var len = parseInt(m[1]);
|
||||
if (m[2] == 'M' || m[2] == 'D') {
|
||||
en += opt.aa? len * 3 : len;
|
||||
} else if (m[2] == 'F' || m[2] == 'G' || m[2] == 'R') {
|
||||
en += len, pseudo = true, fs = 1;
|
||||
} else if (m[2] == 'N') {
|
||||
a.push([t[5], 'paf2gff', 'exon', st, en, 0, t[4], phase, fs]);
|
||||
st = en + len, en += len, phase = 0, fs = 0;
|
||||
} else if (m[2] == 'U') { // ...xGT...AGxx...
|
||||
a.push([t[5], 'paf2gff', 'exon', st, en + 1, 0, t[4], phase, fs]);
|
||||
st = en + len - 2, en += len, phase = 2, fs = 0;
|
||||
} else if (m[2] == 'V') { // ...xxGT...AGx...
|
||||
a.push([t[5], 'paf2gff', 'exon', st, en + 2, 0, t[4], phase, fs]);
|
||||
st = en + len - 1, en += len, phase = 1, fs = 0;
|
||||
}
|
||||
}
|
||||
a.push([t[5], 'paf2gff', 'exon', st, en, 0, t[4], phase, fs]);
|
||||
if (en != t[8] - t[7]) throw Error("inconsistent cigar");
|
||||
if (dist_stop != null && dist_stop == 0)
|
||||
a.push([t[5], 'paf2gff', 'stop_codon', en, en + 3, 0, t[4], '.', 0]);
|
||||
var type = pseudo? 'pseudogene' : 'protein_coding';
|
||||
var attr = ['transcript_id=' + t[0] + '#' + hid, 'transcript_type=' + type].join(";");
|
||||
var trans_attr = 'identity=' + (t[9] / t[10]).toFixed(4);
|
||||
if (np != null) trans_attr += ';positive=' + (np * 3 / t[10]).toFixed(4);
|
||||
trans_attr += ';aa_start=' + t[2];
|
||||
trans_attr += ';aa_end=' + (t[1] - t[3]);
|
||||
if (dist_start != null && dist_start >= 0) trans_attr += ';dist_start_codon=' + dist_start;
|
||||
if (dist_stop != null && dist_stop >= 0) trans_attr += ';dist_stop_codon=' + dist_stop;
|
||||
var trans_st = t[7], trans_en = t[8];
|
||||
if (dist_stop != null && dist_stop == 0) {
|
||||
if (t[4] == '-') trans_st -= 3;
|
||||
else trans_en += 3;
|
||||
}
|
||||
print([t[5], 'paf2gff', 'transcript', trans_st + 1, trans_en, score, t[4], '.', attr + ';' + trans_attr].join("\t"));
|
||||
if (opt.aa && t[4] == '-') {
|
||||
var b = [], len = t[8] - t[7];
|
||||
for (var i = a.length - 1; i >= 0; --i) {
|
||||
var x = len - a[i][3];
|
||||
a[i][3] = len - a[i][4];
|
||||
a[i][4] = x;
|
||||
//a[i][7] = a[i][7] == 0? 0 : 3 - a[i][7]; // not sure if this line is needed
|
||||
b.push(a[i]);
|
||||
}
|
||||
a = b;
|
||||
}
|
||||
for (var i = 0; i < a.length; ++i) {
|
||||
if (!pseudo && a[i][2] == "exon") a[i][2] = "CDS";
|
||||
a[i][3] += t[7] + 1;
|
||||
a[i][4] += t[7];
|
||||
a[i][8] = attr + ";frameshift=" + a[i][8];
|
||||
print(a[i].join("\t"));
|
||||
}
|
||||
}
|
||||
file.close();
|
||||
buf.destroy();
|
||||
}
|
||||
|
||||
function paf_gff2junc(args) {
|
||||
var c, feat = "CDS";
|
||||
while ((c = getopt(args, "f:")) != null) {
|
||||
if (c == 'f') feat = getopt.arg;
|
||||
}
|
||||
if (getopt.ind == args.length) {
|
||||
print("Usage: paftools.js gff2junc [-f feature] <in.gff3>");
|
||||
return;
|
||||
}
|
||||
var buf = new Bytes();
|
||||
var file = args[getopt.ind] == "-"? new File() : new File(args[getopt.ind]);
|
||||
|
||||
function process_a(a) {
|
||||
if (a.length < 2) return;
|
||||
a = a.sort(function(x, y) { return x[4] - y[4] });
|
||||
for (var i = 1; i < a.length; ++i)
|
||||
print([a[i][1], a[i-1][5], a[i][4], a[i][0], 0, a[i][7]].join("\t"));
|
||||
}
|
||||
|
||||
var a = [];
|
||||
while (file.readline(buf) >= 0) {
|
||||
var m, t = buf.toString().split("\t");
|
||||
if (t[0][0] == '#') continue;
|
||||
if (t[2].toLowerCase() != feat.toLowerCase()) continue;
|
||||
//print(t.join("\t"));
|
||||
if ((m = /\bParent=([^;]+)/.exec(t[8])) == null) {
|
||||
warn("Can't find Parent");
|
||||
continue;
|
||||
}
|
||||
t[3] = parseInt(t[3]) - 1;
|
||||
t[4] = parseInt(t[4]);
|
||||
t.unshift(m[1]);
|
||||
if (a.length > 0 && a[0][0] != m[1]) {
|
||||
process_a(a);
|
||||
a.length = 0;
|
||||
a.push(t);
|
||||
} else a.push(t);
|
||||
}
|
||||
process_a(a);
|
||||
file.close();
|
||||
buf.destroy();
|
||||
}
|
||||
|
||||
/*************************
|
||||
***** main function *****
|
||||
*************************/
|
||||
@@ -3098,6 +3668,9 @@ function main(args)
|
||||
print(" sam2paf convert SAM to PAF");
|
||||
print(" delta2paf convert MUMmer's delta to PAF");
|
||||
print(" gff2bed convert GTF/GFF3 to BED12");
|
||||
print(" gff2junc convert GFF3 to junction BED");
|
||||
print(" longcs2seq convert long-cs PAF to sequences");
|
||||
// print(" paf2gff convert PAF to GFF3 (tested for miniprot only)");
|
||||
print("");
|
||||
print(" stat collect basic mapping information in PAF/SAM");
|
||||
print(" asmstat collect basic assembly information");
|
||||
@@ -3115,6 +3688,7 @@ function main(args)
|
||||
print(" mason2fq convert mason2-simulated SAM to FASTQ");
|
||||
print(" pbsim2fq convert PBSIM-simulated MAF to FASTQ");
|
||||
print(" junceval evaluate splice junction consistency with known annotations");
|
||||
print(" exoneval evaluate exon-level consistency with known annotations");
|
||||
print(" ov-eval evaluate read overlap sensitivity using read-to-ref mapping");
|
||||
exit(1);
|
||||
}
|
||||
@@ -3125,6 +3699,7 @@ function main(args)
|
||||
else if (cmd == 'delta2paf') paf_delta2paf(args);
|
||||
else if (cmd == 'splice2bed') paf_splice2bed(args);
|
||||
else if (cmd == 'gff2bed') paf_gff2bed(args);
|
||||
else if (cmd == 'gff2junc') paf_gff2junc(args);
|
||||
else if (cmd == 'stat') paf_stat(args);
|
||||
else if (cmd == 'asmstat') paf_asmstat(args);
|
||||
else if (cmd == 'asmgene') paf_asmgene(args);
|
||||
@@ -3138,10 +3713,13 @@ function main(args)
|
||||
else if (cmd == 'mason2fq') paf_mason2fq(args);
|
||||
else if (cmd == 'pbsim2fq') paf_pbsim2fq(args);
|
||||
else if (cmd == 'junceval') paf_junceval(args);
|
||||
else if (cmd == 'exoneval') paf_exoneval(args);
|
||||
else if (cmd == 'ov-eval') paf_ov_eval(args);
|
||||
else if (cmd == 'vcfstat') paf_vcfstat(args);
|
||||
else if (cmd == 'sveval') paf_sveval(args);
|
||||
else if (cmd == 'vcfsel') paf_vcfsel(args);
|
||||
else if (cmd == 'longcs2seq') paf_longcs2seq(args);
|
||||
else if (cmd == 'paf2gff') paf_paf2gff(args);
|
||||
else if (cmd == 'version') print(paftools_version);
|
||||
else throw Error("unrecognized command: " + cmd);
|
||||
}
|
||||
|
||||
@@ -13,6 +13,8 @@
|
||||
#define MM_DBG_PRINT_QNAME 0x2
|
||||
#define MM_DBG_PRINT_SEED 0x4
|
||||
#define MM_DBG_PRINT_ALN_SEQ 0x8
|
||||
#define MM_DBG_PRINT_CHAIN 0x10
|
||||
#define MM_DBG_SEED_FREQ 0x20
|
||||
|
||||
#define MM_SEED_LONG_JOIN (1ULL<<40)
|
||||
#define MM_SEED_IGNORE (1ULL<<41)
|
||||
@@ -22,6 +24,9 @@
|
||||
#define MM_SEED_SEG_SHIFT 48
|
||||
#define MM_SEED_SEG_MASK (0xffULL<<(MM_SEED_SEG_SHIFT))
|
||||
|
||||
#define MM_JUNC_ANNO 0x1
|
||||
#define MM_JUNC_MISC 0x2
|
||||
|
||||
#ifndef kroundup32
|
||||
#define kroundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
||||
#endif
|
||||
@@ -31,6 +36,7 @@
|
||||
|
||||
#define MALLOC(type, len) ((type*)malloc((len) * sizeof(type)))
|
||||
#define CALLOC(type, len) ((type*)calloc((len), sizeof(type)))
|
||||
#define REALLOC(type, ptr, cnt) ((type*)realloc((ptr), (cnt) * sizeof(type)))
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
@@ -50,6 +56,12 @@ typedef struct {
|
||||
mm128_t *a;
|
||||
} mm_seg_t;
|
||||
|
||||
typedef struct {
|
||||
int32_t off, off2, cnt;
|
||||
int16_t strand;
|
||||
uint16_t flag;
|
||||
} mm_idx_jjump1_t;
|
||||
|
||||
double cputime(void);
|
||||
double realtime(void);
|
||||
long peakrss(void);
|
||||
@@ -67,19 +79,23 @@ double mm_event_identity(const mm_reg1_t *r);
|
||||
int mm_write_sam_hdr(const mm_idx_t *mi, const char *rg, const char *ver, int argc, char *argv[]);
|
||||
void mm_write_paf(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag);
|
||||
void mm_write_paf3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len);
|
||||
void mm_write_paf4(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, void *km, int64_t opt_flag, int rep_len, int n_seg, int seg_idx);
|
||||
void mm_write_sam(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r, int n_regs, const mm_reg1_t *regs);
|
||||
void mm_write_sam2(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regs, const mm_reg1_t *const* regs, void *km, int64_t opt_flag);
|
||||
void mm_write_sam3(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, int seg_idx, int reg_idx, int n_seg, const int *n_regss, const mm_reg1_t *const* regss, void *km, int64_t opt_flag, int rep_len);
|
||||
void mm_write_junc(kstring_t *s, const mm_idx_t *mi, const mm_bseq1_t *t, const mm_reg1_t *r);
|
||||
|
||||
// indexing related in index.c
|
||||
void mm_idxopt_init(mm_idxopt_t *opt);
|
||||
const uint64_t *mm_idx_get(const mm_idx_t *mi, uint64_t minier, int *n);
|
||||
int32_t mm_idx_cal_max_occ(const mm_idx_t *mi, float f);
|
||||
int mm_idx_getseq2(const mm_idx_t *mi, int is_rev, uint32_t rid, uint32_t st, uint32_t en, uint8_t *seq);
|
||||
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a);
|
||||
mm_reg1_t *mm_gen_regs(void *km, uint32_t hash, int qlen, int n_u, uint64_t *u, mm128_t *a, int is_qstrand);
|
||||
int mm_idx_bed_read(mm_idx_t *mi, const char *fn, int read_junc);
|
||||
int mm_idx_jjump_read(mm_idx_t *mi, const char *fn, int flag, int min_sc);
|
||||
const mm_idx_jjump1_t *mm_idx_jump_get(const mm_idx_t *db, int32_t cid, int32_t st, int32_t en, int32_t *n);
|
||||
|
||||
mm128_t *mm_chain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float gap_scale,
|
||||
int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
||||
// chaining in lchain.c
|
||||
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
int is_cdna, int n_segs, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km);
|
||||
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
@@ -96,8 +112,12 @@ void mm_select_sub_multi(void *km, float pri_ratio, float pri1, float pri2, int
|
||||
int mm_filter_strand_retained(int n_regs, mm_reg1_t *r);
|
||||
void mm_filter_regs(const mm_mapopt_t *opt, int qlen, int *n_regs, mm_reg1_t *regs);
|
||||
void mm_hit_sort(void *km, int *n_regs, mm_reg1_t *r, float alt_diff_frac);
|
||||
void mm_set_mapq(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr);
|
||||
void mm_set_mapq2(void *km, int n_regs, mm_reg1_t *regs, int min_chain_sc, int match_sc, int rep_len, int is_sr, int is_splice);
|
||||
void mm_update_dp_max(int qlen, int n_regs, mm_reg1_t *regs, float frac, int a, int b);
|
||||
void mm_jump_split(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq, mm_reg1_t *r, int32_t ts_strand);
|
||||
|
||||
mm_reg1_t *mm_align_skeleton(void *km, const mm_mapopt_t *opt, const mm_idx_t *mi, int qlen, const char *qstr, int *n_regs_, mm_reg1_t *regs, mm128_t *a);
|
||||
void mm_enlarge_cigar(mm_reg1_t *r, uint32_t n_cigar);
|
||||
|
||||
void mm_est_err(const mm_idx_t *mi, int qlen, int n_regs, mm_reg1_t *regs, const mm128_t *a, int32_t n, const uint64_t *mini_pos);
|
||||
|
||||
@@ -105,6 +125,8 @@ mm_seg_t *mm_seg_gen(void *km, uint32_t hash, int n_segs, const int *qlens, int
|
||||
void mm_seg_free(void *km, int n_segs, mm_seg_t *segs);
|
||||
void mm_pair(void *km, int max_gap_ref, int dp_bonus, int sub_diff, int match_sc, const int *qlens, int *n_regs, mm_reg1_t **regs);
|
||||
|
||||
void mm_jump_split(void *km, const mm_idx_t *mi, const mm_mapopt_t *opt, int32_t qlen, const uint8_t *qseq, mm_reg1_t *r, int32_t ts_strand);
|
||||
|
||||
FILE *mm_split_init(const char *prefix, const mm_idx_t *mi);
|
||||
mm_idx_t *mm_split_merge_prep(const char *prefix, int n_splits, FILE **fp, uint32_t *n_seq_part);
|
||||
int mm_split_merge(int n_segs, const char **fn, const mm_mapopt_t *opt, int n_split_idx);
|
||||
|
||||
@@ -8,7 +8,7 @@ void mm_idxopt_init(mm_idxopt_t *opt)
|
||||
opt->k = 15, opt->w = 10, opt->flag = 0;
|
||||
opt->bucket_bits = 14;
|
||||
opt->mini_batch_size = 50000000;
|
||||
opt->batch_size = 4000000000ULL;
|
||||
opt->batch_size = 8000000000ULL;
|
||||
}
|
||||
|
||||
void mm_mapopt_init(mm_mapopt_t *opt)
|
||||
@@ -45,6 +45,7 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
||||
opt->alt_drop = 0.15f;
|
||||
|
||||
opt->a = 2, opt->b = 4, opt->q = 4, opt->e = 2, opt->q2 = 24, opt->e2 = 1;
|
||||
opt->transition = 0;
|
||||
opt->sc_ambi = 1;
|
||||
opt->zdrop = 400, opt->zdrop_inv = 200;
|
||||
opt->end_bonus = -1;
|
||||
@@ -54,13 +55,15 @@ void mm_mapopt_init(mm_mapopt_t *opt)
|
||||
opt->max_clip_ratio = 1.0f;
|
||||
opt->mini_batch_size = 500000000;
|
||||
opt->max_sw_mat = 100000000;
|
||||
opt->cap_kalloc = 1000000000;
|
||||
opt->cap_kalloc = 500000000;
|
||||
|
||||
opt->rank_min_len = 500;
|
||||
opt->rank_frac = 0.9f;
|
||||
|
||||
opt->pe_ori = 0; // FF
|
||||
opt->pe_bonus = 33;
|
||||
|
||||
opt->jump_min_match = 3;
|
||||
}
|
||||
|
||||
void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
||||
@@ -74,6 +77,7 @@ void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
||||
if (opt->max_mid_occ > opt->min_mid_occ && opt->mid_occ > opt->max_mid_occ)
|
||||
opt->mid_occ = opt->max_mid_occ;
|
||||
}
|
||||
if (opt->bw_long < opt->bw) opt->bw_long = opt->bw;
|
||||
if (mm_verbose >= 3)
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] mid_occ = %d\n", __func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), opt->mid_occ);
|
||||
}
|
||||
@@ -89,7 +93,7 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
if (preset == 0) {
|
||||
mm_idxopt_init(io);
|
||||
mm_mapopt_init(mo);
|
||||
} else if (strcmp(preset, "map-ont") == 0) { // this is the same as the default
|
||||
} else if (strcmp(preset, "lr") == 0 || strcmp(preset, "map-ont") == 0) { // this is the same as the default
|
||||
} else if (strcmp(preset, "ava-ont") == 0) {
|
||||
io->flag = 0, io->k = 15, io->w = 5;
|
||||
mo->flag |= MM_F_ALL_CHAINS | MM_F_NO_DIAG | MM_F_NO_DUAL | MM_F_NO_LJOIN;
|
||||
@@ -104,16 +108,33 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
mo->min_chain_score = 100, mo->pri_ratio = 0.0f, mo->max_chain_skip = 25;
|
||||
mo->bw_long = mo->bw;
|
||||
mo->occ_dist = 0;
|
||||
} else if (strcmp(preset, "map-hifi") == 0 || strcmp(preset, "map-ccs") == 0) {
|
||||
} else if (strcmp(preset, "lr:hq") == 0 || strcmp(preset, "map-hifi") == 0 || strcmp(preset, "map-ccs") == 0) {
|
||||
io->flag = 0, io->k = 19, io->w = 19;
|
||||
mo->max_gap = 10000;
|
||||
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1;
|
||||
mo->occ_dist = 500;
|
||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
||||
mo->min_dp_max = 200;
|
||||
if (strcmp(preset, "map-hifi") == 0 || strcmp(preset, "map-ccs") == 0) {
|
||||
mo->a = 1, mo->b = 4, mo->q = 6, mo->q2 = 26, mo->e = 2, mo->e2 = 1;
|
||||
mo->min_dp_max = 200;
|
||||
}
|
||||
} else if (strcmp(preset, "lr:hqae") == 0) { // high-quality assembly evaluation
|
||||
io->flag = 0, io->k = 25, io->w = 51;
|
||||
mo->flag |= MM_F_RMQ;
|
||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
||||
mo->rmq_inner_dist = 5000;
|
||||
mo->occ_dist = 200;
|
||||
mo->best_n = 100;
|
||||
mo->chain_gap_scale = 5.0f;
|
||||
} else if (strcmp(preset, "map-iclr-prerender") == 0) {
|
||||
io->flag = 0, io->k = 15;
|
||||
mo->b = 6, mo->transition = 1;
|
||||
mo->q = 10, mo->q2 = 50;
|
||||
} else if (strcmp(preset, "map-iclr") == 0) {
|
||||
io->flag = 0, io->k = 19;
|
||||
mo->b = 6, mo->transition = 4;
|
||||
mo->q = 10, mo->q2 = 50;
|
||||
} else if (strncmp(preset, "asm", 3) == 0) {
|
||||
io->flag = 0, io->k = 19, io->w = 19;
|
||||
mo->bw = mo->bw_long = 100000;
|
||||
mo->bw = 1000, mo->bw_long = 100000;
|
||||
mo->max_gap = 10000;
|
||||
mo->flag |= MM_F_RMQ;
|
||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
||||
@@ -145,7 +166,7 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
mo->mid_occ = 1000;
|
||||
mo->max_occ = 5000;
|
||||
mo->mini_batch_size = 50000000;
|
||||
} else if (strncmp(preset, "splice", 6) == 0 || strcmp(preset, "cdna") == 0) {
|
||||
} else if (strcmp(preset, "splice") == 0 || strcmp(preset, "splice:hq") == 0 || strcmp(preset, "splice:sr") == 0 || strcmp(preset, "cdna") == 0) {
|
||||
io->flag = 0, io->k = 15, io->w = 5;
|
||||
mo->flag |= MM_F_SPLICE | MM_F_SPLICE_FOR | MM_F_SPLICE_REV | MM_F_SPLICE_FLANK;
|
||||
mo->max_sw_mat = 0;
|
||||
@@ -153,13 +174,31 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
mo->a = 1, mo->b = 2, mo->q = 2, mo->e = 1, mo->q2 = 32, mo->e2 = 0;
|
||||
mo->noncan = 9;
|
||||
mo->junc_bonus = 9;
|
||||
mo->junc_pen = 5;
|
||||
mo->zdrop = 200, mo->zdrop_inv = 100; // because mo->a is halved
|
||||
if (strcmp(preset, "splice:hq") == 0)
|
||||
mo->junc_bonus = 5, mo->b = 4, mo->q = 6, mo->q2 = 24;
|
||||
if (strcmp(preset, "splice:hq") == 0) {
|
||||
mo->noncan = 5, mo->b = 4, mo->q = 6, mo->q2 = 24;
|
||||
} else if (strcmp(preset, "splice:sr") == 0) {
|
||||
mo->flag |= MM_F_NO_PRINT_2ND | MM_F_2_IO_THREADS | MM_F_HEAP_SORT | MM_F_FRAG_MODE | MM_F_WEAK_PAIRING | MM_F_SR_RNA;
|
||||
mo->noncan = 5, mo->b = 4, mo->q = 6, mo->q2 = 24;
|
||||
mo->min_chain_score = 25;
|
||||
mo->min_dp_max = 40;
|
||||
mo->min_ksw_len = 20;
|
||||
mo->pe_ori = 0<<1|1; // FR
|
||||
mo->best_n = 10;
|
||||
mo->mini_batch_size = 100000000;
|
||||
}
|
||||
} else return -1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int mm_max_spsc_bonus(const mm_mapopt_t *mo)
|
||||
{
|
||||
int max_sc = (mo->q2 + 1) / 2 - 1;
|
||||
max_sc = max_sc > mo->q2 - mo->q? max_sc : mo->q2 - mo->q;
|
||||
return max_sc;
|
||||
}
|
||||
|
||||
int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
||||
{
|
||||
if (mo->bw > mo->bw_long) {
|
||||
@@ -214,6 +253,11 @@ int mm_check_opt(const mm_idxopt_t *io, const mm_mapopt_t *mo)
|
||||
fprintf(stderr, "[ERROR]\033[1;31m scoring system violating ({-O}+{-E})+({-O2}+{-E2}) <= 127\033[0m\n");
|
||||
return -1;
|
||||
}
|
||||
if (mo->sc_ambi < 0 || mo->sc_ambi >= mo->b) {
|
||||
if (mm_verbose >= 1)
|
||||
fprintf(stderr, "[ERROR]\033[1;31m --score-N should be within [0,{-B})\033[0m\n");
|
||||
return -1;
|
||||
}
|
||||
if (mo->zdrop < mo->zdrop_inv) {
|
||||
if (mm_verbose >= 1)
|
||||
fprintf(stderr, "[ERROR]\033[1;31m Z-drop should not be less than inversion-Z-drop\033[0m\n");
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
[build-system]
|
||||
requires = ["setuptools", "wheel", "Cython"]
|
||||
+3
-1
@@ -77,7 +77,9 @@ This constructor accepts the following arguments:
|
||||
|
||||
* **min_chain_score**: minimum chaing score
|
||||
|
||||
* **bw**: chaining and alignment band width
|
||||
* **bw**: chaining and alignment band width (initial chaining and extension)
|
||||
|
||||
* **bw_long**: chaining and alignment band width (RMQ-based rechaining and closing gaps)
|
||||
|
||||
* **best_n**: max number of alignments to return
|
||||
|
||||
|
||||
+3
-3
@@ -71,13 +71,13 @@ static inline void mm_reset_timer(void)
|
||||
}
|
||||
|
||||
extern unsigned char seq_comp_table[256];
|
||||
static inline mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char *seq1, const char *seq2, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt)
|
||||
static inline mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char* seqname, const char *seq1, const char *seq2, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt)
|
||||
{
|
||||
mm_reg1_t *r;
|
||||
|
||||
Py_BEGIN_ALLOW_THREADS
|
||||
if (seq2 == 0) {
|
||||
r = mm_map(mi, strlen(seq1), seq1, n_regs, b, opt, NULL);
|
||||
r = mm_map(mi, strlen(seq1), seq1, n_regs, b, opt, seqname);
|
||||
} else {
|
||||
int _n_regs[2];
|
||||
mm_reg1_t *regs[2];
|
||||
@@ -94,7 +94,7 @@ static inline mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char *seq1, const
|
||||
seq[1][i] = seq_comp_table[t];
|
||||
}
|
||||
if (len[1]&1) seq[1][len[1]>>1] = seq_comp_table[(uint8_t)seq[1][len[1]>>1]];
|
||||
mm_map_frag(mi, 2, len, (const char**)seq, _n_regs, regs, b, opt, NULL);
|
||||
mm_map_frag(mi, 2, len, (const char**)seq, _n_regs, regs, b, opt, seqname);
|
||||
for (i = 0; i < _n_regs[1]; ++i)
|
||||
regs[1][i].rev = !regs[1][i].rev;
|
||||
*n_regs = _n_regs[0] + _n_regs[1];
|
||||
|
||||
+5
-2
@@ -36,9 +36,10 @@ cdef extern from "minimap.h":
|
||||
float alt_drop
|
||||
|
||||
int a, b, q, e, q2, e2
|
||||
int transition
|
||||
int sc_ambi
|
||||
int noncan
|
||||
int junc_bonus
|
||||
int junc_bonus, junc_pen
|
||||
int zdrop, zdrop_inv
|
||||
int end_bonus
|
||||
int min_dp_max
|
||||
@@ -51,6 +52,8 @@ cdef extern from "minimap.h":
|
||||
|
||||
int pe_ori, pe_bonus
|
||||
|
||||
int jump_min_match;
|
||||
|
||||
float mid_occ_frac
|
||||
float q_occ_frac
|
||||
int32_t min_mid_occ
|
||||
@@ -128,7 +131,7 @@ cdef extern from "cmappy.h":
|
||||
|
||||
void mm_reg2hitpy(const mm_idx_t *mi, mm_reg1_t *r, mm_hitpy_t *h)
|
||||
void mm_free_reg1(mm_reg1_t *r)
|
||||
mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char *seq1, const char *seq2, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt)
|
||||
mm_reg1_t *mm_map_aux(const mm_idx_t *mi, const char* seqname, const char *seq1, const char *seq2, int *n_regs, mm_tbuf_t *b, const mm_mapopt_t *opt)
|
||||
char *mappy_fetch_seq(const mm_idx_t *mi, const char *name, int st, int en, int *l)
|
||||
mm_idx_t *mappy_idx_seq(int w, int k, int is_hpc, int bucket_bits, const char *seq, int l)
|
||||
|
||||
|
||||
+23
-7
@@ -3,7 +3,7 @@ from libc.stdlib cimport free
|
||||
cimport cmappy
|
||||
import sys
|
||||
|
||||
__version__ = '2.23'
|
||||
__version__ = '2.29'
|
||||
|
||||
cmappy.mm_reset_timer()
|
||||
|
||||
@@ -96,6 +96,7 @@ cdef class Alignment:
|
||||
a = [str(self._q_st), str(self._q_en), strand, self._ctg, str(self._ctg_len), str(self._r_st), str(self._r_en),
|
||||
str(self._mlen), str(self._blen), str(self._mapq), tp, ts, "cg:Z:" + self.cigar_str]
|
||||
if self._cs != "": a.append("cs:Z:" + self._cs)
|
||||
if self._MD != "": a.append("MD:Z:" + self._MD)
|
||||
return "\t".join(a)
|
||||
|
||||
cdef class ThreadBuffer:
|
||||
@@ -112,7 +113,7 @@ cdef class Aligner:
|
||||
cdef cmappy.mm_idxopt_t idx_opt
|
||||
cdef cmappy.mm_mapopt_t map_opt
|
||||
|
||||
def __cinit__(self, fn_idx_in=None, preset=None, k=None, w=None, min_cnt=None, min_chain_score=None, min_dp_score=None, bw=None, best_n=None, n_threads=3, fn_idx_out=None, max_frag_len=None, extra_flags=None, seq=None, scoring=None):
|
||||
def __cinit__(self, fn_idx_in=None, preset=None, k=None, w=None, min_cnt=None, min_chain_score=None, min_dp_score=None, bw=None, bw_long=None, best_n=None, n_threads=3, fn_idx_out=None, max_frag_len=None, extra_flags=None, seq=None, scoring=None, sc_ambi=None, max_chain_skip=None):
|
||||
self._idx = NULL
|
||||
cmappy.mm_set_opt(NULL, &self.idx_opt, &self.map_opt) # set the default options
|
||||
if preset is not None:
|
||||
@@ -125,6 +126,7 @@ cdef class Aligner:
|
||||
if min_chain_score is not None: self.map_opt.min_chain_score = min_chain_score
|
||||
if min_dp_score is not None: self.map_opt.min_dp_max = min_dp_score
|
||||
if bw is not None: self.map_opt.bw = bw
|
||||
if bw_long is not None: self.map_opt.bw_long = bw_long
|
||||
if best_n is not None: self.map_opt.best_n = best_n
|
||||
if max_frag_len is not None: self.map_opt.max_frag_len = max_frag_len
|
||||
if extra_flags is not None: self.map_opt.flag |= extra_flags
|
||||
@@ -136,6 +138,8 @@ cdef class Aligner:
|
||||
self.map_opt.q2, self.map_opt.e2 = scoring[4], scoring[5]
|
||||
if len(scoring) >= 7:
|
||||
self.map_opt.sc_ambi = scoring[6]
|
||||
if sc_ambi is not None: self.map_opt.sc_ambi = sc_ambi
|
||||
if max_chain_skip is not None: self.map_opt.max_chain_skip = max_chain_skip
|
||||
|
||||
cdef cmappy.mm_idx_reader_t *r;
|
||||
|
||||
@@ -161,7 +165,7 @@ cdef class Aligner:
|
||||
def __bool__(self):
|
||||
return (self._idx != NULL)
|
||||
|
||||
def map(self, seq, seq2=None, buf=None, cs=False, MD=False, max_frag_len=None, extra_flags=None):
|
||||
def map(self, seq, seq2=None, name=None, buf=None, cs=False, MD=False, max_frag_len=None, extra_flags=None):
|
||||
cdef cmappy.mm_reg1_t *regs
|
||||
cdef cmappy.mm_hitpy_t h
|
||||
cdef ThreadBuffer b
|
||||
@@ -172,6 +176,7 @@ cdef class Aligner:
|
||||
cdef cmappy.mm_mapopt_t map_opt
|
||||
|
||||
if self._idx == NULL: return
|
||||
if ((self.map_opt.flag & 4) and (self._idx.flag & 2)): return
|
||||
map_opt = self.map_opt
|
||||
if max_frag_len is not None: map_opt.max_frag_len = max_frag_len
|
||||
if extra_flags is not None: map_opt.flag |= extra_flags
|
||||
@@ -182,11 +187,20 @@ cdef class Aligner:
|
||||
km = cmappy.mm_tbuf_get_km(b._b)
|
||||
|
||||
_seq = seq if isinstance(seq, bytes) else seq.encode()
|
||||
if name is not None:
|
||||
_name = name if isinstance(name, bytes) else name.encode()
|
||||
|
||||
if seq2 is None:
|
||||
regs = cmappy.mm_map_aux(self._idx, _seq, NULL, &n_regs, b._b, &map_opt)
|
||||
if name is None:
|
||||
regs = cmappy.mm_map_aux(self._idx, NULL, _seq, NULL, &n_regs, b._b, &map_opt)
|
||||
else:
|
||||
regs = cmappy.mm_map_aux(self._idx, _name, _seq, NULL, &n_regs, b._b, &map_opt)
|
||||
else:
|
||||
_seq2 = seq2 if isinstance(seq2, bytes) else seq2.encode()
|
||||
regs = cmappy.mm_map_aux(self._idx, _seq, _seq2, &n_regs, b._b, &map_opt)
|
||||
if name is None:
|
||||
regs = cmappy.mm_map_aux(self._idx, NULL, _seq, _seq2, &n_regs, b._b, &map_opt)
|
||||
else:
|
||||
regs = cmappy.mm_map_aux(self._idx, _name, _seq, _seq2, &n_regs, b._b, &map_opt)
|
||||
|
||||
try:
|
||||
i = 0
|
||||
@@ -197,11 +211,12 @@ cdef class Aligner:
|
||||
c = h.cigar32[k]
|
||||
cigar.append([c>>4, c&0xf])
|
||||
if cs or MD: # generate the cs and/or the MD tag, if requested
|
||||
_cur_seq = _seq2 if h.seg_id > 0 and seq2 is not None else _seq
|
||||
if cs:
|
||||
l_cs_str = cmappy.mm_gen_cs(km, &cs_str, &m_cs_str, self._idx, ®s[i], _seq, 1)
|
||||
l_cs_str = cmappy.mm_gen_cs(km, &cs_str, &m_cs_str, self._idx, ®s[i], _cur_seq, 1)
|
||||
_cs = cs_str[:l_cs_str] if isinstance(cs_str, str) else cs_str[:l_cs_str].decode()
|
||||
if MD:
|
||||
l_cs_str = cmappy.mm_gen_MD(km, &cs_str, &m_cs_str, self._idx, ®s[i], _seq)
|
||||
l_cs_str = cmappy.mm_gen_MD(km, &cs_str, &m_cs_str, self._idx, ®s[i], _cur_seq)
|
||||
_MD = cs_str[:l_cs_str] if isinstance(cs_str, str) else cs_str[:l_cs_str].decode()
|
||||
yield Alignment(h.ctg, h.ctg_len, h.ctg_start, h.ctg_end, h.strand, h.qry_start, h.qry_end, h.mapq, cigar, h.is_primary, h.mlen, h.blen, h.NM, h.trans_strand, h.seg_id, _cs, _MD)
|
||||
cmappy.mm_free_reg1(®s[i])
|
||||
@@ -217,6 +232,7 @@ cdef class Aligner:
|
||||
cdef int l
|
||||
cdef char *s
|
||||
if self._idx == NULL: return
|
||||
if ((self.map_opt.flag & 4) and (self._idx.flag & 2)): return
|
||||
s = cmappy.mappy_fetch_seq(self._idx, name.encode(), start, end, &l)
|
||||
if l == 0: return None
|
||||
r = s[:l] if isinstance(s, str) else s[:l].decode()
|
||||
|
||||
+5
-3
@@ -5,7 +5,7 @@ import getopt
|
||||
import mappy as mp
|
||||
|
||||
def main(argv):
|
||||
opts, args = getopt.getopt(argv[1:], "x:n:m:k:w:r:c")
|
||||
opts, args = getopt.getopt(argv[1:], "x:n:m:k:w:r:cM")
|
||||
if len(args) < 2:
|
||||
print("Usage: minimap2.py [options] <ref.fa>|<ref.mmi> <query.fq>")
|
||||
print("Options:")
|
||||
@@ -16,10 +16,11 @@ def main(argv):
|
||||
print(" -w INT minimizer window length")
|
||||
print(" -r INT band width")
|
||||
print(" -c output the cs tag")
|
||||
print(" -M output the MD tag")
|
||||
sys.exit(1)
|
||||
|
||||
preset = min_cnt = min_sc = k = w = bw = None
|
||||
out_cs = False
|
||||
out_cs = out_MD = False
|
||||
for opt, arg in opts:
|
||||
if opt == '-x': preset = arg
|
||||
elif opt == '-n': min_cnt = int(arg)
|
||||
@@ -28,11 +29,12 @@ def main(argv):
|
||||
elif opt == '-k': k = int(arg)
|
||||
elif opt == '-w': w = int(arg)
|
||||
elif opt == '-c': out_cs = True
|
||||
elif opt == '-M': out_MD = True
|
||||
|
||||
a = mp.Aligner(args[0], preset=preset, min_cnt=min_cnt, min_chain_score=min_sc, k=k, w=w, bw=bw)
|
||||
if not a: raise Exception("ERROR: failed to load/build index file '{}'".format(args[0]))
|
||||
for name, seq, qual in mp.fastx_read(args[1]): # read one sequence
|
||||
for h in a.map(seq, cs=out_cs): # traverse hits
|
||||
for h in a.map(seq, cs=out_cs, MD=out_MD): # traverse hits
|
||||
print('{}\t{}\t{}'.format(name, len(seq), h))
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -7,7 +7,7 @@ void mm_seed_mz_flt(void *km, mm128_v *mv, int32_t q_occ_max, float q_occ_frac)
|
||||
mm128_t *a;
|
||||
size_t i, j, st;
|
||||
if (mv->n <= q_occ_max || q_occ_frac <= 0.0f || q_occ_max <= 0) return;
|
||||
KMALLOC(km, a, mv->n);
|
||||
a = Kmalloc(km, mm128_t, mv->n);
|
||||
for (i = 0; i < mv->n; ++i)
|
||||
a[i].x = mv->a[i].x, a[i].y = i;
|
||||
radix_sort_128x(a, a + mv->n);
|
||||
@@ -112,7 +112,8 @@ mm_seed_t *mm_collect_matches(void *km, int *_n_m, int qlen, int max_occ, int ma
|
||||
}
|
||||
for (i = 0, n_m = 0, *rep_len = 0, *n_a = 0; i < n_m0; ++i) {
|
||||
mm_seed_t *q = &m[i];
|
||||
//fprintf(stderr, "X\t%d\t%d\t%d\n", q->q_pos>>1, q->n, q->flt);
|
||||
if (mm_dbg_flag & MM_DBG_SEED_FREQ)
|
||||
fprintf(stderr, "SF\t%d\t%d\t%d\n", q->q_pos>>1, q->n, q->flt);
|
||||
if (q->flt) {
|
||||
int en = (q->q_pos >> 1) + 1, st = en - q->q_span;
|
||||
if (st > rep_en) {
|
||||
|
||||
@@ -23,7 +23,7 @@ def readme():
|
||||
|
||||
setup(
|
||||
name = 'mappy',
|
||||
version = '2.23',
|
||||
version = '2.29',
|
||||
url = 'https://github.com/lh3/minimap2',
|
||||
description = 'Minimap2 python binding',
|
||||
long_description = readme(),
|
||||
@@ -33,7 +33,7 @@ setup(
|
||||
keywords = 'sequence-alignment',
|
||||
scripts = ['python/minimap2.py'],
|
||||
ext_modules = [Extension('mappy',
|
||||
sources = ['python/mappy.pyx', 'align.c', 'bseq.c', 'lchain.c', 'seed.c', 'format.c', 'hit.c', 'index.c', 'pe.c', 'options.c',
|
||||
sources = ['python/mappy.pyx', 'align.c', 'bseq.c', 'lchain.c', 'seed.c', 'format.c', 'hit.c', 'index.c', 'pe.c', 'jump.c', 'options.c',
|
||||
'ksw2_extd2_sse.c', 'ksw2_exts2_sse.c', 'ksw2_extz2_sse.c', 'ksw2_ll_sse.c',
|
||||
'kalloc.c', 'kthread.c', 'map.c', 'misc.c', 'sdust.c', 'sketch.c', 'esterr.c', 'splitidx.c'],
|
||||
depends = ['minimap.h', 'bseq.h', 'kalloc.h', 'kdq.h', 'khash.h', 'kseq.h', 'ksort.h',
|
||||
|
||||
Reference in New Issue
Block a user