mirror of
https://github.com/lh3/minimap2.git
synced 2026-09-25 01:18:12 +08:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fe35e679e9 | ||
|
|
e25aa5ee74 | ||
|
|
3bde3450a0 | ||
|
|
36942ff711 | ||
|
|
d3a89d34d4 | ||
|
|
c8f0a35c40 | ||
|
|
fcaadc22b7 | ||
|
|
db37fc43a7 | ||
|
|
a8f1fa8ea3 | ||
|
|
b276772890 |
@@ -1,3 +1,17 @@
|
||||
Release 2.24-r1122 (26 December 2021)
|
||||
-------------------------------------
|
||||
|
||||
This release improves alignment around long poorly aligned regions. Older
|
||||
minimap2 may chain through such regions in rare cases which may result in
|
||||
missing alignments later. The issue has become worse since the the change of
|
||||
the chaining algorithm in v2.19. v2.23 implements an incomplete remedy. This
|
||||
release provides a better solution with a X-drop-like heuristic and by enabling
|
||||
two-bandwidth chaining in the assembly mode.
|
||||
|
||||
(2.24: 26 December 2021, r1122)
|
||||
|
||||
|
||||
|
||||
Release 2.23-r1111 (18 November 2021)
|
||||
-------------------------------------
|
||||
|
||||
|
||||
@@ -74,8 +74,8 @@ Detailed evaluations are available from the [minimap2 paper][doi] or the
|
||||
Minimap2 is optimized for x86-64 CPUs. You can acquire precompiled binaries from
|
||||
the [release page][release] with:
|
||||
```sh
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.23/minimap2-2.23_x64-linux.tar.bz2 | tar -jxvf -
|
||||
./minimap2-2.23_x64-linux/minimap2
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.24/minimap2-2.24_x64-linux.tar.bz2 | tar -jxvf -
|
||||
./minimap2-2.24_x64-linux/minimap2
|
||||
```
|
||||
If you want to compile from the source, you need to have a C compiler, GNU make
|
||||
and zlib development files installed. Then type `make` in the source code
|
||||
|
||||
+2
-2
@@ -31,8 +31,8 @@ To acquire the data used in this cookbook and to install minimap2 and paftools,
|
||||
please follow the command lines below:
|
||||
```sh
|
||||
# install minimap2 executables
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.23/minimap2-2.23_x64-linux.tar.bz2 | tar jxf -
|
||||
cp minimap2-2.23_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.24/minimap2-2.24_x64-linux.tar.bz2 | tar jxf -
|
||||
cp minimap2-2.24_x64-linux/{minimap2,k8,paftools.js} . # copy executables
|
||||
export PATH="$PATH:"`pwd` # put the current directory on PATH
|
||||
# download example datasets
|
||||
curl -L https://github.com/lh3/minimap2/releases/download/v2.10/cookbook-data.tgz | tar zxf -
|
||||
|
||||
@@ -279,7 +279,7 @@ int mm_filter_strand_retained(int n_regs, mm_reg1_t *r)
|
||||
int i, k;
|
||||
for (i = k = 0; i < n_regs; ++i) {
|
||||
int p = r[i].parent;
|
||||
if (!r[i].strand_retained || r[i].div < r[p].div * 5.0f) {
|
||||
if (!r[i].strand_retained || r[i].div < r[p].div * 5.0f || r[i].div < 0.01f) {
|
||||
if (k < i) r[k++] = r[i];
|
||||
else ++k;
|
||||
}
|
||||
|
||||
@@ -6,7 +6,25 @@
|
||||
#include "kalloc.h"
|
||||
#include "krmq.h"
|
||||
|
||||
uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_t *p, int32_t *v, int32_t *t, int32_t min_cnt, int32_t min_sc, int32_t *n_u_, int32_t *n_v_)
|
||||
static int64_t mg_chain_bk_end(int32_t max_drop, const mm128_t *z, const int32_t *f, const int64_t *p, int32_t *t, int64_t k)
|
||||
{
|
||||
int64_t i = z[k].y, end_i = -1, max_i = i;
|
||||
int32_t max_s = 0;
|
||||
if (i < 0 || t[i] != 0) return i;
|
||||
do {
|
||||
int32_t s;
|
||||
t[i] = 2;
|
||||
end_i = i = p[i];
|
||||
s = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (s > max_s) max_s = s, max_i = i;
|
||||
else if (max_s - s > max_drop) break;
|
||||
} while (i >= 0 && t[i] == 0);
|
||||
for (i = z[k].y; i >= 0 && i != end_i; i = p[i]) // reset modified t[]
|
||||
t[i] = 0;
|
||||
return max_i;
|
||||
}
|
||||
|
||||
uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_t *p, int32_t *v, int32_t *t, int32_t min_cnt, int32_t min_sc, int32_t max_drop, int32_t *n_u_, int32_t *n_v_)
|
||||
{
|
||||
mm128_t *z;
|
||||
uint64_t *u;
|
||||
@@ -24,26 +42,32 @@ uint64_t *mg_chain_backtrack(void *km, int64_t n, const int32_t *f, const int64_
|
||||
|
||||
memset(t, 0, n * 4);
|
||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // precompute n_u
|
||||
int64_t n_v0 = n_v;
|
||||
int32_t sc;
|
||||
for (i = z[k].y; i >= 0 && t[i] == 0; i = p[i])
|
||||
++n_v, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
++n_u;
|
||||
else n_v = n_v0;
|
||||
if (t[z[k].y] == 0) {
|
||||
int64_t n_v0 = n_v, end_i;
|
||||
int32_t sc;
|
||||
end_i = mg_chain_bk_end(max_drop, z, f, p, t, k);
|
||||
for (i = z[k].y; i != end_i; i = p[i])
|
||||
++n_v, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
++n_u;
|
||||
else n_v = n_v0;
|
||||
}
|
||||
}
|
||||
KMALLOC(km, u, n_u);
|
||||
memset(t, 0, n * 4);
|
||||
for (k = n_z - 1, n_v = n_u = 0; k >= 0; --k) { // populate u[]
|
||||
int64_t n_v0 = n_v;
|
||||
int32_t sc;
|
||||
for (i = z[k].y; i >= 0 && t[i] == 0; i = p[i])
|
||||
v[n_v++] = i, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
u[n_u++] = (uint64_t)sc << 32 | (n_v - n_v0);
|
||||
else n_v = n_v0;
|
||||
if (t[z[k].y] == 0) {
|
||||
int64_t n_v0 = n_v, end_i;
|
||||
int32_t sc;
|
||||
end_i = mg_chain_bk_end(max_drop, z, f, p, t, k);
|
||||
for (i = z[k].y; i != end_i; i = p[i])
|
||||
v[n_v++] = i, t[i] = 1;
|
||||
sc = i < 0? z[k].x : (int32_t)z[k].x - f[i];
|
||||
if (sc >= min_sc && n_v > n_v0 && n_v - n_v0 >= min_cnt)
|
||||
u[n_u++] = (uint64_t)sc << 32 | (n_v - n_v0);
|
||||
else n_v = n_v0;
|
||||
}
|
||||
}
|
||||
kfree(km, z);
|
||||
assert(n_v < INT32_MAX);
|
||||
@@ -124,7 +148,7 @@ static inline int32_t comput_sc(const mm128_t *ai, const mm128_t *aj, int32_t ma
|
||||
mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int max_iter, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
int is_cdna, int n_seg, int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||
{ // TODO: make sure this works when n has more than 32 bits
|
||||
int32_t *f, *t, *v, n_u, n_v, mmax_f = 0;
|
||||
int32_t *f, *t, *v, n_u, n_v, mmax_f = 0, max_drop = bw;
|
||||
int64_t *p, i, j, max_ii, st = 0, n_iter = 0;
|
||||
uint64_t *u;
|
||||
|
||||
@@ -135,6 +159,7 @@ mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int
|
||||
}
|
||||
if (max_dist_x < bw) max_dist_x = bw;
|
||||
if (max_dist_y < bw && !is_cdna) max_dist_y = bw;
|
||||
if (is_cdna) max_drop = INT32_MAX;
|
||||
KMALLOC(km, p, n);
|
||||
KMALLOC(km, f, n);
|
||||
KMALLOC(km, v, n);
|
||||
@@ -181,7 +206,7 @@ mm128_t *mg_lchain_dp(int max_dist_x, int max_dist_y, int bw, int max_skip, int
|
||||
if (mmax_f < max_f) mmax_f = max_f;
|
||||
}
|
||||
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, &n_u, &n_v);
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, max_drop, &n_u, &n_v);
|
||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||
kfree(km, p); kfree(km, f); kfree(km, t);
|
||||
if (n_u == 0) {
|
||||
@@ -225,7 +250,7 @@ static inline int32_t comput_sc_simple(const mm128_t *ai, const mm128_t *aj, flo
|
||||
mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_skip, int cap_rmq_size, int min_cnt, int min_sc, float chn_pen_gap, float chn_pen_skip,
|
||||
int64_t n, mm128_t *a, int *n_u_, uint64_t **_u, void *km)
|
||||
{
|
||||
int32_t *f,*t, *v, n_u, n_v, mmax_f = 0, max_rmq_size = 0;
|
||||
int32_t *f,*t, *v, n_u, n_v, mmax_f = 0, max_rmq_size = 0, max_drop = bw;
|
||||
int64_t *p, i, i0, st = 0, st_inner = 0, n_iter = 0;
|
||||
uint64_t *u;
|
||||
lc_elem_t *root = 0, *root_inner = 0;
|
||||
@@ -333,7 +358,7 @@ mm128_t *mg_lchain_rmq(int max_dist, int max_dist_inner, int bw, int max_chn_ski
|
||||
}
|
||||
km_destroy(mem_mp);
|
||||
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, &n_u, &n_v);
|
||||
u = mg_chain_backtrack(km, n, f, p, v, t, min_cnt, min_sc, max_drop, &n_u, &n_v);
|
||||
*n_u_ = n_u, *_u = u; // NB: note that u[] may not be sorted by score here
|
||||
kfree(km, p); kfree(km, f); kfree(km, t);
|
||||
if (n_u == 0) {
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
#include "mmpriv.h"
|
||||
#include "ketopt.h"
|
||||
|
||||
#define MM_VERSION "2.23-r1111"
|
||||
#define MM_VERSION "2.24-r1122"
|
||||
|
||||
#ifdef __linux__
|
||||
#include <sys/resource.h>
|
||||
@@ -76,6 +76,8 @@ static ko_longopt_t long_options[] = {
|
||||
{ "cap-kalloc", ko_required_argument, 349 },
|
||||
{ "q-occ-frac", ko_required_argument, 350 },
|
||||
{ "chain-skip-scale",ko_required_argument,351 },
|
||||
{ "print-chains", ko_no_argument, 352 },
|
||||
{ "no-hash-name", ko_no_argument, 353 },
|
||||
{ "help", ko_no_argument, 'h' },
|
||||
{ "max-intron-len", ko_required_argument, 'G' },
|
||||
{ "version", ko_no_argument, 'V' },
|
||||
@@ -233,6 +235,8 @@ int main(int argc, char *argv[])
|
||||
else if (c == 348) opt.flag |= MM_F_QSTRAND | MM_F_NO_INV; // --qstrand
|
||||
else if (c == 349) opt.cap_kalloc = mm_parse_num(o.arg); // --cap-kalloc
|
||||
else if (c == 350) opt.q_occ_frac = atof(o.arg); // --q-occ-frac
|
||||
else if (c == 352) mm_dbg_flag |= MM_DBG_PRINT_CHAIN; // --print-chains
|
||||
else if (c == 353) opt.flag |= MM_F_NO_HASH_NAME; // --no-hash-name
|
||||
else if (c == 330) {
|
||||
fprintf(stderr, "[WARNING] \033[1;31m --lj-min-ratio has been deprecated.\033[0m\n");
|
||||
} else if (c == 314) { // --frag
|
||||
|
||||
@@ -248,7 +248,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
if (qlen_sum == 0 || n_segs <= 0 || n_segs > MM_MAX_SEG) return;
|
||||
if (opt->max_qlen > 0 && qlen_sum > opt->max_qlen) return;
|
||||
|
||||
hash = qname? __ac_X31_hash_string(qname) : 0;
|
||||
hash = qname && !(opt->flag & MM_F_NO_HASH_NAME)? __ac_X31_hash_string(qname) : 0;
|
||||
hash ^= __ac_Wang_hash(qlen_sum) + __ac_Wang_hash(opt->seed);
|
||||
hash = __ac_Wang_hash(hash);
|
||||
|
||||
@@ -328,7 +328,7 @@ void mm_map_frag(const mm_idx_t *mi, int n_segs, const int *qlens, const char **
|
||||
mm_hit_sort(b->km, &n_regs0, regs0, opt->alt_drop); // this step can be merged into mm_gen_regs(); will do if this shows up in profile
|
||||
}
|
||||
|
||||
if (mm_dbg_flag & MM_DBG_PRINT_SEED)
|
||||
if (mm_dbg_flag & (MM_DBG_PRINT_SEED|MM_DBG_PRINT_CHAIN))
|
||||
for (j = 0; j < n_regs0; ++j)
|
||||
for (i = regs0[j].as; i < regs0[j].as + regs0[j].cnt; ++i)
|
||||
fprintf(stderr, "CN\t%d\t%s\t%d\t%c\t%d\t%d\t%d\n", j, mi->seq[a[i].x<<1>>33].name, (int32_t)a[i].x, "+-"[a[i].x>>63], (int32_t)a[i].y, (int32_t)(a[i].y>>32&0xff),
|
||||
|
||||
@@ -39,6 +39,7 @@
|
||||
#define MM_F_RMQ (0x80000000LL)
|
||||
#define MM_F_QSTRAND (0x100000000LL)
|
||||
#define MM_F_NO_INV (0x200000000LL)
|
||||
#define MM_F_NO_HASH_NAME (0x400000000LL)
|
||||
|
||||
#define MM_I_HPC 0x1
|
||||
#define MM_I_NO_SEQ 0x2
|
||||
|
||||
+9
-6
@@ -1,4 +1,4 @@
|
||||
.TH minimap2 1 "18 November 2021" "minimap2-2.23 (r1111)" "Bioinformatics tools"
|
||||
.TH minimap2 1 "18 December 2021" "minimap2-2.24 (r1122)" "Bioinformatics tools"
|
||||
.SH NAME
|
||||
.PP
|
||||
minimap2 - mapping and alignment between collections of DNA sequences
|
||||
@@ -77,7 +77,7 @@ SAM format.
|
||||
Minimizer k-mer length [15]
|
||||
.TP
|
||||
.BI -w \ INT
|
||||
Minimizer window size [2/3 of k-mer length]. A minimizer is the smallest k-mer
|
||||
Minimizer window size [10]. A minimizer is the smallest k-mer
|
||||
in a window of w consecutive k-mers.
|
||||
.TP
|
||||
.B -H
|
||||
@@ -318,6 +318,9 @@ faster for short reads, but slower for long reads. [no]
|
||||
.B --no-pairing
|
||||
Treat two reads in a pair as independent reads. The mate related fields in SAM
|
||||
are still properly populated.
|
||||
.TP
|
||||
.B --no-hash-name
|
||||
Produce the same alignment for identical sequences regardless of their sequence names.
|
||||
.SS Alignment options
|
||||
.TP 10
|
||||
.BI -A \ INT
|
||||
@@ -562,7 +565,7 @@ Align older PacBio continuous long (CLR) reads to a reference genome
|
||||
.B asm5
|
||||
Long assembly to reference mapping
|
||||
.RB ( -k19
|
||||
.B -w19 -U50,500 --rmq -r100k -g10k -A1 -B19 -O39,81 -E3,1 -s200 -z200
|
||||
.B -w19 -U50,500 --rmq -r1k,100k -g10k -A1 -B19 -O39,81 -E3,1 -s200 -z200
|
||||
.BR -N50 ).
|
||||
Typically, the alignment will not extend to regions with 5% or higher sequence
|
||||
divergence. Only use this preset if the average divergence is far below 5%.
|
||||
@@ -570,14 +573,14 @@ divergence. Only use this preset if the average divergence is far below 5%.
|
||||
.B asm10
|
||||
Long assembly to reference mapping
|
||||
.RB ( -k19
|
||||
.B -w19 -U50,500 --rmq -r100k -g10k -A1 -B9 -O16,41 -E2,1 -s200 -z200
|
||||
.B -w19 -U50,500 --rmq -r1k,100k -g10k -A1 -B9 -O16,41 -E2,1 -s200 -z200
|
||||
.BR -N50 ).
|
||||
Up to 10% sequence divergence.
|
||||
.TP
|
||||
.B asm20
|
||||
Long assembly to reference mapping
|
||||
.RB ( -k19
|
||||
.B -w10 -U50,500 --rmq -r100k -g10k -A1 -B4 -O6,26 -E2,1 -s200 -z200
|
||||
.B -w10 -U50,500 --rmq -r1k,100k -g10k -A1 -B4 -O6,26 -E2,1 -s200 -z200
|
||||
.BR -N50 ).
|
||||
Up to 20% sequence divergence.
|
||||
.TP
|
||||
@@ -603,7 +606,7 @@ Long-read splice alignment for PacBio CCS reads
|
||||
.B sr
|
||||
Short single-end reads without splicing
|
||||
.RB ( -k21
|
||||
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -b0 -r100 -p.5 -N20 -f1000,5000 -n2 -m20
|
||||
.B -w11 --sr --frag=yes -A2 -B8 -O12,32 -E2,1 -b0 -r100 -p.5 -N20 -f1000,5000 -n2 -m25
|
||||
.B -s40 -g100 -2K50m --heap-sort=yes
|
||||
.BR --secondary=no ).
|
||||
.TP
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env k8
|
||||
|
||||
var paftools_version = '2.23-r1111';
|
||||
var paftools_version = '2.24-r1122';
|
||||
|
||||
/*****************************
|
||||
***** Library functions *****
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#define MM_DBG_PRINT_QNAME 0x2
|
||||
#define MM_DBG_PRINT_SEED 0x4
|
||||
#define MM_DBG_PRINT_ALN_SEQ 0x8
|
||||
#define MM_DBG_PRINT_CHAIN 0x10
|
||||
|
||||
#define MM_SEED_LONG_JOIN (1ULL<<40)
|
||||
#define MM_SEED_IGNORE (1ULL<<41)
|
||||
|
||||
@@ -74,6 +74,7 @@ void mm_mapopt_update(mm_mapopt_t *opt, const mm_idx_t *mi)
|
||||
if (opt->max_mid_occ > opt->min_mid_occ && opt->mid_occ > opt->max_mid_occ)
|
||||
opt->mid_occ = opt->max_mid_occ;
|
||||
}
|
||||
if (opt->bw_long < opt->bw) opt->bw_long = opt->bw;
|
||||
if (mm_verbose >= 3)
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] mid_occ = %d\n", __func__, realtime() - mm_realtime0, cputime() / (realtime() - mm_realtime0), opt->mid_occ);
|
||||
}
|
||||
@@ -113,7 +114,7 @@ int mm_set_opt(const char *preset, mm_idxopt_t *io, mm_mapopt_t *mo)
|
||||
mo->min_dp_max = 200;
|
||||
} else if (strncmp(preset, "asm", 3) == 0) {
|
||||
io->flag = 0, io->k = 19, io->w = 19;
|
||||
mo->bw = mo->bw_long = 100000;
|
||||
mo->bw = 1000, mo->bw_long = 100000;
|
||||
mo->max_gap = 10000;
|
||||
mo->flag |= MM_F_RMQ;
|
||||
mo->min_mid_occ = 50, mo->max_mid_occ = 500;
|
||||
|
||||
+1
-1
@@ -3,7 +3,7 @@ from libc.stdlib cimport free
|
||||
cimport cmappy
|
||||
import sys
|
||||
|
||||
__version__ = '2.23'
|
||||
__version__ = '2.24'
|
||||
|
||||
cmappy.mm_reset_timer()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user