mirror of
https://github.com/chhylp123/hifiasm.git
synced 2026-10-01 15:28:12 +08:00
code clean
This commit is contained in:
+169
@@ -1,5 +1,6 @@
|
||||
#include <stdio.h>
|
||||
#include <math.h>
|
||||
#include <assert.h>
|
||||
#include "htab.h"
|
||||
#include "ksort.h"
|
||||
#include "Hash_Table.h"
|
||||
@@ -780,3 +781,171 @@ void ha_sort_list_by_anchor(overlap_region_alloc *overlap_list)
|
||||
{
|
||||
ks_introsort_or_xs(overlap_list->length, overlap_list->list);
|
||||
}
|
||||
|
||||
|
||||
void minimizers_gen(ha_abufl_t *ab, char* rs, int64_t rl, uint64_t mz_w, uint64_t mz_k, Candidates_list *cl, kvec_t_u8_warp* k_flag,
|
||||
void *ha_flt_tab, ha_pt_t *ha_idx, kvec_t_u64_warp* dbg_ct, st_mt_t *sp, uint32_t high_occ)
|
||||
{
|
||||
uint64_t i, k, l; int n, j; ha_mzl_t *z; seedl_t *s;
|
||||
if(high_occ < 1) high_occ = 1;
|
||||
clear_Candidates_list(cl); ab->mz.n = 0, ab->n_a = 0;
|
||||
|
||||
// get the list of anchors
|
||||
mz2_ha_sketch(rs, rl, mz_w, mz_k, 0, !(asm_opt.flag & HA_F_NO_HPC), &ab->mz, ha_flt_tab, asm_opt.mz_sample_dist, k_flag, dbg_ct, NULL, -1, asm_opt.dp_min_len, -1, sp, asm_opt.mz_rewin, 0, NULL);
|
||||
|
||||
// minimizer of queried read
|
||||
if (ab->mz.m > ab->old_mz_m) {
|
||||
ab->old_mz_m = ab->mz.m;
|
||||
REALLOC(ab->seed, ab->old_mz_m);
|
||||
}
|
||||
|
||||
for (i = 0, ab->n_a = 0; i < ab->mz.n; ++i) {
|
||||
ab->seed[i].a = ha_ptl_get(ha_idx, ab->mz.a[i].x, &n);
|
||||
ab->seed[i].n = n;
|
||||
ab->n_a += n;
|
||||
}
|
||||
|
||||
if (ab->n_a > ab->m_a) {
|
||||
ab->m_a = ab->n_a;
|
||||
REALLOC(ab->a, ab->m_a);
|
||||
}
|
||||
|
||||
for (i = 0, k = 0; i < ab->mz.n; ++i) {
|
||||
///z is one of the minimizer
|
||||
z = &ab->mz.a[i]; s = &ab->seed[i];
|
||||
for (j = 0; j < s->n; ++j) {
|
||||
const ha_idxposl_t *y = &s->a[j];
|
||||
anchor1_t *an = &ab->a[k++];
|
||||
uint8_t rev = z->rev == y->rev? 0 : 1;
|
||||
an->other_off = y->pos;
|
||||
an->self_off = rev? rl - 1 - (z->pos + 1 - z->span) : z->pos;
|
||||
///an->cnt: cnt<<8|span
|
||||
an->cnt = s->n; if(an->cnt > ((uint32_t)(0xffffffu))) an->cnt = 0xffffffu;
|
||||
an->cnt <<= 8; an->cnt |= ((z->span <= ((uint32_t)(0xffu)))?z->span:((uint32_t)(0xffu)));
|
||||
an->srt = (uint64_t)y->rid<<33 | (uint64_t)rev<<32 | an->other_off;
|
||||
}
|
||||
}
|
||||
|
||||
radix_sort_ha_an1(ab->a, ab->a + ab->n_a);
|
||||
for (k = 1, l = 0; k <= ab->n_a; ++k) {
|
||||
if (k == ab->n_a || ab->a[k].srt != ab->a[l].srt) {
|
||||
if (k - l > 1)
|
||||
radix_sort_ha_an2(ab->a + l, ab->a + k);
|
||||
l = k;
|
||||
}
|
||||
}
|
||||
|
||||
// copy over to _cl_
|
||||
if (ab->m_a >= (uint64_t)cl->size) {
|
||||
cl->size = ab->m_a;
|
||||
REALLOC(cl->list, cl->size);
|
||||
}
|
||||
|
||||
for (k = 0; k < ab->n_a; ++k) {
|
||||
k_mer_hit *p = &cl->list[k];
|
||||
p->readID = ab->a[k].srt >> 33;
|
||||
p->strand = ab->a[k].srt >> 32 & 1;
|
||||
p->offset = ab->a[k].other_off;
|
||||
p->self_offset = ab->a[k].self_off;
|
||||
if((ab->a[k].cnt>>8) <= high_occ){
|
||||
p->cnt = 1;
|
||||
}
|
||||
else{
|
||||
p->cnt = 1 + (((ab->a[k].cnt>>8) + (high_occ<<1) - 1)/(high_occ<<1));
|
||||
p->cnt = pow(p->cnt, 1.1);
|
||||
}
|
||||
if(p->cnt > ((uint32_t)(0xffffffu))) p->cnt = 0xffffffu;
|
||||
p->cnt <<= 8; p->cnt |= (((uint32_t)(0xffu))&(ab->a[k].cnt));
|
||||
}
|
||||
cl->length = ab->n_a;
|
||||
}
|
||||
|
||||
void lchain_gen(Candidates_list* cl, overlap_region_alloc* ol, uint64_t rid, uint64_t rl, All_reads* rdb,
|
||||
const ul_idx_t *udb, uint32_t beg_tail, overlap_region* tf, uint64_t max_n_chain,
|
||||
int64_t max_skip, int64_t max_iter, int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate, int64_t quick_check)
|
||||
{
|
||||
uint64_t i, k, l, m, sm, cn = cl->length;
|
||||
clear_overlap_region_alloc(ol);
|
||||
clear_fake_cigar(&(tf->f_cigar));
|
||||
|
||||
///calculate_overlap_region_by_chaining(cl, overlap_list, chain_idx, rid, rl, NULL, uref, bw_thres, keep_whole_chain, f_cigar);
|
||||
for (l = 0, k = 1, m = 0; k <= cn; k++) {
|
||||
if((k == cn) || (cl->list[k].readID != cl->list[l].readID)
|
||||
|| (cl->list[k].strand != cl->list[l].strand)) {
|
||||
if(cl->list[l].readID != rid) {
|
||||
tf->x_id = rid;
|
||||
tf->x_pos_strand = cl->list[l].strand;
|
||||
tf->y_id = cl->list[l].readID;
|
||||
tf->y_pos_strand = 0;///always 0
|
||||
sm = lchain_dp(cl->list+l, k-l, cl->list+m, &(cl->chainDP), tf, max_skip, max_iter, max_dis, chn_pen_gap, chn_pen_skip, bw_rate,
|
||||
rl, rdb?Get_READ_LENGTH((*rdb), (*tf).y_id):udb->ug->u.a[(*tf).y_id].len, quick_check);
|
||||
assert(sm > 0);
|
||||
if(ovlp_chain_gen(ol, tf, rl, rdb?Get_READ_LENGTH((*rdb), (*tf).y_id):udb->ug->u.a[(*tf).y_id].len, beg_tail)) {
|
||||
m += sm;
|
||||
}
|
||||
}
|
||||
l = k;
|
||||
}
|
||||
}
|
||||
cl->length = m;
|
||||
|
||||
|
||||
|
||||
if (ol->length > max_n_chain) {
|
||||
int32_t w, n[4], s[4]; overlap_region t;
|
||||
n[0] = n[1] = n[2] = n[3] = 0, s[0] = s[1] = s[2] = s[3] = 0;
|
||||
ks_introsort_or_ss(ol->length, ol->list);
|
||||
for (i = 0; i < ol->length; ++i) {
|
||||
const overlap_region *r = &(ol->list[i]);
|
||||
w = ha_ov_type(r, rl);
|
||||
++n[w];
|
||||
if (((uint64_t)n[w]) == max_n_chain) s[w] = r->shared_seed;
|
||||
}
|
||||
if (s[0] > 0 || s[1] > 0 || s[2] > 0 || s[3] > 0) {
|
||||
// n[0] = n[1] = n[2] = n[3] = 0;
|
||||
for (i = 0, k = 0; i < ol->length; ++i) {
|
||||
overlap_region *r = &(ol->list[i]);
|
||||
w = ha_ov_type(r, rl);
|
||||
// ++n[w];
|
||||
// if (((int)n[w] <= max_n_chain) || (r->shared_seed >= s[w] && s[w] >= (asm_opt.k_mer_length<<1))) {
|
||||
if (r->shared_seed >= s[w]) {
|
||||
if (k != i) {
|
||||
t = ol->list[k];
|
||||
ol->list[k] = ol->list[i];
|
||||
ol->list[i] = t;
|
||||
}
|
||||
++k;
|
||||
}
|
||||
}
|
||||
ol->length = k;
|
||||
}
|
||||
}
|
||||
ks_introsort_or_xs(ol->length, ol->list);
|
||||
}
|
||||
|
||||
void set_lchain_dp_op(uint32_t is_accurate, uint32_t mz_k, int64_t *max_skip, int64_t *max_iter, int64_t *max_dis, double *chn_pen_gap, double *chn_pen_skip, int64_t *quick_check)
|
||||
{
|
||||
double div, pen_gap, pen_skip, tmp;
|
||||
if(is_accurate) {
|
||||
(*quick_check) = 1; (*max_skip) = 25; (*max_iter) = 5000; (*max_dis) = 5000; div = 0.01; pen_gap = 0.5f; pen_skip = 0.0005f;
|
||||
} else {
|
||||
(*quick_check) = 0; (*max_skip) = 25; (*max_iter) = 5000; (*max_dis) = 5000; div = 0.1; pen_gap = 0.5f; pen_skip = 0.0005f;
|
||||
}
|
||||
tmp = expf(-div * (double)mz_k);///0.60049557881 -> HiFi; 0.18268352405 -> ont
|
||||
*chn_pen_gap = pen_gap * tmp;///0.300247789405 -> HiFi; 0.091341762025 -> ont
|
||||
///0.000300247789405 -> HiFi (>3330 will be negative);
|
||||
//0.000091341762025 -> ont (>10947 will be negative);
|
||||
*chn_pen_skip = pen_skip * tmp;
|
||||
}
|
||||
|
||||
void ul_map_lchain(ha_abufl_t *ab, int64_t rid, char* rs, uint64_t rl, uint64_t mz_w, uint64_t mz_k, const ul_idx_t *uref, overlap_region_alloc *overlap_list, overlap_region_alloc *overlap_list_hp, Candidates_list *cl, double bw_thres,
|
||||
int max_n_chain, int keep_whole_chain, kvec_t_u8_warp* k_flag, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct, st_mt_t *sp, uint32_t high_occ, uint32_t is_accurate)
|
||||
{
|
||||
extern void *ha_flt_tab;
|
||||
extern ha_pt_t *ha_idx;
|
||||
int64_t max_skip, max_iter, max_dis, quick_check; double chn_pen_gap, chn_pen_skip;
|
||||
set_lchain_dp_op(is_accurate, mz_k, &max_skip, &max_iter, &max_dis, &chn_pen_gap, &chn_pen_skip, &quick_check);
|
||||
minimizers_gen(ab, rs, rl, mz_w, mz_k, cl, k_flag, ha_flt_tab, ha_idx, dbg_ct, sp, high_occ);
|
||||
lchain_gen(cl, overlap_list, rid, rl, NULL, uref, keep_whole_chain, f_cigar, max_n_chain, max_skip, max_iter, max_dis, chn_pen_gap, chn_pen_skip, bw_thres, quick_check);
|
||||
///no need to sort here, overlap_list has been sorted at lchain_gen
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user