mirror of
https://github.com/chhylp123/hifiasm.git
synced 2026-10-02 00:18:11 +08:00
integer correction
This commit is contained in:
+30
-9
@@ -16280,6 +16280,18 @@ uint32_t cmp_untig_graph(ma_ug_t *src, ma_ug_t *dest)
|
||||
fprintf(stderr, "src->g->seq[%u].len: %u, dest->g->seq[%u].len: %u\n",
|
||||
v, src->g->seq[v].len, v, dest->g->seq[v].len);
|
||||
}
|
||||
|
||||
if(src->u.a[v].n != dest->u.a[v].n)
|
||||
{
|
||||
fprintf(stderr, "src->u.a[%u].n: %u, dest->u.a[%u].n: %u\n",
|
||||
v, src->u.a[v].n, v, dest->u.a[v].n);
|
||||
}
|
||||
|
||||
if(src->u.a[v].a[0] != dest->u.a[v].a[0] ||
|
||||
src->u.a[v].a[src->u.a[v].n-1] != dest->u.a[v].a[dest->u.a[v].n - 1])
|
||||
{
|
||||
fprintf(stderr, "unequal beg/end node\n");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -28778,7 +28790,7 @@ void reset_bub(bubble_type* bub, ma_ug_t *ug, trans_chain* back_ug_chain, kvec_a
|
||||
destory_bubbles(bub);
|
||||
memset(bub, 0, sizeof(bubble_type));
|
||||
|
||||
new_rtg_edges->a.n = 0;
|
||||
if(new_rtg_edges) new_rtg_edges->a.n = 0;
|
||||
///classify_untigs(ug, sg, coverage_cut, sources, reverse_sources, ruIndex, new_rtg_edges, max_hang, min_ovlp);
|
||||
identify_bubbles(ug, bub, back_ug_chain->ir_het, NULL);
|
||||
// update_bubble_chain(ug, bub, 0, 1);
|
||||
@@ -31398,19 +31410,23 @@ char *get_outfile_name(char* output_file_name)
|
||||
}
|
||||
|
||||
void gen_ug_opt_t(ug_opt_t *opt, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, int64_t max_hang, int64_t min_ovlp,
|
||||
int64_t gap_fuzz, int64_t min_dp, uint64_t* readLen, ma_sub_t *coverage_cut, R_to_U* ruIndex)
|
||||
int64_t gap_fuzz, int64_t min_dp, uint64_t* readLen, ma_sub_t *coverage_cut, R_to_U* ruIndex, long long tipsLen,
|
||||
float tip_drop_ratio, long long stops_threshold, float chimeric_rate, float drop_ratio, bub_label_t* b_mask_t)
|
||||
{
|
||||
memset(opt, 0, sizeof((*opt)));
|
||||
opt->sources = sources; opt->reverse_sources = reverse_sources; opt->max_hang = max_hang;
|
||||
opt->min_ovlp = min_ovlp; opt->gap_fuzz = gap_fuzz; opt->min_dp = min_dp; opt->readLen = readLen;
|
||||
opt->coverage_cut = coverage_cut; opt->ruIndex = ruIndex;
|
||||
opt->coverage_cut = coverage_cut; opt->ruIndex = ruIndex; opt->tipsLen = tipsLen;
|
||||
opt->tip_drop_ratio = tip_drop_ratio; opt->stops_threshold = stops_threshold;
|
||||
opt->chimeric_rate = chimeric_rate; opt->drop_ratio = drop_ratio; opt->b_mask_t = b_mask_t;
|
||||
}
|
||||
|
||||
void create_ul_info(ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, int64_t max_hang, int64_t min_ovlp, int64_t gap_fuzz,
|
||||
int64_t min_dp, uint64_t* readLen, ma_sub_t *coverage_cut, R_to_U* ruIndex)
|
||||
int64_t min_dp, uint64_t* readLen, ma_sub_t *coverage_cut, R_to_U* ruIndex, long long tipsLen, float tip_drop_ratio, long long stops_threshold, float chimeric_rate, float drop_ratio, bub_label_t* b_mask_t)
|
||||
{
|
||||
ug_opt_t opt;
|
||||
gen_ug_opt_t(&opt, sources, reverse_sources, max_hang, min_ovlp, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex);
|
||||
gen_ug_opt_t(&opt, sources, reverse_sources, max_hang, min_ovlp, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex,
|
||||
tipsLen, tip_drop_ratio, stops_threshold, chimeric_rate, drop_ratio, b_mask_t);
|
||||
ul_load(&opt);
|
||||
}
|
||||
|
||||
@@ -31480,13 +31496,16 @@ ma_sub_t **coverage_cut_ptr, int debug_g)
|
||||
{
|
||||
memset(R_INF.trio_flag, AMBIGU, R_INF.total_reads*sizeof(uint8_t));
|
||||
}
|
||||
|
||||
if(asm_opt.ar) init_all_ul_t(&UL_INF, &R_INF);
|
||||
// if (asm_opt.flag & HA_F_VERBOSE_GFA) {
|
||||
// write_debug_graph(NULL, sources, coverage_cut, output_file_name, reverse_sources, ruIndex, &UL_INF);
|
||||
// debug_gfa:;
|
||||
// }
|
||||
///should recover edges from sources by using UL alignments
|
||||
if(asm_opt.ar) create_ul_info(sources, reverse_sources, max_hang_length, mini_overlap_length, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex);
|
||||
if(asm_opt.ar) {
|
||||
create_ul_info(sources, reverse_sources, max_hang_length, mini_overlap_length, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex,
|
||||
(asm_opt.max_short_tip*2), 0.15, 3, 0.05, 0.9, &b_mask_t);
|
||||
}
|
||||
|
||||
clean_weak_ma_hit_t(sources, reverse_sources, n_read, asm_opt.ar?UL_COV_THRES:(uint32_t)-1);
|
||||
sg = gen_init_sg(min_dp, n_read, mini_overlap_length, max_hang_length, gap_fuzz, sources, readLen, ruIndex,
|
||||
@@ -31519,14 +31538,16 @@ ma_sub_t **coverage_cut_ptr, int debug_g)
|
||||
free(unlean_name);
|
||||
}
|
||||
|
||||
gen_ug_opt_t(&uopt, sources, reverse_sources, max_hang_length, mini_overlap_length, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex);
|
||||
gen_ug_opt_t(&uopt, sources, reverse_sources, max_hang_length, mini_overlap_length, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex,
|
||||
(asm_opt.max_short_tip*2), 0.15, 3, 0.05, 0.9, &b_mask_t);
|
||||
ul_clean_gfa(&uopt, sg, sources, reverse_sources, ruIndex, clean_round, min_ovlp_drop_ratio, max_ovlp_drop_ratio,
|
||||
0.6, asm_opt.max_short_tip, &b_mask_t, !!asm_opt.ar, ha_opt_triobin(&asm_opt), UL_COV_THRES, o_file);
|
||||
|
||||
if (asm_opt.flag & HA_F_VERBOSE_GFA) {
|
||||
write_debug_graph(sg, sources, coverage_cut, output_file_name, reverse_sources, ruIndex, &UL_INF);
|
||||
debug_gfa:;
|
||||
gen_ug_opt_t(&uopt, sources, reverse_sources, max_hang_length, mini_overlap_length, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex);
|
||||
gen_ug_opt_t(&uopt, sources, reverse_sources, max_hang_length, mini_overlap_length, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex,
|
||||
(asm_opt.max_short_tip*2), 0.15, 3, 0.05, 0.9, &b_mask_t);
|
||||
}
|
||||
if(asm_opt.ar) ul_realignment_gfa(&uopt, sg);
|
||||
print_debug_gfa(sg, NULL, coverage_cut, "UL.debug", sources, ruIndex, max_hang_length, mini_overlap_length);
|
||||
|
||||
+1952
-5
File diff suppressed because it is too large
Load Diff
@@ -6416,15 +6416,15 @@ asg_arc_t *p, uint32_t check_het)
|
||||
if(x_1_b_id) id1 = (*x_1_b_id);
|
||||
if((x_0 != (uint32_t)-1) && (x_1 != (uint32_t)-1))
|
||||
{
|
||||
if(((x_0>>1) == (x_1>>1)))
|
||||
if(((x_0>>1) == (x_1>>1)))///if we would like to find a edge bridging two nearby bubbles
|
||||
{
|
||||
if(x_0_b_id == NULL && x_1_b_id == NULL)
|
||||
{
|
||||
get_bub_id(bub, x_0>>1, &id0, &id1, check_het);
|
||||
}
|
||||
|
||||
get_bubbles(bub, id0, &beg_0, &sink_0, &a, &n, NULL);
|
||||
get_bubbles(bub, id1, &beg_1, &sink_1, &a, &n, NULL);
|
||||
get_bubbles(bub, id0, &beg_0, &sink_0, &a, &n, NULL);//first bubble
|
||||
get_bubbles(bub, id1, &beg_1, &sink_1, &a, &n, NULL);//second bubble
|
||||
|
||||
|
||||
ori_0 = (uint64_t)-1;
|
||||
@@ -6716,10 +6716,11 @@ void detect_bub_graph(bubble_type* bub, asg_t *untig_sg)
|
||||
uint64_t pLen, rLEN, r_hetLen;
|
||||
ma_utg_t *u = NULL;
|
||||
asg_arc_t *t = NULL;
|
||||
for (i = 0; i < ug->u.n; i++)
|
||||
for (i = 0; i < ug->u.n; i++)///bubble chain graph; bubbles within the same chain have been merged
|
||||
{
|
||||
u = &(ug->u.a[i]);
|
||||
if(u->n == 0) continue;
|
||||
///u is a bubble chain
|
||||
for (k = pLen = rLEN = r_hetLen = beg_idx = 0, end_idx = -1; k < u->n; k++)
|
||||
{
|
||||
rId = u->a[k]>>33;
|
||||
@@ -6727,7 +6728,7 @@ void detect_bub_graph(bubble_type* bub, asg_t *untig_sg)
|
||||
get_bubbles(bub, rId, ori == 1?&root:&r_root, ori == 0?&root:&r_root, NULL, NULL, NULL);
|
||||
|
||||
t = NULL;
|
||||
if(k+1 < u->n) t = &(arc_first(bg, u->a[k]>>32));
|
||||
if(k+1 < u->n) t = &(arc_first(bg, u->a[k]>>32));//edge between two bubbles within the same chain
|
||||
|
||||
pLen += bg->seq[rId].len;///path length in bubble
|
||||
if(end_idx < beg_idx) ///first bubble
|
||||
@@ -6739,7 +6740,7 @@ void detect_bub_graph(bubble_type* bub, asg_t *untig_sg)
|
||||
|
||||
if(t)
|
||||
{
|
||||
if(t->el == 0)
|
||||
if(t->el == 0)///there is tangle between two bubbles
|
||||
{
|
||||
if(end_idx >= beg_idx)
|
||||
{
|
||||
@@ -6752,7 +6753,7 @@ void detect_bub_graph(bubble_type* bub, asg_t *untig_sg)
|
||||
pLen = rLEN = r_hetLen = 0;
|
||||
beg_idx = k + 1; end_idx = k;
|
||||
}
|
||||
else
|
||||
else///two bubbles directly connected with each other
|
||||
{
|
||||
pLen += t->ol;
|
||||
rLEN += t->ol;
|
||||
@@ -6784,7 +6785,7 @@ void get_bub_graph(ma_ug_t* ug, bubble_type* bub)
|
||||
uint32_t *pre = NULL; MALLOC(pre, n_vtx);
|
||||
uint32_t pre_id, adjecent, bub_occ;
|
||||
asg_t *bub_g = asg_init();
|
||||
for (v = 0; v < bub->f_bub; v++)
|
||||
for (v = 0; v < bub->f_bub; v++)///all bubbles
|
||||
{
|
||||
uint64_t pathbase;
|
||||
uint32_t beg, sink;
|
||||
@@ -6793,11 +6794,12 @@ void get_bub_graph(ma_ug_t* ug, bubble_type* bub)
|
||||
bub_g->seq[v].c = PRIMARY_LABLE;
|
||||
}
|
||||
|
||||
//check all unitigs
|
||||
//check all unitigs, instead of bubble nodes
|
||||
for (v = 0; v < n_vtx; ++v)
|
||||
{
|
||||
if(sg->seq[v>>1].del) continue;
|
||||
if(bub->b_s_idx.a[v>>1] == (uint64_t)-1) continue; ///if (v>>1) is not a beg or sink of bubbles
|
||||
///one node might be the beg/sink node of at most two bubbles
|
||||
bub_occ = connect_bub_occ(bub, v>>1, bub->check_het);
|
||||
if(bub_occ == 0) continue;
|
||||
if(bub_occ == 2)
|
||||
@@ -6812,6 +6814,7 @@ void get_bub_graph(ma_ug_t* ug, bubble_type* bub)
|
||||
}
|
||||
if(ma_2_bub_arc(bub, v, NULL, (uint32_t)-1, NULL, &t, bub->check_het) == 0) continue;
|
||||
|
||||
///v is the beg/sink node of only one bubble
|
||||
get_shortest_path(v, &pq, sg, pre);
|
||||
for (k = 0; k < pq.dis.n; k++)
|
||||
{
|
||||
@@ -6833,7 +6836,7 @@ void get_bub_graph(ma_ug_t* ug, bubble_type* bub)
|
||||
|
||||
if(adjecent == 0)
|
||||
{
|
||||
if(ma_2_bub_arc(bub, v, NULL, k^1, NULL, &t, bub->check_het))
|
||||
if(ma_2_bub_arc(bub, v, NULL, k^1, NULL, &t, bub->check_het))///edges spanning tangles
|
||||
{
|
||||
t.el = 0; t.ol = pq.dis.a[k] + sg->seq[k>>1].len;
|
||||
p = asg_arc_pushp(bub_g);
|
||||
@@ -6963,7 +6966,7 @@ uint8_t* vis_flag, uint32_t vis_flag_n, kvec_t_u32_warp* stack, asg_t *bsg, asg_
|
||||
radix_sort_u32(broken->a.a, broken->a.a + broken->a.n);
|
||||
for (i = n = 0, pre = (uint32_t)-1; i < broken->a.n; i++)
|
||||
{
|
||||
if((broken->a.a[i]>>1) == (pre>>1)) continue;
|
||||
if((broken->a.a[i]>>1) == (pre>>1)) continue;///skip same node like v and v^1
|
||||
pre = broken->a.a[i];
|
||||
broken->a.a[n] = pre;
|
||||
n++;
|
||||
@@ -8046,7 +8049,7 @@ void resolve_bubble_chain_tangle(ma_ug_t* ug, bubble_type* bub)
|
||||
occ_idx.a[occ_idx.n - k - 1] = tmp;
|
||||
}
|
||||
|
||||
for (k = 0; k < bub_ug->g->n_seq; k++)
|
||||
for (k = 0; k < bub_ug->g->n_seq; k++)///start from the longest chain
|
||||
{
|
||||
v = ((uint32_t)(occ_idx.a[k]))<<1;
|
||||
if(is_used[v] == 0 && asg_arc_n(bub_ug->g, v) > 0)
|
||||
@@ -8192,9 +8195,9 @@ void update_bubble_chain(ma_ug_t* ug, bubble_type* bub, uint32_t is_middle, uint
|
||||
///uint64_t end_thres;
|
||||
|
||||
uint8_t *bsg_idx = NULL; CALLOC(bsg_idx, n_vtx>>1);
|
||||
for (i = 0; i < bub_ug->u.n; i++)
|
||||
for (i = 0; i < bub_ug->u.n; i++)///label all unitigs within the bubble chains
|
||||
{
|
||||
u = &(bub_ug->u.a[i]);
|
||||
u = &(bub_ug->u.a[i]);///a bubble chain
|
||||
if(u->n == 0) continue;
|
||||
for (k_i = 0; k_i < u->n; k_i++)
|
||||
{
|
||||
@@ -8213,7 +8216,7 @@ void update_bubble_chain(ma_ug_t* ug, bubble_type* bub, uint32_t is_middle, uint
|
||||
new_bub = bub->b_g->n_seq;
|
||||
for (i = 0; i < bub_ug->u.n; i++)
|
||||
{
|
||||
u = &(bub_ug->u.a[i]);
|
||||
u = &(bub_ug->u.a[i]);///bubble chain
|
||||
if(u->n == 0) continue;
|
||||
///end_thres = calculate_chain_weight(u, bub, ug, &x);
|
||||
if(is_middle)
|
||||
@@ -8223,7 +8226,7 @@ void update_bubble_chain(ma_ug_t* ug, bubble_type* bub, uint32_t is_middle, uint
|
||||
if(k_i+1 >= u->n) continue;
|
||||
///note: must igore .del here, since bsg might be changed
|
||||
t = &(arc_first(bsg, u->a[k_i]>>32));
|
||||
if(t->el == 1) continue;
|
||||
if(t->el == 1) continue;///if a[k_i] and a[k_i+1] are directly connected without any tangle involoved
|
||||
|
||||
rId_0 = u->a[k_i]>>33;
|
||||
ori_0 = u->a[k_i]>>32&1;
|
||||
@@ -8233,7 +8236,7 @@ void update_bubble_chain(ma_ug_t* ug, bubble_type* bub, uint32_t is_middle, uint
|
||||
ori_1 = (u->a[k_i+1]>>32&1)^1;
|
||||
get_bubbles(bub, rId_1, ori_1 == 1?&root_1:NULL, ori_1 == 0?&root_1:NULL, NULL, NULL, NULL);
|
||||
|
||||
broken.a.n = 0;
|
||||
broken.a.n = 0;///just collect all nodes between root_0 and root_1
|
||||
get_related_bub_nodes(&broken, bub, &pq, sg, pre, root_0, root_1, NULL);
|
||||
get_related_bub_nodes(&broken, bub, &pq, sg, pre, root_1, root_0, NULL);
|
||||
///no need to cut the edge, we still have chance to flip by chain
|
||||
@@ -9353,8 +9356,8 @@ void build_bub_graph(ma_ug_t* ug, bubble_type* bub)
|
||||
// detect_bub_graph(bub, ug->g, 1);
|
||||
update_bubble_chain(ug, bub, 1, 0);
|
||||
///print_bubble_chain(bub, "second round");
|
||||
update_bubble_chain(ug, bub, 0, 1);
|
||||
resolve_bubble_chain_tangle(ug, bub);
|
||||
update_bubble_chain(ug, bub, 0, 1);///resolve tangles within bubble chains
|
||||
resolve_bubble_chain_tangle(ug, bub);///resolve tangles between bubble chains
|
||||
}
|
||||
|
||||
void get_forward_distance(uint32_t src, uint32_t dest, asg_t *sg, hc_links* link, MT* M)
|
||||
|
||||
@@ -8465,13 +8465,14 @@ void print_ovlp_src_bl_stat(all_ul_t *x, const ug_opt_t *uopt)
|
||||
__func__, tt[0]+tt[1]+tt[2]+tt[3], tt[1], tt[2], tt[3]);
|
||||
}
|
||||
|
||||
void gen_ul_vec_rid_t(all_ul_t *x)
|
||||
void gen_ul_vec_rid_t(all_ul_t *x, All_reads *rdb, ma_ug_t *ug)
|
||||
{
|
||||
ul_vec_rid_t *ridx = &(x->ridx);
|
||||
uint64_t k, i, l, m, *a, a_n; ul_vec_t *p = NULL;
|
||||
ridx->idx.n = ridx->idx.m = R_INF.total_reads + 1; CALLOC(ridx->idx.a, ridx->idx.n);
|
||||
uint64_t k, i, l, m, *a, a_n, idx_n; ul_vec_t *p = NULL;
|
||||
idx_n = (rdb?rdb->total_reads:ug->u.n);
|
||||
ridx->idx.n = ridx->idx.m = idx_n + 1; CALLOC(ridx->idx.a, ridx->idx.n);
|
||||
|
||||
for (k = 0; k < x->n; k++) {
|
||||
for (k = 0; k < x->n; k++) {///each UL read
|
||||
p = &(x->a[k]);
|
||||
for (i = 0; i < p->bb.n; i++) {
|
||||
if(p->bb.a[i].base) continue;
|
||||
@@ -8486,7 +8487,7 @@ void gen_ul_vec_rid_t(all_ul_t *x)
|
||||
}
|
||||
|
||||
ridx->occ.n = ridx->occ.m = l; MALLOC(ridx->occ.a, ridx->occ.n);
|
||||
for (k = 0; k < R_INF.total_reads; k++) {
|
||||
for (k = 0; k < idx_n; k++) {
|
||||
a = ridx->occ.a + ridx->idx.a[k];
|
||||
a_n = ridx->idx.a[k+1] - ridx->idx.a[k];
|
||||
if(a_n) a[a_n-1] = 0;
|
||||
@@ -8504,9 +8505,105 @@ void gen_ul_vec_rid_t(all_ul_t *x)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
uint32_t ugl_cover_check(uint64_t is, uint64_t ie, ma_utg_t *u)
|
||||
{
|
||||
if(is == 0 && ie == u->len) return 1;
|
||||
uint64_t l, i, us, ue;
|
||||
for (i = l = 0; i < u->n; i++) {
|
||||
us = l; ue = l + Get_READ_LENGTH(R_INF, (u->a[i]>>33));
|
||||
if(is <= us && ie >= ue) return 1;
|
||||
if(us >= ie) break;
|
||||
l += (uint32_t)u->a[i];
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void update_ug_arch_ul(void *data, long i, int tid) // callback for kt_for()
|
||||
{
|
||||
const ma_ug_t *ug = (ma_ug_t *)data;
|
||||
asg_arc_t *e = &(ug->g->arc[i]); e->ou = 0;
|
||||
uint32_t v = e->ul>>32, w = e->v, k, uv, uw;
|
||||
uint64_t *a, a_n; uc_block_t *p, *n;
|
||||
a = UL_INF.ridx.occ.a + UL_INF.ridx.idx.a[v>>1];
|
||||
a_n = UL_INF.ridx.idx.a[(v>>1)+1] - UL_INF.ridx.idx.a[v>>1];
|
||||
for (k = 0; k < a_n; k++) {
|
||||
p = &(UL_INF.a[a[k]>>32].bb.a[(uint32_t)(a[k])]);
|
||||
if(p->base || (!p->el) || (!p->pchain)) continue;
|
||||
uv = (((uint32_t)(p->hid))<<1)|((uint32_t)(p->rev));
|
||||
if((uv == v) && (p->aidx != (uint32_t)-1)) {
|
||||
n = &(UL_INF.a[a[k]>>32].bb.a[p->aidx]);
|
||||
// if(!((!n->base)&&(n->el)&&(n->pchain)&&(n->pidx==((uint32_t)(a[k]))))) {
|
||||
// fprintf(stderr, "k::%u, n->base::%u, n->el::%u, n->pchain::%u, n->pidx::%u, p->aidx::%u\n",
|
||||
// k, n->base, n->el, n->pchain, n->pidx, p->aidx);
|
||||
// }
|
||||
assert((!n->base)&&(n->el)&&(n->pchain)&&(n->pidx==((uint32_t)(a[k]))));
|
||||
uw = (((uint32_t)(n->hid))<<1)|((uint32_t)(n->rev));
|
||||
if(uw == w) e->ou++;
|
||||
}
|
||||
|
||||
if(((uv^1) == v) && (p->pidx != (uint32_t)-1)) {
|
||||
n = &(UL_INF.a[a[k]>>32].bb.a[p->pidx]);
|
||||
assert((!n->base)&&(n->el)&&(n->pchain)&&(n->aidx==((uint32_t)(a[k]))));
|
||||
uw = (((uint32_t)(n->hid))<<1)|((uint32_t)(n->rev)); uw ^= 1;
|
||||
if(uw == w) e->ou++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void filter_short_ulalignments(void *data, long i, int tid) // callback for kt_for()
|
||||
{
|
||||
const ma_ug_t *ug = (ma_ug_t *)data;
|
||||
uc_block_t *a = NULL; uc_block_t *p; int64_t k, a_n; uint32_t z, fz, lz, l, bz;
|
||||
a = UL_INF.a[i].bb.a; a_n = UL_INF.a[i].bb.n;
|
||||
for (k = a_n - 1; k >= 0; k--) {
|
||||
p = &(a[k]);
|
||||
if(p->base || (!p->el) || (!p->pchain)) continue;
|
||||
if((p->pidx == (uint32_t)-1) && (p->aidx == (uint32_t)-1)) {
|
||||
if(!ugl_cover_check(p->ts, p->te, &(ug->u.a[p->hid]))) p->pchain = 0;
|
||||
continue;
|
||||
}
|
||||
if(p->pidx == (uint32_t)-1) continue;
|
||||
if(ugl_cover_check(p->ts, p->te, &(ug->u.a[p->hid]))) continue;
|
||||
for (z = p->pidx; z != (uint32_t)-1; z = a[z].pidx) {
|
||||
if(ugl_cover_check(a[z].ts, a[z].te, &(ug->u.a[a[z].hid]))) break;
|
||||
}
|
||||
lz = z; fz = p->aidx; l = 0; if(fz != (uint32_t)-1) l = a[fz].pdis;
|
||||
for (z = k; z != lz; z = bz) {
|
||||
bz = a[z].pidx; l += a[z].pdis;
|
||||
a[z].pidx = a[z].pdis = a[z].aidx = (uint32_t)-1; a[z].pchain = 0;
|
||||
}
|
||||
|
||||
if(fz != (uint32_t)-1 && lz != (uint32_t)-1) {
|
||||
a[fz].pdis = l; a[fz].pidx = lz; a[lz].aidx = fz;
|
||||
} else if(fz != (uint32_t)-1) {
|
||||
a[fz].pdis = a[fz].pidx = (uint32_t)-1;
|
||||
} else if(lz != (uint32_t)-1) {
|
||||
a[lz].aidx = (uint32_t)-1;
|
||||
}
|
||||
}
|
||||
|
||||
// for (k = a_n - 1; k >= 0; k--) {
|
||||
// p = &(a[k]);
|
||||
// if(p->base || (!p->el) || (!p->pchain)) continue;
|
||||
// if(p->pidx != (uint32_t)-1) {
|
||||
// assert(a[p->pidx].aidx == (uint32_t)k);
|
||||
// }
|
||||
// if(p->aidx != (uint32_t)-1) {
|
||||
// assert(a[p->aidx].pidx == (uint32_t)k);
|
||||
// }
|
||||
// }
|
||||
}
|
||||
|
||||
|
||||
void filter_ul_ug(ma_ug_t *ug)
|
||||
{
|
||||
kt_for(asm_opt.thread_num, filter_short_ulalignments, ug, UL_INF.n);
|
||||
}
|
||||
|
||||
|
||||
int32_t find_ul_block_max_reverse(int32_t n, const uc_block_t *a, uint32_t x)
|
||||
{
|
||||
int32_t s = 0, e = n;
|
||||
@@ -8667,6 +8764,7 @@ static void update_ovlp_src(void *data, long i, int tid) // callback for kt_for(
|
||||
// fprintf(stderr, "--i->%d, a_n->%lu--\n", i, a_n);
|
||||
}
|
||||
|
||||
|
||||
uint64_t* get_hifi2ul_list(all_ul_t *x, uint64_t hid, uint64_t* a_n)
|
||||
{
|
||||
(*a_n) = x->ridx.idx.a[hid+1] - x->ridx.idx.a[hid];
|
||||
@@ -8710,7 +8808,7 @@ int scall_ul_pipeline(uldat_t* sl, const enzyme *fn)
|
||||
fprintf(stderr, "[M::%s::] ==> # reads: %lu, # bases: %lu\n", __func__, UL_INF.n, sl->total_base);
|
||||
fprintf(stderr, "[M::%s::] ==> # bases: %lu; # corrected bases: %lu; # recorrected bases: %lu\n",
|
||||
__func__, sl->num_bases, sl->num_corrected_bases, sl->num_recorrected_bases);
|
||||
gen_ul_vec_rid_t(&UL_INF);
|
||||
gen_ul_vec_rid_t(&UL_INF, &R_INF, NULL);
|
||||
return 1;
|
||||
}
|
||||
|
||||
@@ -8718,13 +8816,7 @@ int scall_ul_pipeline(uldat_t* sl, const enzyme *fn)
|
||||
int rescall_ul_pipeline(uldat_t* sl, const enzyme *fn)
|
||||
{
|
||||
double index_time = yak_realtime();
|
||||
int32_t i; uint32_t k, rlen;
|
||||
///clean UL_INF
|
||||
for (k = 0; k < UL_INF.n; k++) {
|
||||
rlen = UL_INF.a[k].rlen;
|
||||
free(UL_INF.a[k].bb.a); free(UL_INF.a[k].N_site.a); free(UL_INF.a[k].r_base.a);
|
||||
memset(&(UL_INF.a[k]), 0, sizeof(UL_INF.a[k])); UL_INF.a[k].rlen = rlen;
|
||||
}
|
||||
int32_t i;
|
||||
|
||||
for (i = 0; i < fn->n; i++){
|
||||
gzFile fp;
|
||||
@@ -10009,6 +10101,7 @@ int32_t load_all_ul_t(all_ul_t *x, char* file_name, All_reads *hR, ma_ug_t *ug)
|
||||
return 0;
|
||||
}
|
||||
|
||||
destory_all_ul_t(x);
|
||||
memset(x, 0, sizeof(*x)); x->hR = hR; init_aux_table();
|
||||
uint64_t k; ul_vec_t *p = NULL;
|
||||
|
||||
@@ -10089,7 +10182,7 @@ uint64_t ul_refine_alignment(const ug_opt_t *uopt, asg_t *sg)
|
||||
init_uldat_t(&sl, NULL, NULL, &opt, CHUNK_SIZE, asm_opt.thread_num, uopt, uu); sl.rg = sg;
|
||||
if(work_ul_gchains(&sl)) {
|
||||
free(UL_INF.ridx.idx.a); free(UL_INF.ridx.occ.a); memset(&(UL_INF.ridx), 0, sizeof(UL_INF.ridx));
|
||||
gen_ul_vec_rid_t(&UL_INF);
|
||||
gen_ul_vec_rid_t(&UL_INF, &R_INF, NULL);
|
||||
kt_for(sl.n_thread, update_ovlp_src, &sl, R_INF.total_reads);
|
||||
kt_for(sl.n_thread, update_ovlp_src_bl, &sl, R_INF.total_reads);
|
||||
destroy_ul_idx_t(uu);
|
||||
@@ -10114,6 +10207,18 @@ uint32_t dd_ug(asg_t *sg, ma_ug_t *ug, ma_sub_t* coverage_cut, ma_hit_t_alloc* s
|
||||
}
|
||||
|
||||
|
||||
void clear_all_ul_t(all_ul_t *x)
|
||||
{
|
||||
uint64_t k, rlen;
|
||||
for (k = 0; k < x->n; k++) {
|
||||
rlen = x->a[k].rlen;
|
||||
free(x->a[k].bb.a); free(x->a[k].N_site.a); free(x->a[k].r_base.a);
|
||||
memset(&(x->a[k]), 0, sizeof(x->a[k])); x->a[k].rlen = rlen;
|
||||
}
|
||||
free(x->ridx.idx.a); free(x->ridx.occ.a); memset(&(x->ridx), 0, sizeof((x->ridx)));
|
||||
}
|
||||
|
||||
|
||||
|
||||
ma_ug_t *ul_realignment(const ug_opt_t *uopt, asg_t *sg)
|
||||
{
|
||||
@@ -10131,11 +10236,14 @@ ma_ug_t *ul_realignment(const ug_opt_t *uopt, asg_t *sg)
|
||||
// dd_ug(sg, ug, uopt->coverage_cut, uopt->sources, uopt->ruIndex, "UL.sa");
|
||||
// debug_sl_compress_base_disk_0(&sl, asm_opt.ar);
|
||||
// detect_outlier_len("ul_realignment");
|
||||
clear_all_ul_t(&UL_INF);
|
||||
if(!load_all_ul_t(&UL_INF, gfa_name, &R_INF, ug)) {
|
||||
gen_UL_reovlps(&sl, ug, sg, gfa_name, cutoff);
|
||||
write_all_ul_t(&UL_INF, gfa_name, ug);
|
||||
}
|
||||
|
||||
filter_ul_ug(ug);
|
||||
gen_ul_vec_rid_t(&UL_INF, NULL, ug);
|
||||
kt_for(asm_opt.thread_num, update_ug_arch_ul, ug, ug->g->n_arc);
|
||||
// print_all_ul_t_stat(&UL_INF);
|
||||
// kt_for(sl.n_thread, update_ovlp_src, &sl, R_INF.total_reads);
|
||||
// kt_for(sl.n_thread, update_ovlp_src_bl, &sl, R_INF.total_reads);
|
||||
|
||||
@@ -10,5 +10,6 @@ uint64_t ul_refine_alignment(const ug_opt_t *uopt, asg_t *sg);
|
||||
ma_ug_t *ul_realignment(const ug_opt_t *uopt, asg_t *sg);
|
||||
int32_t write_all_ul_t(all_ul_t *x, char* file_name, ma_ug_t *ug);
|
||||
int32_t load_all_ul_t(all_ul_t *x, char* file_name, All_reads *hR, ma_ug_t *ug);
|
||||
uint32_t ugl_cover_check(uint64_t is, uint64_t ie, ma_utg_t *u);
|
||||
|
||||
#endif
|
||||
|
||||
Reference in New Issue
Block a user