This commit is contained in:
chhylp123
2022-04-13 16:15:03 -04:00
parent 75f8048b18
commit 9c79bdfe7f
10 changed files with 1860 additions and 267 deletions
+1 -1
View File
@@ -232,7 +232,7 @@ void init_opt(hifiasm_opt_t* asm_opt)
asm_opt->kpt_rate = -1; asm_opt->kpt_rate = -1;
asm_opt->infor_cov = 3; asm_opt->infor_cov = 3;
asm_opt->s_hap_cov = 3; asm_opt->s_hap_cov = 3;
asm_opt->ul_error_rate = 0.15; asm_opt->ul_error_rate = 0.2/**0.15**/;
asm_opt->is_dbg_het_cnt = 0; asm_opt->is_dbg_het_cnt = 0;
} }
+99 -102
View File
@@ -54,6 +54,8 @@ KRADIX_SORT_INIT(u_trans_qs, u_trans_t, u_trans_qs_key, member_size(u_trans_t, q
#define u_trans_ts_key(a) ((a).ts) #define u_trans_ts_key(a) ((a).ts)
KRADIX_SORT_INIT(u_trans_ts, u_trans_t, u_trans_ts_key, member_size(u_trans_t, ts)) KRADIX_SORT_INIT(u_trans_ts, u_trans_t, u_trans_ts_key, member_size(u_trans_t, ts))
#define UL_COV_THRES 2
KSORT_INIT_GENERIC(uint32_t) KSORT_INIT_GENERIC(uint32_t)
typedef struct { typedef struct {
@@ -1881,7 +1883,7 @@ ma_sub_t* max_left, ma_sub_t* max_right, float overlap_rate, uint32_t trio_flag)
} }
void collect_sides(uint32_t rid, ma_hit_t_alloc* pafs, all_ul_t *x, uint64_t rLen, ma_sub_t* max_left, ma_sub_t* max_right) void collect_sides(uint32_t rid, ma_hit_t_alloc* pafs, all_ul_t *x, uint64_t rLen, ma_sub_t* max_left, ma_sub_t* max_right, uint64_t ul_thres)
{ {
long long j; long long j;
uint32_t qs, qe; uint32_t qs, qe;
@@ -1915,33 +1917,27 @@ void collect_sides(uint32_t rid, ma_hit_t_alloc* pafs, all_ul_t *x, uint64_t rLe
if(x) { if(x) {
uint64_t *a = NULL, a_n, k; uint64_t *a = NULL, a_n, k;
uc_block_t *p = NULL; uc_block_t *p = NULL; uint64_t cc = 0;
a = get_hifi2ul_list(x, rid, &a_n); a = get_hifi2ul_list(x, rid, &a_n);
for (k = 0; k < a_n; k++) { for (k = 0; k < a_n; k++) {
p = &(x->a[a[k]>>32].bb.a[(uint32_t)(a[k])]); p = &(x->a[a[k]>>32].bb.a[(uint32_t)(a[k])]);
if(p->base/**->hid&x->mm**/) continue;///should not happen if(p->base||(!p->el)) continue;
qs = p->ts; qe = p->te;///note here is ts && te, instead of qs && qe qs = p->ts; qe = p->te;///note here is ts && te, instead of qs && qe
///for UL, we only use overlaps which cover the whole HiFi read
///overlaps from left side if(qs == 0 && qe == rLen){
if(qs == 0){ cc++;
if(qs < max_left->s) max_left->s = qs; if(cc >= ul_thres) break;
if(qe > max_left->e) max_left->e = qe;
} }
}
///overlaps from right side if(cc >= ul_thres) {
if(qe == rLen){ max_left->s = 0; max_left->e = rLen;
if(qs < max_right->s) max_right->s = qs; max_right->s = 0; max_right->e = rLen;
if(qe > max_right->e) max_right->e = qe;
}
///note: if (qs == 0 && qe == rLen)
///this overlap would be added to both b_left and b_right
///that is what we want
} }
} }
} }
void collect_contain(ma_hit_t_alloc* paf1, ma_hit_t_alloc* paf2, uint64_t rLen, void collect_contain(ma_hit_t_alloc* paf1, ma_hit_t_alloc* paf2, uint64_t rLen,
ma_sub_t* max_left, ma_sub_t* max_right, float overlap_rate, all_ul_t *x, uint64_t xid) ma_sub_t* max_left, ma_sub_t* max_right, float overlap_rate)
{ {
long long j, new_left_e, new_right_s; long long j, new_left_e, new_right_s;
new_left_e = max_left->e; new_left_e = max_left->e;
@@ -2007,35 +2003,6 @@ ma_sub_t* max_left, ma_sub_t* max_right, float overlap_rate, all_ul_t *x, uint64
} }
} }
if(x) {
uint64_t *a = NULL, a_n, k;
uc_block_t *p = NULL;
a = get_hifi2ul_list(x, xid, &a_n);
for (k = 0; k < a_n; k++) {
p = &(x->a[a[k]>>32].bb.a[(uint32_t)(a[k])]);
if(p->base/**->hid&x->mm**/) continue;///should not happen
qs = p->ts; qe = p->te;///note here is ts && te, instead of qs && qe
///check contained overlaps
if(qs != 0 && qe != rLen)
{
///[qs, qe), [max_left.s, max_left.e)
if(qs < max_left->e && qe > max_left->e && max_left->e - qs > (overlap_rate * (qe -qs)))
{
///if(qe > max_left->e) max_left->e = qe;
if(qe > max_left->e && qe > new_left_e) new_left_e = qe;
}
///[qs, qe), [max_right.s, max_right.e)
if(qs < max_right->s && qe > max_right->s && qe - max_right->s > (overlap_rate * (qe -qs)))
{
///if(qs < max_right->s) max_right->s = qs;
if(qs < max_right->s && qs < new_right_s) new_right_s = qs;
}
}
}
}
max_left->e = new_left_e; max_left->e = new_left_e;
max_right->s = new_right_s; max_right->s = new_right_s;
} }
@@ -2068,17 +2035,15 @@ char* bq, char* bt)
long long j; long long j;
uint32_t qs, qe; uint32_t qs, qe;
for (j = 0; j < paf->length; j++) for (j = 0; j < paf->length; j++) {
{
if(paf->buffer[j].del) continue; if(paf->buffer[j].del) continue;
qs = Get_qs(paf->buffer[j]); qs = Get_qs(paf->buffer[j]);
qe = Get_qe(paf->buffer[j]); qe = Get_qe(paf->buffer[j]);
///[interval_s, interval_e) must be at least contained at one of the [qs, qe) ///[interval_s, interval_e) must be at least contained at one of the [qs, qe)
if(qs<=interval_s && qe>=interval_e) if(qs<=interval_s && qe>=interval_e) {
{ if((paf->buffer[j].el) ||
if(boundary_verify(interval_s, interval_e, &(paf->buffer[j]), bq, bt, &R_INF) == 0) (boundary_verify(interval_s, interval_e, &(paf->buffer[j]), bq, bt, &R_INF) == 0)) {
{
return 1; return 1;
} }
} }
@@ -2147,11 +2112,11 @@ void print_overlaps(ma_hit_t_alloc* paf, long long rLen, long long interval_s, l
void detect_chimeric_reads(ma_hit_t_alloc* paf, long long n_read, uint64_t* readLen, void detect_chimeric_reads(ma_hit_t_alloc* paf, long long n_read, uint64_t* readLen,
ma_sub_t* coverage_cut, float shift_rate, all_ul_t *x) ma_sub_t* coverage_cut, float shift_rate, all_ul_t *x, uint64_t ul_thres)
{ {
double startTime = Get_T(); double startTime = Get_T();
init_aux_table(); init_aux_table();
long long i, rLen, /**cov,**/ n_simple_remove = 0, n_complex_remove = 0, n_complex_remove_real = 0; long long i, rLen, n_simple_remove = 0, n_complex_remove = 0, n_complex_remove_real = 0;
uint32_t interval_s, interval_e; uint32_t interval_s, interval_e;
ma_sub_t max_left, max_right; ma_sub_t max_left, max_right;
kvec_t(char) b_q = {0,0,0}; kvec_t(char) b_q = {0,0,0};
@@ -2164,25 +2129,22 @@ ma_sub_t* coverage_cut, float shift_rate, all_ul_t *x)
max_left.s = max_right.s = rLen; max_left.s = max_right.s = rLen;
max_left.e = max_right.e = 0; max_left.e = max_right.e = 0;
///we just need to check UL alignment here as we only need UL which covers the whole HiFi read
collect_sides(i, paf, x, rLen, &max_left, &max_right); collect_sides(i, paf, x, rLen, &max_left, &max_right, ul_thres);
///collect_sides(&(rev_paf[i]), rLen, &max_left, &max_right); ///collect_sides(&(rev_paf[i]), rLen, &max_left, &max_right);
///that means this read is an end node ///that means this read is an end node
if(max_left.s == rLen || max_right.s == rLen) if(max_left.s == rLen || max_right.s == rLen)
{ {
continue; continue;
} }
collect_contain(&(paf[i]), NULL, rLen, &max_left, &max_right, 0.1);
collect_contain(&(paf[i]), NULL, rLen, &max_left, &max_right, 0.1, x, i);
///collect_contain(&(paf[i]), &(rev_paf[i]), rLen, &max_left, &max_right, 0.1); ///collect_contain(&(paf[i]), &(rev_paf[i]), rLen, &max_left, &max_right, 0.1);
////shift_rate should be (asm_opt.max_ov_diff_final*2) ////shift_rate should be (asm_opt.max_ov_diff_final*2)
///this read is a normal read ///this read is a normal read
if (max_left.e > max_right.s && (max_left.e - max_right.s >= rLen * shift_rate)) if (max_left.e > max_right.s && (max_left.e - max_right.s >= rLen * shift_rate))
{ {
continue; continue;
} }
///simple chimeric reads ///simple chimeric reads
if(max_left.e <= max_right.s) if(max_left.e <= max_right.s)
{ {
@@ -9930,11 +9892,11 @@ int asg_cut_internal(asg_t *g, int max_ext)
void clean_weak_ma_hit_t(ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, long long num_sources) void clean_weak_ma_hit_t(ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, long long num_sources, uint32_t ou_thres)
{ {
double startTime = Get_T(); double startTime = Get_T();
long long i, j, index; long long i, j, index;
uint32_t qn, tn; uint32_t qn, tn, ou;
for (i = 0; i < num_sources; i++) for (i = 0; i < num_sources; i++)
{ {
@@ -9944,9 +9906,9 @@ void clean_weak_ma_hit_t(ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_source
tn = Get_tn(sources[i].buffer[j]); tn = Get_tn(sources[i].buffer[j]);
if(sources[i].buffer[j].del) continue; if(sources[i].buffer[j].del) continue;
ou = (sources[i].buffer[j].bl&((uint32_t)0x3fffffff));
//if this is a weak overlap //if this is a weak overlap
if(sources[i].buffer[j].ml == 0) if((sources[i].buffer[j].ml == 0) && ((ou_thres==((uint32_t)-1)) || (ou < ou_thres)))
{ {
if( if(
!check_weak_ma_hit(&(sources[qn]), reverse_sources, tn, !check_weak_ma_hit(&(sources[qn]), reverse_sources, tn,
@@ -14696,20 +14658,22 @@ ma_hit_t_alloc* sources, R_to_U* ruIndex, int max_hang, int min_ovlp)
kvec_asg_arc_t_warp new_rtg_edges; kvec_asg_arc_t_warp new_rtg_edges;
kv_init(new_rtg_edges.a); kv_init(new_rtg_edges.a);
if(ug == NULL) ug = ma_ug_gen(read_g); if(ug == NULL) {
ug = ma_ug_gen(read_g);
uint32_t i; } else {
for (i = 0; i < ug->u.n; ++i) uint32_t i;
{ for (i = 0; i < ug->u.n; ++i)
ma_utg_t *u = &ug->u.a[i];
if(u->m == 0 || ug->g->seq[i].c == ALTER_LABLE)
{ {
asg_seq_del(ug->g, i); ma_utg_t *u = &ug->u.a[i];
if(ug->u.a[i].m!=0) if(u->m == 0 || ug->g->seq[i].c == ALTER_LABLE)
{ {
ug->u.a[i].m = ug->u.a[i].n = 0; asg_seq_del(ug->g, i);
free(ug->u.a[i].a); if(ug->u.a[i].m!=0)
ug->u.a[i].a = NULL; {
ug->u.a[i].m = ug->u.a[i].n = 0;
free(ug->u.a[i].a);
ug->u.a[i].a = NULL;
}
} }
} }
} }
@@ -31155,22 +31119,50 @@ char *get_outfile_name(char* output_file_name)
return buf; return buf;
} }
void gen_ug_opt_t(ug_opt_t *opt, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, int64_t max_hang, int64_t min_ovlp,
int64_t gap_fuzz, int64_t min_dp, uint64_t* readLen, ma_sub_t *coverage_cut, R_to_U* ruIndex)
{
memset(opt, 0, sizeof((*opt)));
opt->sources = sources; opt->reverse_sources = reverse_sources; opt->max_hang = max_hang;
opt->min_ovlp = min_ovlp; opt->gap_fuzz = gap_fuzz; opt->min_dp = min_dp; opt->readLen = readLen;
opt->coverage_cut = coverage_cut; opt->ruIndex = ruIndex;
}
void create_ul_info(ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, int64_t max_hang, int64_t min_ovlp, int64_t gap_fuzz, void create_ul_info(ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, int64_t max_hang, int64_t min_ovlp, int64_t gap_fuzz,
int64_t min_dp, uint64_t* readLen, ma_sub_t *coverage_cut, R_to_U* ruIndex) int64_t min_dp, uint64_t* readLen, ma_sub_t *coverage_cut, R_to_U* ruIndex)
{ {
ug_opt_t opt; memset(&opt, 0, sizeof(opt)); ug_opt_t opt;
opt.sources = sources; gen_ug_opt_t(&opt, sources, reverse_sources, max_hang, min_ovlp, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex);
opt.reverse_sources = reverse_sources;
opt.max_hang = max_hang;
opt.min_ovlp = min_ovlp;
opt.gap_fuzz = gap_fuzz;
opt.min_dp = min_dp;
opt.readLen = readLen;
opt.coverage_cut = coverage_cut;
opt.ruIndex = ruIndex;
ul_load(&opt); ul_load(&opt);
} }
void rescue_src_ul(ma_hit_t_alloc* src, uint64_t n_read, uint64_t occ)
{
uint64_t k, i;
for (k = 0; k < n_read; k++) {
for (i = 0; i < src[k].length; i++) {
if(!src[k].buffer[i].del) continue;
if(src[k].buffer[i].bl>=occ) src[k].buffer[i].del = 0;
}
}
}
asg_t *gen_init_sg(int32_t min_dp, uint64_t n_read, int64_t mini_overlap_length, int64_t max_hang_length, int64_t gap_fuzz,
ma_hit_t_alloc* src, uint64_t* readLen, R_to_U* ruIndex, bub_label_t *b_mask_t, ma_sub_t** cov, all_ul_t *ul)
{
asg_t *sg = NULL;
if(ul) rescue_src_ul(src, n_read, UL_COV_THRES);
ma_hit_sub(min_dp, src, n_read, readLen, mini_overlap_length, cov);
detect_chimeric_reads(src, n_read, readLen, *cov, asm_opt.max_ov_diff_final*2.0, ul, UL_COV_THRES);
ma_hit_cut(src, n_read, readLen, mini_overlap_length, cov);
ma_hit_flt(src, n_read, *cov, max_hang_length, mini_overlap_length);
ma_hit_contained_advance(src, n_read, *cov, ruIndex, max_hang_length, mini_overlap_length);
sg = ma_sg_gen(src, n_read, *cov, max_hang_length, mini_overlap_length);
asg_arc_del_trans(sg, gap_fuzz);
init_bub_label_t(b_mask_t, MIN(10, asm_opt.thread_num), sg->n_seq);
asm_opt.coverage = get_coverage(src, *cov, n_read);
return sg;
}
void clean_graph( void clean_graph(
int min_dp, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, int min_dp, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
@@ -31184,6 +31176,11 @@ ma_sub_t **coverage_cut_ptr, int debug_g)
ma_sub_t *coverage_cut = *coverage_cut_ptr; ma_sub_t *coverage_cut = *coverage_cut_ptr;
asg_t *sg = *sg_ptr; asg_t *sg = *sg_ptr;
bub_label_t b_mask_t; bub_label_t b_mask_t;
ug_opt_t uopt;
if(asm_opt.ar) {
gen_ug_opt_t(&uopt, sources, reverse_sources, max_hang_length, mini_overlap_length, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex);
}
if(debug_g) if(debug_g)
{ {
init_bub_label_t(&b_mask_t, MIN(10, asm_opt.thread_num), sg->n_seq); init_bub_label_t(&b_mask_t, MIN(10, asm_opt.thread_num), sg->n_seq);
@@ -31203,15 +31200,14 @@ ma_sub_t **coverage_cut_ptr, int debug_g)
{ {
memset(R_INF.trio_flag, AMBIGU, R_INF.total_reads*sizeof(uint8_t)); memset(R_INF.trio_flag, AMBIGU, R_INF.total_reads*sizeof(uint8_t));
} }
if(asm_opt.ar) { ///should recover edges from sources by using UL alignments
create_ul_info(sources, reverse_sources, max_hang_length, mini_overlap_length, gap_fuzz, if(asm_opt.ar) create_ul_info(sources, reverse_sources, max_hang_length, mini_overlap_length, gap_fuzz, min_dp, readLen, coverage_cut, ruIndex);
min_dp, readLen, coverage_cut, ruIndex);
exit(1); clean_weak_ma_hit_t(sources, reverse_sources, n_read, asm_opt.ar?UL_COV_THRES:(uint32_t)-1);
} else { sg = gen_init_sg(min_dp, n_read, mini_overlap_length, max_hang_length, gap_fuzz, sources, readLen, ruIndex,
// sg = build_init_sg(sources, reverse_sources, n_read, min_dp, readLen, mini_overlap_length, max_hang_length, &b_mask_t, &coverage_cut, asm_opt.ar?&UL_INF:NULL);
// coverage_cut, ruIndex); // if(asm_opt.ar) exit(1);
clean_weak_ma_hit_t(sources, reverse_sources, n_read); /**
}
///print_binned_reads(sources, n_read, coverage_cut); ///print_binned_reads(sources, n_read, coverage_cut);
///ma_hit_sub is just use to init coverage_cut, ///ma_hit_sub is just use to init coverage_cut,
@@ -31229,7 +31225,7 @@ ma_sub_t **coverage_cut_ptr, int debug_g)
init_bub_label_t(&b_mask_t, MIN(10, asm_opt.thread_num), sg->n_seq); init_bub_label_t(&b_mask_t, MIN(10, asm_opt.thread_num), sg->n_seq);
asg_arc_del_trans(sg, gap_fuzz); asg_arc_del_trans(sg, gap_fuzz);
asm_opt.coverage = get_coverage(sources, coverage_cut, n_read); asm_opt.coverage = get_coverage(sources, coverage_cut, n_read);
**/
if(VERBOSE >= 1) if(VERBOSE >= 1)
{ {
char* unlean_name = (char*)malloc(strlen(output_file_name)+25); char* unlean_name = (char*)malloc(strlen(output_file_name)+25);
@@ -31237,8 +31233,9 @@ ma_sub_t **coverage_cut_ptr, int debug_g)
output_read_graph(sg, coverage_cut, unlean_name, n_read); output_read_graph(sg, coverage_cut, unlean_name, n_read);
free(unlean_name); free(unlean_name);
} }
ul_clean_gfa(sg, sources, reverse_sources, ruIndex, clean_round, min_ovlp_drop_ratio, max_ovlp_drop_ratio, ul_clean_gfa(&uopt, sg, sources, reverse_sources, ruIndex, clean_round, min_ovlp_drop_ratio, max_ovlp_drop_ratio,
0.6, asm_opt.max_short_tip, &b_mask_t, !!asm_opt.ar, ha_opt_triobin(&asm_opt)); 0.6, asm_opt.max_short_tip, &b_mask_t, !!asm_opt.ar, ha_opt_triobin(&asm_opt), UL_COV_THRES);
print_debug_gfa(sg, NULL, coverage_cut, "UL.debug", sources, ruIndex, max_hang_length, mini_overlap_length);
/** /**
asg_cut_tip(sg, asm_opt.max_short_tip); asg_cut_tip(sg, asm_opt.max_short_tip);
///debug_info_of_specfic_node("m64043_200505_112554/8849050/ccs", sg, "inner_1"); ///debug_info_of_specfic_node("m64043_200505_112554/8849050/ccs", sg, "inner_1");
@@ -31380,13 +31377,13 @@ ma_sub_t **coverage_cut_ptr, int debug_g)
if(asm_opt.flag & HA_F_PARTITION) asm_opt.flag -= HA_F_PARTITION; if(asm_opt.flag & HA_F_PARTITION) asm_opt.flag -= HA_F_PARTITION;
output_poly_trio(sg, coverage_cut, o_file, sources, reverse_sources, (asm_opt.max_short_tip*2), 0.15, 3, ruIndex, output_poly_trio(sg, coverage_cut, o_file, sources, reverse_sources, (asm_opt.max_short_tip*2), 0.15, 3, ruIndex,
0.05, 0.9, max_hang_length, mini_overlap_length, 0, &b_mask_t, asm_opt.polyploidy); 0.05, 0.9, max_hang_length, mini_overlap_length, 0, &b_mask_t, asm_opt.polyploidy);
} }/**
else if(asm_opt.ar) else if(asm_opt.ar)
{ {
if(asm_opt.flag & HA_F_PARTITION) asm_opt.flag -= HA_F_PARTITION; if(asm_opt.flag & HA_F_PARTITION) asm_opt.flag -= HA_F_PARTITION;
output_ul_graph(sg, coverage_cut, o_file, sources, reverse_sources, (asm_opt.max_short_tip*2), output_ul_graph(sg, coverage_cut, o_file, sources, reverse_sources, (asm_opt.max_short_tip*2),
0.15, 3, ruIndex, 0.05, 0.9, max_hang_length, mini_overlap_length, gap_fuzz, &b_mask_t); 0.15, 3, ruIndex, 0.05, 0.9, max_hang_length, mini_overlap_length, gap_fuzz, &b_mask_t);
} }**/
else if (ha_opt_triobin(&asm_opt) && ha_opt_hic(&asm_opt)) else if (ha_opt_triobin(&asm_opt) && ha_opt_hic(&asm_opt))
{ {
if(asm_opt.flag & HA_F_PARTITION) asm_opt.flag -= HA_F_PARTITION; if(asm_opt.flag & HA_F_PARTITION) asm_opt.flag -= HA_F_PARTITION;
@@ -31424,7 +31421,7 @@ ma_sub_t **coverage_cut_ptr, int debug_g)
*coverage_cut_ptr = coverage_cut; *coverage_cut_ptr = coverage_cut;
*sg_ptr = sg; *sg_ptr = sg;
destory_bub_label_t(&b_mask_t); destory_bub_label_t(&b_mask_t);
free(o_file); free(o_file); if(asm_opt.ar) destory_all_ul_t(&UL_INF);
fprintf(stderr, "Inconsistency threshold for low-quality regions in BED files: %u%%\n", asm_opt.bed_inconsist_rate); fprintf(stderr, "Inconsistency threshold for low-quality regions in BED files: %u%%\n", asm_opt.bed_inconsist_rate);
} }
+13
View File
@@ -225,6 +225,16 @@ typedef struct {
kvec_t(uint64_t) interval; kvec_t(uint64_t) interval;
} ucov_t; } ucov_t;
typedef struct {
uint32_t u, off, pos;
} utg_rid_dt;
typedef struct {
uint32_t *idx;
kvec_t(utg_rid_dt) p;
asg_t *rg;
} utg_rid_t;
typedef struct { typedef struct {
kvec_t(uint64_t) idx; kvec_t(uint64_t) idx;
kvec_t(utg_ct_t) rids; kvec_t(utg_ct_t) rids;
@@ -242,6 +252,7 @@ typedef struct {
ucov_t *cc; ucov_t *cc;
ucov_t *cr; ucov_t *cr;
ul_contain *ct; ul_contain *ct;
utg_rid_t *r_ug;
// cvert_t *nug; // cvert_t *nug;
// kv_ul_ov_t *ov; // kv_ul_ov_t *ov;
} ul_idx_t; } ul_idx_t;
@@ -1059,6 +1070,8 @@ int asg_topocut_aux(asg_t *g, uint32_t v, int max_ext);
int asg_arc_del_triangular_directly(asg_t *g, long long min_edge_length, int asg_arc_del_triangular_directly(asg_t *g, long long min_edge_length,
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex); ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex);
int asg_arc_del_short_diploid_by_exact(asg_t *g, int max_ext, ma_hit_t_alloc* sources); int asg_arc_del_short_diploid_by_exact(asg_t *g, int max_ext, ma_hit_t_alloc* sources);
uint32_t print_debug_gfa(asg_t *read_g, ma_ug_t *ug, ma_sub_t* coverage_cut, const char* output_file_name,
ma_hit_t_alloc* sources, R_to_U* ruIndex, int max_hang, int min_ovlp);
#define JUNK_COV 5 #define JUNK_COV 5
#define DISCARD_RATE 0.8 #define DISCARD_RATE 0.8
+24 -18
View File
@@ -982,7 +982,7 @@ void append_ul_t(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str,
np->n = id_l; MALLOC(np->a, np->n+1); memcpy(np->a, id, id_l); np->a[id_l] = '\0'; np->n = id_l; MALLOC(np->a, np->n+1); memcpy(np->a, id, id_l); np->a[id_l] = '\0';
} }
if(str) { if(str||str_l) {
if(rid == NULL) { if(rid == NULL) {
kv_pushp(ul_vec_t, *x, &p); kv_pushp(ul_vec_t, *x, &p);
memset(p, 0, sizeof(*p)); memset(p, 0, sizeof(*p));
@@ -994,7 +994,7 @@ void append_ul_t(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str,
} }
p = &(x->a[(*rid)]); p = &(x->a[(*rid)]);
} }
// if((*rid) == 23) fprintf(stderr, "#rid->%lu, on->%ld\n", *rid, on);
p->bb.n = p->N_site.n = p->r_base.n = 0; p->dd = 0; p->bb.n = p->N_site.n = p->r_base.n = 0; p->dd = 0;
p->rlen = str_l; p->rlen = str_l;
@@ -1002,31 +1002,36 @@ void append_ul_t(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str,
if(o == NULL || on == 0) on = 0; if(o == NULL || on == 0) on = 0;
for (i = on-1, st = et = str_l; i >= 0; i--) { for (i = on-1, st = et = str_l; i >= 0; i--) {
z = &(o[i]); z = &(o[i]);
mine = MIN(et, ((int64_t)z->qe)); maxs = MAX(st, ((int64_t)z->qs)); if(z->el) {
ovlp = mine - maxs; mine = MIN(et, ((int64_t)z->qe)); maxs = MAX(st, ((int64_t)z->qs));
ovlp = mine - maxs;
if(ovlp < 0) {///push original bases if(ovlp < 0) {///push original bases
kv_pushp(uc_block_t, p->bb, &b); kv_pushp(uc_block_t, p->bb, &b);
b->hid = 0/**x->mm**/; b->rev = 0; b->base = 1; b->pchain = 0; b->el = 0; b->hid = 0/**x->mm**/; b->rev = 0; b->base = 1; b->pchain = 0; b->el = 0;
b->qe = maxs; b->qs = b->qe + ovlp; bl += (b->qe-b->qs); b->qe = maxs; b->qs = b->qe + ovlp; bl += (b->qe-b->qs);
o_l = (b->qs >= UL_FLANK?UL_FLANK:b->qs); o_l = (b->qs >= UL_FLANK?UL_FLANK:b->qs);
o_r = ((str_l-b->qe)>=UL_FLANK?UL_FLANK:(str_l-b->qe)); o_r = ((str_l-b->qe)>=UL_FLANK?UL_FLANK:(str_l-b->qe));
b->hid |= (o_l<<15); b->hid |= o_r; b->hid |= (o_l<<15); b->hid |= o_r;
b->qs -= o_l; b->qe += o_r; b->qs -= o_l; b->qe += o_r;
b->ts = p->r_base.n; b->te = b->ts + B4L(b->qe-b->qs); b->ts = p->r_base.n; b->te = b->ts + B4L(b->qe-b->qs);
kv_resize(uint8_t, p->r_base, b->te); p->r_base.n = b->te; kv_resize(uint8_t, p->r_base, b->te); p->r_base.n = b->te;
ha_encode_base(p->r_base.a+b->ts, str+b->qs, b->qe-b->qs, &(p->N_site), b->qs); // if(!str) fprintf(stderr, "+rid->%lu\n", *rid);
ha_encode_base(p->r_base.a+b->ts, str+b->qs, b->qe-b->qs, &(p->N_site), b->qs);
}
st = MIN(st, z->qs);
} }
///push ovlp bases ///push ovlp bases
kv_pushp(uc_block_t, p->bb, &b); kv_pushp(uc_block_t, p->bb, &b);
b->hid = (z->tn<<1)>>1; b->rev = z->rev; b->base = 0; b->el = 1; b->hid = (z->tn<<1)>>1; b->rev = z->rev; b->base = 0; b->el = z->el;
b->pchain = ((z->tn&((uint32_t)(0x80000000)))?1:0); b->pchain = ((z->tn&((uint32_t)(0x80000000)))?1:0);
b->qs = z->qs; b->qe = z->qe; b->qs = z->qs; b->qe = z->qe;
b->ts = z->ts; b->te = z->te; b->ts = z->ts; b->te = z->te;
if(b->pchain) pc++; if(b->pchain) pc++;
st = MIN(st, z->qs); // st = MIN(st, z->qs);
} }
if(st > 0) {///push original bases if(st > 0) {///push original bases
@@ -1039,6 +1044,7 @@ void append_ul_t(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str,
b->qs -= o_l; b->qe += o_r; b->qs -= o_l; b->qe += o_r;
b->ts = p->r_base.n; b->te = b->ts + B4L(b->qe-b->qs); b->ts = p->r_base.n; b->te = b->ts + B4L(b->qe-b->qs);
kv_resize(uint8_t, p->r_base, b->te); p->r_base.n = b->te; kv_resize(uint8_t, p->r_base, b->te); p->r_base.n = b->te;
// if(!str) fprintf(stderr, "-rid->%lu, st->%ld, str_l->%ld\n", *rid, st, str_l);
ha_encode_base(p->r_base.a+b->ts, str+b->qs, b->qe-b->qs, &(p->N_site), b->qs); ha_encode_base(p->r_base.a+b->ts, str+b->qs, b->qe-b->qs, &(p->N_site), b->qs);
// push_subblock_original_bases(str, x, p, end, str_l, 321);//for debug // push_subblock_original_bases(str, x, p, end, str_l, 321);//for debug
} }
@@ -1335,7 +1341,7 @@ uint64_t retrieve_u_cov_region(const ul_idx_t *ul, uint64_t id, uint8_t strand,
if(e>=(a[k]>>32) && e<(a[k+1]>>32)) break; if(e>=(a[k]>>32) && e<(a[k+1]>>32)) break;
} }
// if(s == 54201 && e == 58376 && id == 492) fprintf(stderr, "tcc:%lu\n", tcc);
return tcc; return tcc;
} }
+107 -25
View File
@@ -7,6 +7,7 @@
#include "gfa_ut.h" #include "gfa_ut.h"
#include "CommandLines.h" #include "CommandLines.h"
#include "Correct.h" #include "Correct.h"
#include "inter.h"
#define generic_key(x) (x) #define generic_key(x) (x)
KRADIX_SORT_INIT(srt64, uint64_t, generic_key, 8) KRADIX_SORT_INIT(srt64, uint64_t, generic_key, 8)
@@ -169,7 +170,7 @@ uint32_t asg_arc_cut_tips(asg_t *g, uint32_t max_ext, asg64_v *in, uint32_t is_o
} }
if(mm_ou == (uint32_t)-1) mm_ou = 0; if(mm_ou == (uint32_t)-1) mm_ou = 0;
kv += mm_ou; i += mm_ou; kv += mm_ou; i += mm_ou;
if(i < max_ext + (!!is_ou)) kv_push(uint64_t, *b, (((uint64_t)kv)<<32)|v); if(i < max_ext/** + (!!is_ou)**/) kv_push(uint64_t, *b, (((uint64_t)kv)<<32)|v);
} }
radix_sort_srt64(b->a, b->a + b->n); radix_sort_srt64(b->a, b->a + b->n);
@@ -194,7 +195,7 @@ uint32_t asg_arc_cut_tips(asg_t *g, uint32_t max_ext, asg64_v *in, uint32_t is_o
if(mm_ou == (uint32_t)-1) mm_ou = 0; if(mm_ou == (uint32_t)-1) mm_ou = 0;
i += mm_ou; i += mm_ou;
if(i < max_ext + (!!is_ou)) { if(i < max_ext/** + (!!is_ou)**/) {
for (i = pb; i < b->n; i++) asg_seq_del(g, ((uint32_t)b->a[i])>>1); for (i = pb; i < b->n; i++) asg_seq_del(g, ((uint32_t)b->a[i])>>1);
cnt++; cnt++;
} }
@@ -239,11 +240,13 @@ static void update_sg_uo_t(void *data, long i, int tid)
ma_hit_t_alloc *src = sl->src; asg_t *g = sl->g; ma_hit_t_alloc *src = sl->src; asg_t *g = sl->g;
asg_arc_t *e = &(g->arc[i]); uint32_t k, qn, tn; asg_arc_t *e = &(g->arc[i]); uint32_t k, qn, tn;
ma_hit_t_alloc *x = &(src[e->ul>>33]); ma_hit_t_alloc *x = &(src[e->ul>>33]);
e->ou = 0;
if(e->del) return;
for (k = 0; k < x->length; k++) { for (k = 0; k < x->length; k++) {
qn = Get_qn(x->buffer[k]); qn = Get_qn(x->buffer[k]);
tn = Get_tn(x->buffer[k]); tn = Get_tn(x->buffer[k]);
if(qn == (e->ul>>33) && tn == (e->v>>1)) { if(qn == (e->ul>>33) && tn == (e->v>>1)) {
e->ou = (x->buffer[k].bl&OU_MASK); e->ou = (x->buffer[k].bl>OU_MASK?OU_MASK:x->buffer[k].bl);
break; break;
} }
} }
@@ -254,6 +257,32 @@ void update_sg_uo(asg_t *g, ma_hit_t_alloc *src)
{ {
sset_aux s; s.g = g; s.src = src; sset_aux s; s.g = g; s.src = src;
kt_for(asm_opt.thread_num, update_sg_uo_t, &s, g->n_arc); kt_for(asm_opt.thread_num, update_sg_uo_t, &s, g->n_arc);
uint32_t k, z, nv, occ_a = 0, occ_n = 0; asg_arc_t *av = NULL;
for (k = 0; k < g->n_seq; k++) {
if(g->seq[k].del) continue;
occ_n++;
av = asg_arc_a(g, (k<<1)); nv = asg_arc_n(g, (k<<1));
for (z = 0; z < nv; z++) {
if(av[z].del || av[z].ou == 0) continue;
break;
}
if(z < nv) {
occ_a++;
continue;
}
av = asg_arc_a(g, ((k<<1)+1)); nv = asg_arc_n(g, ((k<<1)+1));
for (z = 0; z < nv; z++) {
if(av[z].del || av[z].ou == 0) continue;
break;
}
if(z < nv) {
occ_a++;
}
}
fprintf(stderr, "[M::%s::] ==> # gfa reads:%u, # covered gfa reads:%u\n", __func__, occ_n, occ_a);
} }
int32_t if_sup_chimeric(ma_hit_t_alloc* src, uint64_t rLen, asg64_v *b, int if_exact) int32_t if_sup_chimeric(ma_hit_t_alloc* src, uint64_t rLen, asg64_v *b, int if_exact)
@@ -328,7 +357,7 @@ int32_t if_sup_chimeric(ma_hit_t_alloc* src, uint64_t rLen, asg64_v *b, int if_e
} }
///remove single node ///remove single node
void asg_arc_cut_chimeric(asg_t *g, ma_hit_t_alloc* src, asg64_v *in) void asg_arc_cut_chimeric(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t ou_thres)
{ {
asg64_v tx = {0,0,0}, *b = NULL; asg64_v tx = {0,0,0}, *b = NULL;
uint32_t v, w, ei[2] = {0}, k, i, n_vtx = g->n_seq<<1; uint32_t v, w, ei[2] = {0}, k, i, n_vtx = g->n_seq<<1;
@@ -344,7 +373,8 @@ void asg_arc_cut_chimeric(asg_t *g, ma_hit_t_alloc* src, asg64_v *in)
assert((g->arc[ei[0]].ul>>32) == v && (g->arc[ei[1]].ul>>32) == (v^1)); assert((g->arc[ei[0]].ul>>32) == v && (g->arc[ei[1]].ul>>32) == (v^1));
if((get_arcs(g, g->arc[ei[0]].v^1, NULL, 0)<2) || (get_arcs(g, g->arc[ei[1]].v^1, NULL, 0)<2)) continue; if((get_arcs(g, g->arc[ei[0]].v^1, NULL, 0)<2) || (get_arcs(g, g->arc[ei[1]].v^1, NULL, 0)<2)) continue;
if(g->arc[ei[0]].el) continue; if(g->arc[ei[0]].el) continue;
if(!if_sup_chimeric(&(src[v>>1]), g->seq[v>>1].len, b, 1)) continue; if(ou_thres!=(uint32_t)-1&&g->arc[ei[0]].ou>=ou_thres&&g->arc[ei[1]].ou>=ou_thres) continue;///UL
if(!if_sup_chimeric(&(src[v>>1]), g->seq[v>>1].len, b, 1)) continue;///HiFi
kv_push(uint64_t, *b, (((uint64_t)(g->arc[ei[0]].ol))<<32)|((uint64_t)(ei[0]))); kv_push(uint64_t, *b, (((uint64_t)(g->arc[ei[0]].ol))<<32)|((uint64_t)(ei[0])));
} }
} }
@@ -429,9 +459,8 @@ void asg_arc_cut_inexact(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, int32_t max
} }
///mm_ol and mm_ou are used to make edge with long indel more easy to be cutted ///mm_ol and mm_ou are used to make edge with long indel more easy to be cutted
mm_ol = MIN(ve->ol, we->ol); mm_ou = MIN(ve->ou, we->ou); mm_ol = MIN(ve->ol, we->ol); mm_ou = MIN(ve->ou, we->ou);
for (i = kv = ol_max = ou_max = 0, /**ve =**/ vmax = NULL; i < nv; ++i) { for (i = kv = ol_max = ou_max = 0, vmax = NULL; i < nv; ++i) {
if(av[i].del) continue; if(av[i].del) continue;
// if(av[i].v == (w^1)) ve = &(av[i]);
kv++; kv++;
if(is_trio && get_tip_trio_infor(g, av[i].v) == ntrioF) continue; if(is_trio && get_tip_trio_infor(g, av[i].v) == ntrioF) continue;
if(ol_max < av[i].ol) ol_max = av[i].ol, vmax = &(av[i]); if(ol_max < av[i].ol) ol_max = av[i].ol, vmax = &(av[i]);
@@ -442,13 +471,12 @@ void asg_arc_cut_inexact(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, int32_t max
// } // }
if (kv < 1) continue; if (kv < 1) continue;
if (kv >= 2) { if (kv >= 2) {
if (/**ve->ol**/mm_ol >= ol_max) continue; if (mm_ol >= ol_max) continue;
if (is_ou && /**ve->ou**/mm_ou >= ou_max) continue; if (is_ou && mm_ou >= ou_max) continue;
} }
for (i = kw = ol_max = ou_max = 0/**, we = NULL**/; i < nw; ++i) { for (i = kw = ol_max = ou_max = 0; i < nw; ++i) {
if(aw[i].del) continue; if(aw[i].del) continue;
// if(aw[i].v == (v^1)) we = &(aw[i]);
kw++; kw++;
if(is_trio && get_tip_trio_infor(g, aw[i].v) == ntrioF) continue; if(is_trio && get_tip_trio_infor(g, aw[i].v) == ntrioF) continue;
if(ol_max < aw[i].ol) ol_max = aw[i].ol; if(ol_max < aw[i].ol) ol_max = aw[i].ol;
@@ -459,8 +487,8 @@ void asg_arc_cut_inexact(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, int32_t max
// } // }
if (kw < 1) continue; if (kw < 1) continue;
if (kw >= 2) { if (kw >= 2) {
if (/**we->ol**/mm_ol >= ol_max) continue; if (mm_ol >= ol_max) continue;
if (is_ou && /**we->ou**/mm_ou >= ou_max) continue; if (is_ou && mm_ou >= ou_max) continue;
} }
if (kv <= 1 && kw <= 1) continue; if (kv <= 1 && kw <= 1) continue;
@@ -698,6 +726,7 @@ uint32_t is_topo, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len)
b->n = 0; b->n = 0;
for (v = 0; v < n_vtx; ++v) { for (v = 0; v < n_vtx; ++v) {
// if((v>>1)==17078) fprintf(stderr, "[M::%s::] v:%u, del:%u, seq_vis:%u\n", __func__, v, g->seq[v>>1].del, g->seq_vis[v]);
if (g->seq[v>>1].del) continue; if (g->seq[v>>1].del) continue;
if(g->seq_vis[v] == 0) { if(g->seq_vis[v] == 0) {
av = asg_arc_a(g, v); nv = asg_arc_n(g, v); av = asg_arc_a(g, v); nv = asg_arc_n(g, v);
@@ -711,6 +740,11 @@ uint32_t is_topo, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len)
for (i = 0; i < nv; ++i) { for (i = 0; i < nv; ++i) {
if(av[i].del) continue; if(av[i].del) continue;
// if((av[i].ul>>33)==287) {
// fprintf(stderr, "++++++%.*s(%c)\t%.*s(%c)\tol:%u\tou:%u\n",
// (int32_t)Get_NAME_LENGTH(R_INF, (av[i].ul>>33)), Get_NAME(R_INF, (av[i].ul>>33)), "+-"[(av[i].ul>>32)&1],
// (int32_t)Get_NAME_LENGTH(R_INF, (av[i].v>>1)), Get_NAME(R_INF, (av[i].v>>1)), "+-"[av[i].v&1], av[i].ol, av[i].ou);
// }
if(max_drop_len && av[i].ol >= (*max_drop_len)) continue; if(max_drop_len && av[i].ol >= (*max_drop_len)) continue;
kv_push(uint64_t, *b, (((uint64_t)av[i].ol)<<32) | ((uint64_t)(av-g->arc+i))); kv_push(uint64_t, *b, (((uint64_t)av[i].ol)<<32) | ((uint64_t)(av-g->arc+i)));
} }
@@ -747,6 +781,11 @@ uint32_t is_topo, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len)
for (i = kv = ol_max = ou_max = 0, /**ve =**/ vl_max = NULL; i < nv; ++i) { for (i = kv = ol_max = ou_max = 0, /**ve =**/ vl_max = NULL; i < nv; ++i) {
if(av[i].del) continue; if(av[i].del) continue;
// if(av[i].v == (w^1)) ve = &(av[i]); // if(av[i].v == (w^1)) ve = &(av[i]);
// if((av[i].ul>>33)==287) {
// fprintf(stderr, "++++++%.*s(%c)\t%.*s(%c)\tol:%u\tou:%u\n",
// (int32_t)Get_NAME_LENGTH(R_INF, (av[i].ul>>33)), Get_NAME(R_INF, (av[i].ul>>33)), "+-"[(av[i].ul>>32)&1],
// (int32_t)Get_NAME_LENGTH(R_INF, (av[i].v>>1)), Get_NAME(R_INF, (av[i].v>>1)), "+-"[av[i].v&1], av[i].ol, av[i].ou);
// }
kv++; kv++;
if(is_trio && get_tip_trio_infor(g, av[i].v) == ntrioF) continue; if(is_trio && get_tip_trio_infor(g, av[i].v) == ntrioF) continue;
if(ol_max < av[i].ol) ol_max = av[i].ol, vl_max = &(av[i]); if(ol_max < av[i].ol) ol_max = av[i].ol, vl_max = &(av[i]);
@@ -754,14 +793,13 @@ uint32_t is_topo, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len)
} }
if (kv < 1) continue; if (kv < 1) continue;
if (kv >= 2) { if (kv >= 2) {
if (/**ve->ol**/mm_ol > ol_max*len_rat) continue; if (mm_ol > ol_max*len_rat) continue;
if (is_ou && /**ve->ou**/mm_ou > ou_max*ou_rat) continue; if (is_ou && mm_ou > ou_max*ou_rat) continue;
} }
for (i = kw = ol_max = ou_max = 0, /**we =**/ wl_max = NULL; i < nw; ++i) { for (i = kw = ol_max = ou_max = 0, wl_max = NULL; i < nw; ++i) {
if(aw[i].del) continue; if(aw[i].del) continue;
// if(aw[i].v == (v^1)) we = &(aw[i]);
kw++; kw++;
if(is_trio && get_tip_trio_infor(g, aw[i].v) == ntrioF) continue; if(is_trio && get_tip_trio_infor(g, aw[i].v) == ntrioF) continue;
if(ol_max < aw[i].ol) ol_max = aw[i].ol, wl_max = &(aw[i]); if(ol_max < aw[i].ol) ol_max = aw[i].ol, wl_max = &(aw[i]);
@@ -769,8 +807,8 @@ uint32_t is_topo, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len)
} }
if (kw < 1) continue; if (kw < 1) continue;
if (kw >= 2) { if (kw >= 2) {
if (/**we->ol**/mm_ol > ol_max*len_rat) continue; if (mm_ol > ol_max*len_rat) continue;
if (is_ou && /**we->ou**/mm_ou > ou_max*ou_rat) continue; if (is_ou && mm_ou > ou_max*ou_rat) continue;
} }
if (kv <= 1 && kw <= 1) continue; if (kv <= 1 && kw <= 1) continue;
@@ -1234,8 +1272,47 @@ void debug_edges(asg64_v *dbg, uint32_t *l, uint32_t l_n) {
} }
} }
void ul_clean_gfa(asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, int64_t clean_round, double min_ovlp_drop_ratio, double max_ovlp_drop_ratio, void print_node(asg_t *sg, uint32_t src)
double ou_drop_rate, int64_t max_tip, bub_label_t *b_mask_t, int32_t is_ou, int32_t is_trio) {
asg_arc_t *av; uint32_t nv, v, i;
v = src<<1;
av = asg_arc_a(sg, v); nv = asg_arc_n(sg, v);
fprintf(stderr, "\n%.*s(%c)\tnv:%u\n",
(int32_t)Get_NAME_LENGTH(R_INF, v>>1), Get_NAME(R_INF, v>>1), "+-"[v&1], nv);
for (i = 0; i < nv; i++) {
fprintf(stderr, "++++++%.*s(%c)<id:%lu>\t%.*s(%c)<id:%u>\tol:%u\tou:%u\tdel:%u\n",
(int32_t)Get_NAME_LENGTH(R_INF, (av[i].ul>>33)), Get_NAME(R_INF, (av[i].ul>>33)), "+-"[(av[i].ul>>32)&1], av[i].ul>>33,
(int32_t)Get_NAME_LENGTH(R_INF, (av[i].v>>1)), Get_NAME(R_INF, (av[i].v>>1)), "+-"[av[i].v&1], av[i].v>>1, av[i].ol, av[i].ou, av[i].del);
}
v = (src<<1)+1;
av = asg_arc_a(sg, v); nv = asg_arc_n(sg, v);
fprintf(stderr, "\n%.*s(%c)\tnv:%u\n",
(int32_t)Get_NAME_LENGTH(R_INF, v>>1), Get_NAME(R_INF, v>>1), "+-"[v&1], nv);
for (i = 0; i < nv; i++) {
fprintf(stderr, "------%.*s(%c)<id:%lu>\t%.*s(%c)<id:%u>\tol:%u\tou:%u\tdel:%u\n",
(int32_t)Get_NAME_LENGTH(R_INF, (av[i].ul>>33)), Get_NAME(R_INF, (av[i].ul>>33)), "+-"[(av[i].ul>>32)&1], av[i].ul>>33,
(int32_t)Get_NAME_LENGTH(R_INF, (av[i].v>>1)), Get_NAME(R_INF, (av[i].v>>1)), "+-"[av[i].v&1], av[i].v>>1, av[i].ol, av[i].ou, av[i].del);
}
}
void print_vw_edge(asg_t *sg, uint32_t v, uint32_t w, const char *cmd)
{
asg_arc_t *av; uint32_t nv, i;
av = asg_arc_a(sg, v); nv = asg_arc_n(sg, v);
for (i = 0; i < nv; i++) {
if(av[i].v == w) {
fprintf(stderr, "[%s]\t%.*s(%c)<id:%lu>\t%.*s(%c)<id:%u>\tol:%u\tou:%u\tdel:%u\n", cmd,
(int32_t)Get_NAME_LENGTH(R_INF, (av[i].ul>>33)), Get_NAME(R_INF, (av[i].ul>>33)), "+-"[(av[i].ul>>32)&1], av[i].ul>>33,
(int32_t)Get_NAME_LENGTH(R_INF, (av[i].v>>1)), Get_NAME(R_INF, (av[i].v>>1)), "+-"[av[i].v&1], av[i].v>>1, av[i].ol, av[i].ou, av[i].del);
break;
}
}
if(i >= nv) fprintf(stderr, "[%s]\tno edges\n", cmd);
}
void ul_clean_gfa(ug_opt_t *uopt, asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, int64_t clean_round, double min_ovlp_drop_ratio, double max_ovlp_drop_ratio,
double ou_drop_rate, int64_t max_tip, bub_label_t *b_mask_t, int32_t is_ou, int32_t is_trio, uint32_t ou_thres)
{ {
#define HARD_OU_DROP 0.75 #define HARD_OU_DROP 0.75
#define HARD_OL_DROP 0.6 #define HARD_OL_DROP 0.6
@@ -1250,16 +1327,16 @@ double ou_drop_rate, int64_t max_tip, bub_label_t *b_mask_t, int32_t is_ou, int3
asg_arc_cut_tips(sg, max_tip, &bu, is_ou); asg_arc_cut_tips(sg, max_tip, &bu, is_ou);
for (i = 0; i < clean_round; i++, drop += step) { for (i = 0; i < clean_round; i++, drop += step) {
if(drop > max_ovlp_drop_ratio) drop = max_ovlp_drop_ratio; if(drop > max_ovlp_drop_ratio) drop = max_ovlp_drop_ratio;
// fprintf(stderr, "(0):i->%ld, drop->%f\n", i, drop);
// print_vw_edge(sg, 34156, 34090, "0");
// stats_chimeric(sg, src, &bu); // stats_chimeric(sg, src, &bu);
if(!is_ou) asg_iterative_semi_circ(sg, src, &bu, max_tip, 1); if(!is_ou) asg_iterative_semi_circ(sg, src, &bu, max_tip, 1);
// fprintf(stderr, "(0):i->%ld, drop->%f\n", i, drop);
asg_arc_identify_simple_bubbles_multi(sg, b_mask_t, 1); asg_arc_identify_simple_bubbles_multi(sg, b_mask_t, 1);
asg_arc_cut_chimeric(sg, src, &bu); asg_arc_cut_chimeric(sg, src, &bu, is_ou?ou_thres:(uint32_t)-1);
asg_arc_cut_tips(sg, max_tip, &bu, is_ou); asg_arc_cut_tips(sg, max_tip, &bu, is_ou);
asg_arc_identify_simple_bubbles_multi(sg, b_mask_t, 0); asg_arc_identify_simple_bubbles_multi(sg, b_mask_t, 0);
// print_edge(sg->arc+45471, "a");
asg_arc_cut_inexact(sg, src, &bu, max_tip, is_ou, is_trio/**, NULL**//**&dbg**/); asg_arc_cut_inexact(sg, src, &bu, max_tip, is_ou, is_trio/**, NULL**//**&dbg**/);
// debug_edges(&dbg, d, 2); // debug_edges(&dbg, d, 2);
asg_arc_cut_tips(sg, max_tip, &bu, is_ou); asg_arc_cut_tips(sg, max_tip, &bu, is_ou);
@@ -1275,6 +1352,7 @@ double ou_drop_rate, int64_t max_tip, bub_label_t *b_mask_t, int32_t is_ou, int3
asg_arc_cut_complex_bub_links(sg, &bu, HARD_OL_DROP, HARD_OU_DROP, is_ou, b_mask_t); asg_arc_cut_complex_bub_links(sg, &bu, HARD_OL_DROP, HARD_OU_DROP, is_ou, b_mask_t);
asg_arc_cut_tips(sg, max_tip, &bu, is_ou); asg_arc_cut_tips(sg, max_tip, &bu, is_ou);
} }
if(!is_ou) asg_iterative_semi_circ(sg, src, &bu, max_tip, 1); if(!is_ou) asg_iterative_semi_circ(sg, src, &bu, max_tip, 1);
asg_arc_identify_simple_bubbles_multi(sg, b_mask_t, 0); asg_arc_identify_simple_bubbles_multi(sg, b_mask_t, 0);
@@ -1292,5 +1370,9 @@ double ou_drop_rate, int64_t max_tip, bub_label_t *b_mask_t, int32_t is_ou, int3
if(!is_ou) asg_cut_semi_circ(sg, LIM_LEN, 1); if(!is_ou) asg_cut_semi_circ(sg, LIM_LEN, 1);
if(is_ou) ul_refine_alignment(uopt, sg);
// print_node(sg, 17078); //print_node(sg, 8311); print_node(sg, 8294);
free(bu.a); free(bu.a);
} }
+3 -3
View File
@@ -2,11 +2,11 @@
#define __GFA_UT__ #define __GFA_UT__
#include "Overlaps.h" #include "Overlaps.h"
void ul_clean_gfa(asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, int64_t clean_round, double min_ovlp_drop_ratio, double max_ovlp_drop_ratio, void ul_clean_gfa(ug_opt_t *uopt, asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, int64_t clean_round, double min_ovlp_drop_ratio, double max_ovlp_drop_ratio,
double ou_drop_rate, int64_t max_tip, bub_label_t *b_mask_t, int32_t is_ou, int32_t is_trio); double ou_drop_rate, int64_t max_tip, bub_label_t *b_mask_t, int32_t is_ou, int32_t is_trio, uint32_t ou_thres);
uint32_t asg_arc_cut_tips(asg_t *g, uint32_t max_ext, asg64_v *in, uint32_t is_ou); uint32_t asg_arc_cut_tips(asg_t *g, uint32_t max_ext, asg64_v *in, uint32_t is_ou);
void asg_iterative_semi_circ(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t normal_len, uint32_t pop_chimer, asg64_v *dbg); void asg_iterative_semi_circ(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t normal_len, uint32_t pop_chimer, asg64_v *dbg);
void asg_arc_cut_chimeric(asg_t *g, ma_hit_t_alloc* src, asg64_v *in); void asg_arc_cut_chimeric(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t ou_thres);
void asg_arc_cut_inexact(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, int32_t max_ext, uint32_t is_ou, uint32_t is_trio); void asg_arc_cut_inexact(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, int32_t max_ext, uint32_t is_ou, uint32_t is_trio);
void asg_arc_cut_length(asg_t *g, asg64_v *in, int32_t max_ext, float len_rat, float ou_rat, uint32_t is_ou, uint32_t is_trio, void asg_arc_cut_length(asg_t *g, asg64_v *in, int32_t max_ext, float len_rat, float ou_rat, uint32_t is_ou, uint32_t is_trio,
uint32_t is_topo, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len); uint32_t is_topo, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len);
+1 -1
View File
@@ -6,7 +6,7 @@
#include "CommandLines.h" #include "CommandLines.h"
typedef struct { typedef struct {
int n, m; size_t n, m;
uint64_t *a; uint64_t *a;
} st_mt_t; } st_mt_t;
+1603 -109
View File
File diff suppressed because it is too large Load Diff
+1
View File
@@ -6,5 +6,6 @@
void ul_resolve(ma_ug_t *ug, const asg_t *rg, const ug_opt_t *uopt, int hap_n); void ul_resolve(ma_ug_t *ug, const asg_t *rg, const ug_opt_t *uopt, int hap_n);
void ul_load(const ug_opt_t *uopt); void ul_load(const ug_opt_t *uopt);
uint64_t* get_hifi2ul_list(all_ul_t *x, uint64_t hid, uint64_t* a_n); uint64_t* get_hifi2ul_list(all_ul_t *x, uint64_t hid, uint64_t* a_n);
void ul_refine_alignment(const ug_opt_t *uopt, asg_t *sg);
#endif #endif
+4 -4
View File
@@ -158,7 +158,7 @@ void debug_pl(const char *str, int len, int w, int k, int is_hpc, ha_mz1_v *p, c
y = yak_hash64_64(kmer[z<<1|0]) + yak_hash64_64(kmer[z<<1|1]); y = yak_hash64_64(kmer[z<<1|0]) + yak_hash64_64(kmer[z<<1|1]);
cnt = hf? ha_ft_cnt(hf, y) : 0; cnt = hf? ha_ft_cnt(hf, y) : 0;
for (dbi = 0; dbi < mt->n; dbi++) for (dbi = 0; dbi < (int32_t)mt->n; dbi++)
{ {
if(p->a[dbi].x == y && p->a[dbi].rid == cnt && p->a[dbi].pos == i && p->a[dbi].rev == z && p->a[dbi].span == kmer_span) if(p->a[dbi].x == y && p->a[dbi].rid == cnt && p->a[dbi].pos == i && p->a[dbi].rev == z && p->a[dbi].span == kmer_span)
{ {
@@ -170,9 +170,9 @@ void debug_pl(const char *str, int len, int w, int k, int is_hpc, ha_mz1_v *p, c
} else l = 0, tq.count = tq.front = 0, kmer_span = 0; } else l = 0, tq.count = tq.front = 0, kmer_span = 0;
} }
if(dbcnt != mt->n) fprintf(stderr, "ERROR\n"); if(dbcnt != (int32_t)mt->n) fprintf(stderr, "ERROR\n");
if(mt->n != (int)p->n) fprintf(stderr, "ERROR\n"); if(mt->n != p->n) fprintf(stderr, "ERROR\n");
for (dbi = 1; dbi < mt->n; dbi++) for (dbi = 1; dbi < (int32_t)mt->n; dbi++)
{ {
if(p->a[dbi].pos <= p->a[dbi-1].pos || (int)mt->a[dbi] <= (int)mt->a[dbi-1]) if(p->a[dbi].pos <= p->a[dbi-1].pos || (int)mt->a[dbi] <= (int)mt->a[dbi-1])
{ {