debug mbg

This commit is contained in:
chhylp123
2021-05-30 09:02:25 -04:00
parent a39f01f4d8
commit e774a83be2
18 changed files with 2497 additions and 207 deletions
+54
View File
@@ -14,6 +14,8 @@
void ha_get_candidates_interface(ha_abuf_t *ab, int64_t rid, UC_Read *ucr, overlap_region_alloc *overlap_list, overlap_region_alloc *overlap_list_hp, Candidates_list *cl, double bw_thres,
int max_n_chain, int keep_whole_chain, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* chain_idx, ma_hit_t_alloc* paf, ma_hit_t_alloc* rev_paf, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct);
void ha_get_ug_candidates(ha_abuf_t *ab, int64_t rid, ma_utg_t *u, ma_utg_v *ua, overlap_region_alloc *overlap_list, Candidates_list *cl, double bw_thres, int max_n_chain, int keep_whole_chain, kvec_t_u8_warp* k_flag,
kvec_t_u64_warp* chain_idx, void *ha_flt_tab, ha_pt_t *ha_idx, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct, double chain_match_rate);
void ha_sort_list_by_anchor(overlap_region_alloc *overlap_list);
All_reads R_INF;
@@ -435,6 +437,7 @@ typedef struct {
kvec_t_u64_warp r_buf;
kvec_t_u8_warp k_flag;
overlap_region tmp_region;
ma_utg_v *ua;
} ha_ovec_buf_t;
ha_ovec_buf_t *ha_ovec_init(int is_final, int save_ov)
@@ -1602,11 +1605,62 @@ void ha_overlap_final(void)
asm_opt.het_cov = het_cov;
}
static void worker_ov_utg(void *data, long i, int tid)
{
ha_ovec_buf_t *b = ((ha_ovec_buf_t**)data)[tid];
if(b->ua->a[i].len == 0) return;
ha_get_ug_candidates(b->ab, i, &(b->ua->a[i]), b->ua, &b->olist, &b->clist,
0.3, asm_opt.polyploidy*5, 0, &(b->k_flag), &b->r_buf, ha_flt_tab, ha_idx,
&(b->tmp_region), NULL, /**0.3**/0);
overlap_region_sort_y_id(b->olist.list, b->olist.length);
ma_hit_sort_tn(R_INF.paf[i].buffer, R_INF.paf[i].length);
ma_hit_sort_tn(R_INF.reverse_paf[i].buffer, R_INF.reverse_paf[i].length);
update_overlaps(&b->olist, &(R_INF.paf[i]), &b->self_read, &b->ovlp_read, 1, 1);
update_overlaps(&b->olist, &(R_INF.reverse_paf[i]), &b->self_read, &b->ovlp_read, 2, 0);
///recover missing exact overlaps
update_exact_overlaps(&b->olist, &b->self_read, &b->ovlp_read);
///Final_phasing(&overlap_list, &cigarline, &g_read, &overlap_read, c2n);
push_final_overlaps(&(R_INF.paf[i]), R_INF.reverse_paf, &b->olist, 1);
push_final_overlaps(&(R_INF.reverse_paf[i]), R_INF.reverse_paf, &b->olist, 2);
}
void ug_idx_build(ma_ug_t *ug, int hap_n)
{
int flag = asm_opt.flag&HA_F_NO_HPC, i;
asm_opt.flag -= flag;
ha_flt_tab = ha_ft_ug_gen(&asm_opt, &(ug->u), hap_n);
ha_idx = ha_pt_ug_gen(&asm_opt, ha_flt_tab, &(ug->u), hap_n);
ha_ovec_buf_t **b = NULL;
// overlap and correct reads
CALLOC(b, asm_opt.thread_num);
for (i = 0; i < asm_opt.thread_num; ++i)
{
b[i] = ha_ovec_init(1, 1);
b[i]->ua = &(ug->u);
}
kt_for(asm_opt.thread_num, worker_ov_utg, b, R_INF.total_reads);
for (i = 0; i < asm_opt.thread_num; ++i)
ha_ovec_destroy(b[i]);
free(b);
ha_ft_destroy(ha_flt_tab);
ha_pt_destroy(ha_idx);
asm_opt.flag += flag;
exit(1);
}
int ha_assemble(void)
{
// debug_mc_g_t(MC_NAME);
debug_mc_gg_t(MC_NAME, 0, 0);
extern void ha_extract_print_list(const All_reads *rs, int n_rounds, const char *o);
int r, hom_cov = -1, ovlp_loaded = 0;
if (asm_opt.load_index_from_disk && load_all_data_from_disk(&R_INF.paf, &R_INF.reverse_paf, asm_opt.output_file_name)) {
+2
View File
@@ -1,6 +1,7 @@
#ifndef __ASSEMBLY__
#define __ASSEMBLY__
#include "CommandLines.h"
#include "Overlaps.h"
#define FORWARD 0
#define REVERSE_COMPLEMENT (0x8000000000000000)
@@ -14,5 +15,6 @@
#define RESEED_HP_RATE 0.9
int ha_assemble(void);
void ug_idx_build(ma_ug_t *ug, int hap_n);
#endif
+164 -1
View File
@@ -193,6 +193,168 @@ int append_inexact_overlap_region_alloc(overlap_region_alloc* list, overlap_regi
resize_fake_cigar(&(list->list[list->length].f_cigar), (tmp->f_cigar.length + 2));
if(add_beg_end == 1)
{
add_fake_cigar(&(list->list[list->length].f_cigar), list->list[list->length].x_pos_s, 0);
}
long long distance_self_pos = tmp->x_pos_e - tmp->x_pos_s;
long long distance_pos = tmp->y_pos_e - tmp->y_pos_s;
long long init_distance_gap = distance_pos - distance_self_pos;
/****************************may have bugs********************************/
///long long pre_distance_gap = init_distance_gap;
long long pre_distance_gap = 0xfffffffffffffff;
/****************************may have bugs********************************/
long long distance_gap;
long long i = 0;
for (i = tmp->f_cigar.length - 1; i >= 0; i--)
{
distance_gap = get_fake_gap_shift(&(tmp->f_cigar), i);
if(distance_gap != pre_distance_gap)
{
pre_distance_gap = distance_gap;
add_fake_cigar(&(list->list[list->length].f_cigar),
get_fake_gap_pos(&(tmp->f_cigar), i), init_distance_gap - pre_distance_gap);
}
}
if(add_beg_end == 1 && get_fake_gap_pos(&(list->list[list->length].f_cigar),
list->list[list->length].f_cigar.length - 1) != (long long)list->list[list->length].x_pos_e)
{
add_fake_cigar(&(list->list[list->length].f_cigar),
list->list[list->length].x_pos_e,
get_fake_gap_shift(&(list->list[list->length].f_cigar),
list->list[list->length].f_cigar.length - 1));
}
}
list->list[list->length].shared_seed = tmp->shared_seed;
list->list[list->length].align_length = 0;
list->list[list->length].is_match = 0;
list->list[list->length].non_homopolymer_errors = 0;
list->list[list->length].strong = 0;
list->length++;
return 1;
}
int append_utg_inexact_overlap_region_alloc(overlap_region_alloc* list, overlap_region* tmp,
ma_utg_v *ua, int add_beg_end)
{
if (list->length + 1 > list->size)
{
list->size = list->size * 2;
list->list = (overlap_region*)realloc(list->list, sizeof(overlap_region)*list->size);
/// need to set new space to be 0
memset(list->list + (list->size/2), 0, sizeof(overlap_region)*(list->size/2));
}
if (list->length!=0 && list->list[list->length - 1].y_id==tmp->y_id)
{
///if(list->list[list->length - 1].shared_seed >= tmp->shared_seed)
if((list->list[list->length - 1].shared_seed > tmp->shared_seed)
||
((list->list[list->length - 1].shared_seed == tmp->shared_seed) &&
(list->list[list->length - 1].overlapLen <= tmp->overlapLen)))
{
return 0;
}
else
{
list->length--;
}
}
if(tmp->x_pos_s <= tmp->y_pos_s)
{
tmp->y_pos_s = tmp->y_pos_s - tmp->x_pos_s;
tmp->x_pos_s = 0;
}
else
{
tmp->x_pos_s = tmp->x_pos_s - tmp->y_pos_s;
tmp->y_pos_s = 0;
}
long long x_right_length = ua->a[tmp->x_id].len - tmp->x_pos_e - 1;
long long y_right_length = ua->a[tmp->y_id].len - tmp->y_pos_e - 1;
if(x_right_length <= y_right_length)
{
tmp->x_pos_e = ua->a[tmp->x_id].len - 1;
tmp->y_pos_e = tmp->y_pos_e + x_right_length;
}
else
{
tmp->x_pos_e = tmp->x_pos_e + y_right_length;
tmp->y_pos_e = ua->a[tmp->y_id].len - 1;
}
if (tmp->x_pos_strand == 1)
{
list->list[list->length].x_id = tmp->x_id;
list->list[list->length].x_pos_e = ua->a[tmp->x_id].len - tmp->x_pos_s - 1;
list->list[list->length].x_pos_s = ua->a[tmp->x_id].len - tmp->x_pos_e - 1;
list->list[list->length].x_pos_strand = 0;
list->list[list->length].y_id = tmp->y_id;
list->list[list->length].y_pos_e = ua->a[tmp->y_id].len - tmp->y_pos_s - 1;
list->list[list->length].y_pos_s = ua->a[tmp->y_id].len - tmp->y_pos_e - 1;
list->list[list->length].y_pos_strand = 1;
resize_fake_cigar(&(list->list[list->length].f_cigar), (tmp->f_cigar.length + 2));
if(add_beg_end == 1)
{
add_fake_cigar(&(list->list[list->length].f_cigar), list->list[list->length].x_pos_s, 0);
}
long long distance_gap;
/****************************may have bugs********************************/
///long long pre_distance_gap = 0;
long long pre_distance_gap = 0xfffffffffffffff;
/****************************may have bugs********************************/
long long i = 0;
for (i = 0; i < (long long)tmp->f_cigar.length; i++)
{
distance_gap = get_fake_gap_shift(&(tmp->f_cigar), i);
if(distance_gap != pre_distance_gap)
{
pre_distance_gap = distance_gap;
add_fake_cigar(&(list->list[list->length].f_cigar),
ua->a[tmp->x_id].len - get_fake_gap_pos(&(tmp->f_cigar), i) - 1,
pre_distance_gap);
}
}
if(add_beg_end == 1 && get_fake_gap_pos(&(list->list[list->length].f_cigar),
list->list[list->length].f_cigar.length - 1) != (long long)list->list[list->length].x_pos_e)
{
add_fake_cigar(&(list->list[list->length].f_cigar),
list->list[list->length].x_pos_e,
get_fake_gap_shift(&(list->list[list->length].f_cigar),
list->list[list->length].f_cigar.length - 1));
}
}
else
{
list->list[list->length].x_id = tmp->x_id;
list->list[list->length].x_pos_e = tmp->x_pos_e;
list->list[list->length].x_pos_s = tmp->x_pos_s;
list->list[list->length].x_pos_strand = tmp->x_pos_strand;
list->list[list->length].y_id = tmp->y_id;
list->list[list->length].y_pos_e = tmp->y_pos_e;
list->list[list->length].y_pos_s = tmp->y_pos_s;
list->list[list->length].y_pos_strand = tmp->y_pos_strand;
resize_fake_cigar(&(list->list[list->length].f_cigar), (tmp->f_cigar.length + 2));
if(add_beg_end == 1)
{
@@ -422,7 +584,7 @@ int32_t ha_chain_check(k_mer_hit *a, int32_t n_a, Chain_Data *dp, int32_t min_sc
}
///double band_width_threshold = 0.05;
void chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result,
long long chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result,
double band_width_threshold, int max_skip, int x_readLen, int y_readLen)
{
long long i, j;
@@ -613,6 +775,7 @@ skip_dp:
i = dp->pre[i];
}
}
return chainLen;
}
void calculate_overlap_region_by_chaining_back(Candidates_list* candidates, overlap_region_alloc* overlap_list,
+3 -2
View File
@@ -187,6 +187,7 @@ void init_window_list_alloc(window_list_alloc* x);
void clear_window_list_alloc(window_list_alloc* x);
void destory_window_list_alloc(window_list_alloc* x);
void resize_window_list_alloc(window_list_alloc* x, long long size);
void chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result, double band_width_threshold, int max_skip, int x_readLen, int y_readLen);
long long chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result, double band_width_threshold, int max_skip, int x_readLen, int y_readLen);
int append_utg_inexact_overlap_region_alloc(overlap_region_alloc* list, overlap_region* tmp,
ma_utg_v *ua, int add_beg_end);
#endif
+255 -132
View File
@@ -12,6 +12,9 @@
#include "hic.h"
#include "kthread.h"
#include "tovlp.h"
#include "Assembly.h"
#include "rcut.h"
#include "horder.h"
uint32_t debug_purge_dup = 0;
@@ -68,12 +71,9 @@ long long min_thres;
uint32_t print_untig_by_read(ma_ug_t *g, const char* name, uint32_t in, ma_hit_t_alloc* sources,
ma_hit_t_alloc* reverse_sources, const char* info);
int asg_pop_bubble_primary_trio(ma_ug_t *ug, uint64_t* i_max_dist, uint32_t positive_flag, uint32_t negative_flag, hap_cov_t *cov, utg_trans_t *o, uint32_t is_update_chain);
void get_utg_ovlp(ma_ug_t **ug, asg_t* read_g, float drop_rate,
ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, ma_sub_t* coverage_cut,
long long tipsLen, float tip_drop_ratio, long long stops_threshold,
R_to_U* ruIndex, float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp,
kvec_asg_arc_t_warp* new_rtg_edges, hap_cov_t **i_cov, bub_label_t* b_mask_t,
uint32_t collect_p_trans, uint32_t collect_p_trans_f);
kv_u_trans_t *get_utg_ovlp(ma_ug_t **ug, asg_t* read_g, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, ma_sub_t* coverage_cut,
R_to_U* ruIndex, int max_hang, int min_ovlp, kvec_asg_arc_t_warp* new_rtg_edges, bub_label_t* b_mask_t, uint8_t* r_het);
void delete_useless_nodes(ma_ug_t **ug);
void init_bub_label_t(bub_label_t* x, uint32_t n_thres, uint32_t n_reads)
{
@@ -10187,7 +10187,7 @@ ma_hit_t_alloc* sources, R_to_U* ruIndex, const char* prefix, FILE *fp)
void ma_ug_print_bed(const ma_ug_t *g, asg_t *read_g, All_reads *RNF, ma_sub_t *coverage_cut,
ma_hit_t_alloc* sources, kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp, uint32_t rate_thres,
const char* prefix, FILE *fp, hap_cov_t *cov)
const char* prefix, FILE *fp, trans_chain* t_ch)
{
UC_Read g_read;
init_UC_Read(&g_read);
@@ -10210,7 +10210,7 @@ const char* prefix, FILE *fp, hap_cov_t *cov)
print_rough_inconsistent_sites(u, j, j+1, read_g, RNF, sources, coverage_cut,
edge, &g_read, &tmp, max_hang, min_ovlp, start, rate_thres, &exact_count,
&total_count, prefix, i+1, fp, cov? &(cov->t_ch->bed.a[i]): NULL);
&total_count, prefix, i+1, fp, t_ch? &(t_ch->bed.a[i]): NULL);
}
}
@@ -13046,13 +13046,9 @@ void write_trans_chain(trans_chain* t_ch, const char *fn)
sprintf(buf, "%s.hic.trans.bin", fn);
FILE* fp = fopen(buf, "w");
fwrite(&t_ch->u_num, sizeof(t_ch->u_num), 1, fp);
fwrite(&t_ch->r_num, sizeof(t_ch->r_num), 1, fp);
fwrite(t_ch->rUidx, sizeof(uint32_t), t_ch->r_num, fp);
fwrite(t_ch->rUpos, sizeof(uint64_t), t_ch->r_num, fp);
fwrite(t_ch->is_r_het, sizeof(uint8_t), t_ch->r_num, fp);
uint32_t i;
fwrite(&t_ch->bed.n, sizeof(t_ch->bed.n), 1, fp);
for (i = 0; i < t_ch->bed.n; i++)
@@ -13086,12 +13082,7 @@ trans_chain* load_hc_hits(const char *fn)
trans_chain *t_ch = NULL;
CALLOC(t_ch, 1);
flag += fread(&t_ch->u_num, sizeof(t_ch->u_num), 1, fp);
flag += fread(&t_ch->r_num, sizeof(t_ch->r_num), 1, fp);
MALLOC(t_ch->rUidx, t_ch->r_num);
flag += fread(t_ch->rUidx, sizeof(uint32_t), t_ch->r_num, fp);
MALLOC(t_ch->rUpos, t_ch->r_num);
flag += fread(t_ch->rUpos, sizeof(uint64_t), t_ch->r_num, fp);
MALLOC(t_ch->is_r_het, t_ch->r_num);
flag += fread(t_ch->is_r_het, sizeof(uint8_t), t_ch->r_num, fp);
@@ -13186,7 +13177,7 @@ long long gap_fuzz, bub_label_t* b_mask_t)
new_rtg_edges.a.n = 0;
ma_ug_print_bed(ug, sg, &R_INF, coverage_cut, sources, &new_rtg_edges,
max_hang, min_ovlp, asm_opt.hic_inconsist_rate, NULL, NULL, cov);
max_hang, min_ovlp, asm_opt.hic_inconsist_rate, NULL, NULL, cov->t_ch);
if((asm_opt.flag & HA_F_VERBOSE_GFA)) write_trans_chain(cov->t_ch, output_file_name);
}
@@ -13199,7 +13190,7 @@ long long gap_fuzz, bub_label_t* b_mask_t)
// fclose(output_file);
// free(gfa_name);
hic_analysis(ug, sg, cov?cov->t_ch:t_ch, &opt);
hic_analysis(ug, sg, cov?cov->t_ch:t_ch, &opt, 0);
if(cov) destory_hap_cov_t(&cov);
if(t_ch) destory_trans_chain(&t_ch);
@@ -13245,6 +13236,100 @@ long long gap_fuzz, bub_label_t* b_mask_t)
0.05, 0.9, max_hang, min_ovlp, 0, b_mask_t);
}
ma_ug_t *get_poly_ug(asg_t *sg, ma_sub_t* coverage_cut, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
int max_hang, int min_ovlp, R_to_U* ruIndex, bub_label_t* b_mask_t)
{
kvec_asg_arc_t_warp new_rtg_edges, d_edges;
kv_init(new_rtg_edges.a); kv_init(d_edges.a);
ma_ug_t *ug = ma_ug_gen_primary(sg, PRIMARY_LABLE);
adjust_utg_advance(sg, ug, reverse_sources, ruIndex, b_mask_t);
asg_t* nsg = (*ug).g;
uint32_t v, n_vtx = nsg->n_seq;
for (v = 0; v < n_vtx; ++v)
{
if(nsg->seq[v].del) continue;
nsg->seq[v].c = PRIMARY_LABLE;
}
delete_useless_nodes(&ug);
renew_utg(&ug, sg, NULL);
ma_ug_seq(ug, sg, coverage_cut, sources, &new_rtg_edges, max_hang, min_ovlp, &d_edges, 1);
kv_destroy(new_rtg_edges.a); kv_destroy(d_edges.a);
horder_clean_sg_by_utg(sg, ug);
return ug;
}
trans_chain* get_hic_polyploid_trans_chain(ma_ug_t *ug, asg_t *sg, ma_sub_t* coverage_cut, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
int max_hang, int min_ovlp, R_to_U* ruIndex, bub_label_t* b_mask_t)
{
uint32_t k;
trans_chain* p = NULL; CALLOC(p, 1);
p->r_num = sg->n_seq; p->u_num = ug->u.n;
kv_malloc(p->bed, p->u_num); p->bed.n = p->u_num;
for (k = 0; k < p->bed.n; k++) kv_init(p->bed.a[k]);
CALLOC(p->is_r_het, p->r_num);
kv_u_trans_t *ta = get_utg_ovlp(&ug, sg, sources, reverse_sources, coverage_cut, ruIndex, max_hang, min_ovlp, NULL, b_mask_t, p->is_r_het);
p->k_trans = *ta; free(ta);
return p;
}
void output_hic_graph_polyploid(asg_t *sg, ma_sub_t* coverage_cut, char* output_file_name,
ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
long long tipsLen, float tip_drop_ratio, long long stops_threshold,
R_to_U* ruIndex, float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp,
long long gap_fuzz, bub_label_t* b_mask_t)
{
ug_opt_t opt; memset(&opt, 0, sizeof(opt));
opt.coverage_cut = coverage_cut;
opt.sources = sources;
opt.reverse_sources = reverse_sources;
opt.tipsLen = (asm_opt.max_short_tip*2);
opt.tip_drop_ratio = 0.15;
opt.stops_threshold = 3;
opt.ruIndex = ruIndex;
opt.chimeric_rate = 0.05;
opt.drop_ratio = 0.9;
opt.max_hang = max_hang;
opt.min_ovlp = min_ovlp;
opt.is_bench = 0;
opt.b_mask_t = b_mask_t;
opt.gap_fuzz = gap_fuzz;
ma_ug_t *ug = get_poly_ug(sg, coverage_cut, sources, reverse_sources, max_hang, min_ovlp, ruIndex, b_mask_t);
trans_chain* t_ch = NULL;
char* gfa_name = (char*)malloc(strlen(output_file_name)+50);
sprintf(gfa_name, "%s.pre.clean_d_utg.noseq.gfa", output_file_name);
FILE* output_file = fopen(gfa_name, "w");
ma_ug_print_simple(ug, sg, coverage_cut, sources, ruIndex, "utg", output_file);
fclose(output_file);
free(gfa_name);
if((asm_opt.flag & HA_F_VERBOSE_GFA)) t_ch = load_hc_hits(output_file_name);
if(!t_ch)
{
t_ch = get_hic_polyploid_trans_chain(ug, sg, coverage_cut, sources, reverse_sources, max_hang, min_ovlp, ruIndex, b_mask_t);
ma_ug_print_bed(ug, sg, &R_INF, coverage_cut, sources, NULL,
max_hang, min_ovlp, asm_opt.hic_inconsist_rate, NULL, NULL, t_ch);
if((asm_opt.flag & HA_F_VERBOSE_GFA)) write_trans_chain(t_ch, output_file_name);
}
hic_analysis(ug, sg, t_ch, &opt, 1);
destory_trans_chain(&t_ch);
ma_ug_destroy(ug);
asg_cleanup(sg);
reduce_hamming_error(sg, sources, coverage_cut, max_hang, min_ovlp, gap_fuzz);
output_trio_unitig_graph(sg, coverage_cut, output_file_name, FATHER, sources, reverse_sources, (asm_opt.max_short_tip*2), 0.15, 3, ruIndex,
0.05, 0.9, max_hang, min_ovlp, 0, b_mask_t);
output_trio_unitig_graph(sg, coverage_cut, output_file_name, MOTHER, sources, reverse_sources, (asm_opt.max_short_tip*2), 0.15, 3, ruIndex,
0.05, 0.9, max_hang, min_ovlp, 0, b_mask_t);
}
void set_trio_flag_by_cov(ma_ug_t *ug, asg_t *read_g, hap_cov_t *cov)
{
kvec_t(uint64_t) idx; kv_init(idx);
@@ -13377,6 +13462,74 @@ void set_trio_flag_by_cov(ma_ug_t *ug, asg_t *read_g, hap_cov_t *cov)
uint64_t get_utg_cov(ma_ug_t *ug, uint32_t uID, asg_t* read_g,
const ma_sub_t* coverage_cut, ma_hit_t_alloc* sources, R_to_U* ruIndex, uint8_t* r_flag)
{
ma_utg_t *u = &(ug->u.a[uID]);
uint32_t k, j, rId, tn, is_Unitig;
long long R_bases = 0, C_bases = 0;
long long cov_in, cov_out;
uint32_t nv, i;
asg_arc_t *av = NULL;
ma_hit_t *h;
if(u->m == 0 || ug->g->seq[uID].del) return 0;
///set
for (k = 0; k < u->n; k++) r_flag[u->a[k]>>33] = 1;
for (i = 0; i < 2; i++)
{
nv = asg_arc_n(ug->g, (uID<<1)+i);
av = asg_arc_a(ug->g, (uID<<1)+i);
for (j = 0; j < nv; j++)
{
if(av[j].del) continue;
u = &(ug->u.a[av[j].v>>1]);
for (k = 0; k < u->n; k++) r_flag[u->a[k]>>33] = 2;
}
}
u = &(ug->u.a[uID]);
cov_in = cov_out = R_bases = 0;
for (k = 0; k < u->n; k++)
{
rId = u->a[k]>>33;
R_bases += (coverage_cut[rId].e - coverage_cut[rId].s);
for (j = 0; j < (uint64_t)(sources[rId].length); j++)
{
h = &(sources[rId].buffer[j]);
///if(h->del) continue;
if(h->el != 1) continue;
tn = Get_tn((*h));
if(read_g->seq[tn].del == 1)
{
///get the id of read that contains it
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
if(tn == (uint32_t)-1 || is_Unitig == 1 || read_g->seq[tn].del == 1) continue;
}
if(read_g->seq[tn].del == 1) continue;
if(r_flag[tn] == 0) continue;
if(r_flag[tn] == 1) cov_in += (Get_qe((*h)) - Get_qs((*h)));
if(r_flag[tn] == 2) cov_out += (Get_qe((*h)) - Get_qs((*h)));
}
}
C_bases = cov_in;
if(cov_out <= (cov_in*0.2)) C_bases += cov_out;
///reset
for (k = 0; k < u->n; k++) r_flag[u->a[k]>>33] = 0;
for (i = 0; i < 2; i++)
{
nv = asg_arc_n(ug->g, (uID<<1)+i);
av = asg_arc_a(ug->g, (uID<<1)+i);
for (j = 0; j < nv; j++)
{
u = &(ug->u.a[av[j].v>>1]);
for (k = 0; k < u->n; k++) r_flag[u->a[k]>>33] = 0;
}
}
return R_bases == 0? 0 : C_bases/R_bases;
}
void print_r_het(hap_cov_t *cov, uint8_t* trio_flag, const char* cmd)
@@ -13683,6 +13836,50 @@ void kt_u_trans_t_symm(kv_u_trans_t *ta, ma_ug_t *ug)
kt_u_trans_t_idx(ta, ug->g->n_seq);
}
void kt_u_trans_t_simple_symm(kv_u_trans_t *ta, uint32_t un, uint32_t symm_add)
{
u_trans_t *a = NULL, *r_a = NULL, *p = NULL;
uint32_t k, n, m;
n = ta->n;
for (k = 0; k < n; k++)
{
if(ta->a[k].del || ta->a[k].qn > ta->a[k].tn) continue;
get_u_trans_spec(ta, ta->a[k].tn, ta->a[k].qn, &r_a, NULL);
a = &(ta->a[k]);
if(!r_a || r_a->del)
{
if(symm_add) kv_pushp(u_trans_t, *ta, &r_a);
else
{
a->del = 1;
continue;
}
}
else
{
if(r_a->nw > a->nw)
{
p = a;
a = r_a;
r_a = p;
}
}
(*r_a) = (*a);
r_a->qn = a->tn; r_a->qs = a->ts; r_a->qe = a->te;
r_a->tn = a->qn; r_a->ts = a->qs; r_a->te = a->qe;
}
for (k = m = 0; k < ta->n; ++k)
{
if(ta->a[k].del) continue;
ta->a[m] = ta->a[k];
m++;
}
ta->n = m;
kt_u_trans_t_idx(ta, un);
}
void debug_u_trans_t(kv_u_trans_t *ta)
{
u_trans_t *a = NULL, *r_a = NULL;
@@ -13914,12 +14111,15 @@ bub_label_t* b_mask_t)
hap_cov_t *cov = NULL;
asg_t *copy_sg = copy_read_graph(sg);
ma_ug_t *copy_ug = copy_untig_graph(ug);
/*******************************for debug************************************/
// adjust_utg_by_primary(&copy_ug, copy_sg, TRIO_THRES, sources, reverse_sources, coverage_cut,
// tipsLen, tip_drop_ratio, stops_threshold, ruIndex, chimeric_rate, drop_ratio,
// max_hang, min_ovlp, &new_rtg_edges, &cov, b_mask_t, 1, 1);
get_utg_ovlp(&copy_ug, copy_sg, TRIO_THRES, sources, reverse_sources, coverage_cut,
tipsLen, tip_drop_ratio, stops_threshold, ruIndex, chimeric_rate, drop_ratio,
max_hang, min_ovlp, &new_rtg_edges, &cov, b_mask_t, 1, 1);
adjust_utg_advance(copy_sg, copy_ug, reverse_sources, ruIndex, b_mask_t);
get_utg_ovlp(&copy_ug, copy_sg, sources, reverse_sources, coverage_cut,
ruIndex, max_hang, min_ovlp, &new_rtg_edges, b_mask_t, NULL);
exit(1);
/*******************************for debug************************************/
print_utg(copy_ug, copy_sg, coverage_cut, output_file_name, sources, ruIndex, max_hang,
min_ovlp, &new_rtg_edges);
ma_ug_destroy(copy_ug);
@@ -24437,119 +24637,40 @@ uint32_t collect_p_trans, uint32_t collect_p_trans_f)
}
}
void get_utg_ovlp(ma_ug_t **ug, asg_t* read_g, float drop_rate,
ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, ma_sub_t* coverage_cut,
long long tipsLen, float tip_drop_ratio, long long stops_threshold,
R_to_U* ruIndex, float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp,
kvec_asg_arc_t_warp* new_rtg_edges, hap_cov_t **i_cov, bub_label_t* b_mask_t,
uint32_t collect_p_trans, uint32_t collect_p_trans_f)
void set_r_het_status(uint8_t* r_het, kv_gg_status *sa, ma_ug_t *ug, uint32_t hapN)
{
fprintf(stderr, "******1******\n");
asg_t* nsg = (*ug)->g;
uint32_t v, n_vtx = nsg->n_seq, k, rId, just_contain;
ma_utg_t* u = NULL;
hap_cov_t *cov = init_hap_cov_t(*ug, read_g, sources, ruIndex, reverse_sources,
coverage_cut, max_hang, min_ovlp, (asm_opt.purge_level_primary>0||i_cov)?1:0);
if(cov->t_ch) set_r_het_flag(*ug, read_g, coverage_cut, sources, ruIndex, cov->t_ch);
adjust_utg_advance(read_g, (*ug), reverse_sources, ruIndex, b_mask_t);
nsg = (*ug)->g;
n_vtx = nsg->n_seq;
for (v = 0; v < n_vtx; ++v)
uint32_t i, k, o, f;
ma_utg_t *u;
mcg_node_t s;
for (i = 0; i < sa->n; i++)
{
if(nsg->seq[v].del) continue;
nsg->seq[v].c = PRIMARY_LABLE;
EvaluateLen((*ug)->u, v) = (*ug)->u.a[v].n;
}
topo_ovlp_collect(*ug, read_g, sources, reverse_sources, coverage_cut, tipsLen, tip_drop_ratio,
stops_threshold, ruIndex, chimeric_rate, drop_ratio, max_hang, min_ovlp, cov);
delete_useless_nodes(ug);
renew_utg(ug, read_g, new_rtg_edges);
if(i_cov && collect_p_trans == 0) goto skip_purge;
if(asm_opt.purge_level_primary > 0)
{
// print_debug_gfa(read_g, *ug, coverage_cut, "debug_purge", sources, ruIndex, asm_opt.max_hang_Len, asm_opt.min_overlap_Len);
just_contain = 0;
if(asm_opt.purge_level_primary == 1) just_contain = 1;
purge_dups(*ug, read_g, coverage_cut, sources, reverse_sources, ruIndex, new_rtg_edges,
asm_opt.purge_simi_thres, asm_opt.purge_overlap_len, max_hang, min_ovlp, drop_ratio,
just_contain, 0, cov, !!(cov->t_ch&&collect_p_trans), collect_p_trans_f);
delete_useless_nodes(ug);
renew_utg(ug, read_g, new_rtg_edges);
}
if (!(asm_opt.flag & HA_F_BAN_POST_JOIN))
{
rescue_missing_overlaps_aggressive(*ug, read_g, sources, coverage_cut, ruIndex, max_hang,
min_ovlp, 0, 1, NULL, b_mask_t);
renew_utg(ug, read_g, new_rtg_edges);
rescue_contained_reads_aggressive(*ug, read_g, sources, coverage_cut, ruIndex, max_hang,
min_ovlp, 10, 0, 1, NULL, NULL, b_mask_t);
renew_utg(ug, read_g, new_rtg_edges);
}
n_vtx = read_g->n_seq;
for (v = 0; v < n_vtx; v++)
{
read_g->seq[v].c = ALTER_LABLE;
}
nsg = (*ug)->g;
n_vtx = nsg->n_seq;
for (v = 0; v < n_vtx; ++v)
{
if(nsg->seq[v].del) continue;
if(nsg->seq[v].c == ALTER_LABLE) continue;
u = &((*ug)->u.a[v]);
if(u->m == 0) continue;
for (k = 0; k < u->n; k++)
{
rId = u->a[k]>>33;
read_g->seq[rId].c = nsg->seq[v].c;
u = &(ug->u.a[i]);
s = sa->a[i].s; o = 0;
while (s) {
o += (s&1); s>>=1;
}
}
n_vtx = read_g->n_seq;
for (v = 0; v < n_vtx; v++)
{
if(read_g->seq[v].c == ALTER_LABLE)
{
asg_seq_drop(read_g, v);
}
f = N_HET;
if(o < hapN) f = C_HET;
for (k = 0; k < u->n; k++) r_het[u->a[k]>>33] = f;
}
if(asm_opt.recover_atg_cov_min == -1024)
{
asm_opt.recover_atg_cov_max = asm_opt.hom_global_coverage/HOM_PEAK_RATE;
asm_opt.recover_atg_cov_min = asm_opt.recover_atg_cov_max * 0.85;
asm_opt.recover_atg_cov_max = INT32_MAX;
}
if(asm_opt.recover_atg_cov_max != INT32_MAX)
{
fprintf(stderr, "[M::%s] primary contig coverage range: [%d, %d]\n",
__func__, asm_opt.recover_atg_cov_min, asm_opt.recover_atg_cov_max);
}
else
{
fprintf(stderr, "[M::%s] primary contig coverage range: [%d, infinity]\n",
__func__, asm_opt.recover_atg_cov_min);
}
skip_purge:
recover_utg_by_coverage(ug, read_g, coverage_cut, sources, ruIndex, cov->t_ch);
if(i_cov)
{
(*i_cov) = cov;
}
else
{
destory_hap_cov_t(&cov);
}
}
kv_u_trans_t *get_utg_ovlp(ma_ug_t **ug, asg_t* read_g, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, ma_sub_t* coverage_cut,
R_to_U* ruIndex, int max_hang, int min_ovlp, kvec_asg_arc_t_warp* new_rtg_edges, bub_label_t* b_mask_t, uint8_t* r_het)
{
///print_debug_gfa(read_g, *ug, coverage_cut, "init", sources, ruIndex, asm_opt.max_hang_Len, asm_opt.min_overlap_Len);
kv_u_trans_t *ta = pt_pdist(*ug, read_g,coverage_cut, sources, new_rtg_edges, max_hang, min_ovlp, 5);
kv_gg_status *sa = init_mc_gg_status(*ug, read_g, coverage_cut, sources, ruIndex,
asm_opt.hom_global_coverage_set?asm_opt.hom_global_coverage:((double)asm_opt.hom_global_coverage)/((double)HOM_PEAK_RATE),
asm_opt.polyploidy);
mc_solve_general(ta, (*ug)->u.n, sa, asm_opt.polyploidy, 1, 1);
if(r_het) set_r_het_status(r_het, sa, *ug, asm_opt.polyploidy);
free(sa->a); free(sa);
return ta;
// ma_ug_seq(*ug, read_g, coverage_cut, sources, new_rtg_edges, max_hang, min_ovlp, 0, 0);
// ug_idx_build(*ug, asm_opt.polyploidy);
// topo_ovlp_collect(*ug, read_g, sources, reverse_sources, coverage_cut, tipsLen, tip_drop_ratio,
// stops_threshold, ruIndex, chimeric_rate, drop_ratio, max_hang, min_ovlp, cov);
}
void output_contig_graph_primary_pre(asg_t *sg, ma_sub_t* coverage_cut, char* output_file_name,
@@ -30695,7 +30816,9 @@ ma_sub_t **coverage_cut_ptr, int debug_g)
else if(ha_opt_hic(&asm_opt))
{
if(asm_opt.flag & HA_F_PARTITION) asm_opt.flag -= HA_F_PARTITION;
output_hic_graph(sg, coverage_cut, o_file, sources, reverse_sources, (asm_opt.max_short_tip*2),
// output_hic_graph(sg, coverage_cut, o_file, sources, reverse_sources, (asm_opt.max_short_tip*2),
// 0.15, 3, ruIndex, 0.05, 0.9, max_hang_length, mini_overlap_length, gap_fuzz, &b_mask_t);
output_hic_graph_polyploid(sg, coverage_cut, o_file, sources, reverse_sources, (asm_opt.max_short_tip*2),
0.15, 3, ruIndex, 0.05, 0.9, max_hang_length, mini_overlap_length, gap_fuzz, &b_mask_t);
}
else if((asm_opt.flag & HA_F_PARTITION) && (asm_opt.purge_level_primary > 0))
+4 -1
View File
@@ -120,7 +120,7 @@ typedef struct {
uint32_t occ;
double nw;
uint8_t f:6, rev:1, del:1;
uint8_t qo:4, to:4;
///uint8_t qo:4, to:4;
} u_trans_t;
typedef struct {
@@ -932,6 +932,7 @@ uint32_t *aux_a, uint32_t aux_n, uint32_t aux_beg, uint64_t *i_aux_len,
ma_ug_t *ug, uint32_t flag, double overall_score, const char* cmd);
int asg_arc_del_trans(asg_t *g, int fuzz);
void kt_u_trans_t_idx(kv_u_trans_t *ta, uint32_t n);
void kt_u_trans_t_simple_symm(kv_u_trans_t *ta, uint32_t un, uint32_t symm_add);
uint32_t get_u_trans_spec(kv_u_trans_t *ta, uint32_t qn, uint32_t tn, u_trans_t **r_a, uint32_t *occ);
int ma_ug_seq(ma_ug_t *g, asg_t *read_g, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources,
kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp, kvec_asg_arc_t_warp *E, uint32_t is_polish);
@@ -989,6 +990,8 @@ inline uint32_t get_offset_adjust(uint32_t offset, uint32_t offsetLen, uint32_t
uint32_t set_utg_offset(uint32_t *a, uint32_t a_n, ma_ug_t *ug, asg_t *read_sg, uint64_t* pos_idx, uint32_t is_clear,
uint32_t only_len);
uint64_t get_utg_cov(ma_ug_t *ug, uint32_t uID, asg_t* read_g,
const ma_sub_t* coverage_cut, ma_hit_t_alloc* sources, R_to_U* ruIndex, uint8_t* r_flag);
#define JUNK_COV 5
#define DISCARD_RATE 0.8
+181
View File
@@ -178,6 +178,187 @@ kvec_t_u64_warp* chain_idx, void *ha_flt_tab, ha_pt_t *ha_idx, overlap_region* f
}
void calculate_ug_chaining(Candidates_list* candidates, overlap_region_alloc* overlap_list, kvec_t_u64_warp* chain_idx,
uint64_t readID, ma_utg_v *ua, double band_width_threshold, int add_beg_end, overlap_region* f_cigar, long long mz_occ, double mz_rate)
{
long long i = 0;
uint64_t current_ID;
uint64_t current_stand;
if (candidates->length == 0)
{
return;
}
long long sub_region_beg;
long long sub_region_end;
long long chain_len;
clear_fake_cigar(&((*f_cigar).f_cigar));
i = 0;
while (i < candidates->length)
{
chain_idx->a.n = 0;
current_ID = candidates->list[i].readID;
current_stand = candidates->list[i].strand;
///reference read
(*f_cigar).x_id = readID;
(*f_cigar).x_pos_strand = current_stand;
///query read
(*f_cigar).y_id = current_ID;
///here the strand of query is always 0
(*f_cigar).y_pos_strand = 0;
sub_region_beg = i;
sub_region_end = i;
i++;
while (i < candidates->length
&&
current_ID == candidates->list[i].readID
&&
current_stand == candidates->list[i].strand)
{
sub_region_end = i;
i++;
}
if ((*f_cigar).x_id == (*f_cigar).y_id)
{
continue;
}
chain_len = chain_DP(candidates->list + sub_region_beg,
sub_region_end - sub_region_beg + 1, &(candidates->chainDP), f_cigar, band_width_threshold,
50, ua->a[(*f_cigar).x_id].len, ua->a[(*f_cigar).y_id].len);
// if ((*f_cigar).x_id != (*f_cigar).y_id)
if ((*f_cigar).x_id != (*f_cigar).y_id && chain_len > mz_occ*mz_rate)
{
append_utg_inexact_overlap_region_alloc(overlap_list, f_cigar, ua, add_beg_end);
}
}
}
void ha_get_ug_candidates(ha_abuf_t *ab, int64_t rid, ma_utg_t *u, ma_utg_v *ua, overlap_region_alloc *overlap_list, Candidates_list *cl, double bw_thres, int max_n_chain, int keep_whole_chain, kvec_t_u8_warp* k_flag,
kvec_t_u64_warp* chain_idx, void *ha_flt_tab, ha_pt_t *ha_idx, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct, double chain_match_rate)
{
uint32_t i;
uint64_t k, l;
// prepare
clear_Candidates_list(cl);
clear_overlap_region_alloc(overlap_list);
ab->mz.n = 0, ab->n_a = 0;
// get the list of anchors
ha_sketch_query(u->s, u->len, asm_opt.mz_win, asm_opt.k_mer_length, 0, !(asm_opt.flag & HA_F_NO_HPC), &ab->mz, ha_flt_tab, k_flag, dbg_ct);
// minimizer of queried read
if (ab->mz.m > ab->old_mz_m) {
ab->old_mz_m = ab->mz.m;
REALLOC(ab->seed, ab->old_mz_m);
}
for (i = 0, ab->n_a = 0; i < ab->mz.n; ++i) {
int n;
ab->seed[i].a = ha_pt_get(ha_idx, ab->mz.a[i].x, &n);
ab->seed[i].n = n;
ab->seed[i].good = 0;
ab->n_a += n;
}
if (ab->n_a > ab->m_a) {
ab->m_a = ab->n_a;
kroundup64(ab->m_a);
REALLOC(ab->a, ab->m_a);
}
for (i = 0, k = 0; i < ab->mz.n; ++i) {
int j;
///z is one of the minimizer
ha_mz1_t *z = &ab->mz.a[i];
seed1_t *s = &ab->seed[i];
for (j = 0; j < s->n; ++j) {
const ha_idxpos_t *y = &s->a[j];
anchor1_t *an = &ab->a[k++];
uint8_t rev = z->rev == y->rev? 0 : 1;
an->other_off = y->pos;
an->self_off = rev? u->len - 1 - (z->pos + 1 - z->span) : z->pos;
an->good = s->good;
an->srt = (uint64_t)y->rid<<33 | (uint64_t)rev<<32 | an->other_off;
}
}
// sort anchors
radix_sort_ha_an1(ab->a, ab->a + ab->n_a);
for (k = 1, l = 0; k <= ab->n_a; ++k) {
if (k == ab->n_a || ab->a[k].srt != ab->a[l].srt) {
if (k - l > 1)
radix_sort_ha_an2(ab->a + l, ab->a + k);
l = k;
}
}
// copy over to _cl_
if (ab->m_a >= (uint64_t)cl->size) {
cl->size = ab->m_a;
REALLOC(cl->list, cl->size);
}
for (k = 0; k < ab->n_a; ++k) {
k_mer_hit *p = &cl->list[k];
p->readID = ab->a[k].srt >> 33;
p->strand = ab->a[k].srt >> 32 & 1;
p->offset = ab->a[k].other_off;
p->self_offset = ab->a[k].self_off;
p->good = ab->a[k].good;
}
cl->length = ab->n_a;
calculate_ug_chaining(cl, overlap_list, chain_idx, rid, ua, bw_thres, keep_whole_chain, f_cigar, ab->mz.n, chain_match_rate);
#if 0
if (overlap_list->length > 0) {
fprintf(stderr, "B\t%ld\t%ld\t%d\n", (long)rid, (long)overlap_list->length, rlen);
for (int i = 0; i < (int)overlap_list->length; ++i) {
overlap_region *r = &overlap_list->list[i];
fprintf(stderr, "C\t%d\t%d\t%d\t%c\t%d\t%ld\t%d\t%d\t%c\t%d\t%d\n", (int)r->x_id, (int)r->x_pos_s, (int)r->x_pos_e, "+-"[r->x_pos_strand],
(int)r->y_id, (long)Get_READ_LENGTH(R_INF, r->y_id), (int)r->y_pos_s, (int)r->y_pos_e, "+-"[r->y_pos_strand], (int)r->shared_seed, ha_ov_type(r, rlen));
}
}
#endif
if ((int)overlap_list->length > max_n_chain) {
int32_t w, n[4], s[4];
n[0] = n[1] = n[2] = n[3] = 0, s[0] = s[1] = s[2] = s[3] = 0;
ks_introsort_or_ss(overlap_list->length, overlap_list->list);
for (i = 0; i < (uint32_t)overlap_list->length; ++i) {
const overlap_region *r = &overlap_list->list[i];
w = ha_ov_type(r, u->len);
++n[w];
if ((int)n[w] == max_n_chain) s[w] = r->shared_seed;
}
if (s[0] > 0 || s[1] > 0 || s[2] > 0 || s[3] > 0) {
for (i = 0, k = 0; i < (uint32_t)overlap_list->length; ++i) {
overlap_region *r = &overlap_list->list[i];
w = ha_ov_type(r, u->len);
if (r->shared_seed >= s[w]) {
if ((uint32_t)k != i) {
overlap_region t;
t = overlap_list->list[k];
overlap_list->list[k] = overlap_list->list[i];
overlap_list->list[i] = t;
}
++k;
}
}
overlap_list->length = k;
}
}
///ks_introsort_or_xs(overlap_list->length, overlap_list->list);
}
void lable_matched_ovlp(overlap_region_alloc* overlap_list, ma_hit_t_alloc* paf)
{
uint64_t j = 0, inner_j = 0;
+250 -34
View File
@@ -44,6 +44,11 @@ KRADIX_SORT_INIT(u_trans_occ, u_trans_t, u_trans_occ_key, member_size(u_trans_t,
#define is_hom_hit(a) ((a).id == (uint64_t)-1)
typedef struct {
kv_gg_status sg;
uint64_t xs;
} psg_t;
typedef struct{
kvec_t(char) name;
kvec_t(uint64_t) name_Len;
@@ -780,7 +785,9 @@ ha_ug_index* build_unitig_index(ma_ug_t *ug, int k, uint64_t up_occ, uint64_t lo
ha_ug_index* idx = NULL; CALLOC(idx, 1);
pldat_t pl; pl.h = idx; pl.is_cnt = 1;
double index_time = yak_realtime(), beg_time;
fprintf(stderr, "sa-0\n");
init_ha_ug_index_opt(idx, ug, k, &pl, up_occ, low_occ, thread_num);
fprintf(stderr, "sa-1\n");
beg_time = yak_realtime();
pl.is_cnt = 1;
@@ -2960,7 +2967,7 @@ uint32_t get_specific_shortest_path(pdq_spec *p)
{
pop_pdq(p->pq, &(p->v), &w);
p->pq->vis.a[p->v] = 1;
if(p->dest[p->v] == p->flag) p->occ++;
if(p->dest && p->dest[p->v] == p->flag) p->occ++;
if(p->occ > p->df_occ) return 0;
av = asg_arc_a(p->sg, p->v);
@@ -2979,7 +2986,8 @@ uint32_t get_specific_shortest_path(pdq_spec *p)
if(p->pre) p->pre[u] = p->v;
}
}
if(p->dest[p->v] == p->flag) return 1;
if(p->dest && p->dest[p->v] == p->flag) return 1;
if(!p->dest) return 1;
}
return 0;
@@ -3073,6 +3081,22 @@ double rate, long long *dis)
}
void set_utg_by_dis(uint32_t v, pdq* pq, asg_t *g, kvec_t_u32_warp *res, uint32_t dis)
{
uint64_t r;
pdq_spec a;
a.v = (uint64_t)-1; a.src = v; a.pq = pq; a.sg = g; a.df_occ = (uint64_t)-1;
a.occ = 0; a.flag = (uint8_t)-1; a.dest = NULL; a.pre = NULL;
reset_pdq(a.pq);
while (1)
{
r = get_specific_shortest_path(&a);
if(!r) return;
if(a.pq->dis.a[a.v] <= dis) kv_push(uint32_t, res->a, a.v);
else return;
}
}
uint64_t LCA_distance(long long d_x, long long d_y, long long xLen, long long yLen, uint8_t* rev)
{
(*rev) = 0;
@@ -9307,14 +9331,37 @@ void get_forward_distance(uint32_t src, uint32_t dest, asg_t *sg, hc_links* link
// ((e->dis>>2)&1)?"back":"forw", e->dis>>3);
}
uint32_t is_same_phase(uint64_t bid, uint64_t eid, H_partition* hap, int8_t *s, mc_gg_status *sa)
{
if(bid == eid) return 1;
if(hap || s)
{
int beg_status, end_status;
beg_status = (hap? get_phase_status(hap, bid):s[bid]);
if(beg_status != 1 && beg_status != -1) return (uint32_t)-1;
end_status = (hap? get_phase_status(hap, eid):s[eid]);
if(end_status != 1 && end_status != -1) return (uint32_t)-1;
if(beg_status == end_status) return 1;
return 0;
}
if(sa)
{
if(sa[bid].s == 0 || sa[eid].s == 0) return (uint32_t)-1;
return !!(sa[bid].s&sa[eid].s);
}
return (uint32_t)-1;
}
int get_trans_rate_function_advance(ha_ug_index* idx, kvec_pe_hit* hits, hc_links* link, bubble_type* bub,
H_partition* hap, int8_t *s, trans_idx* dis)
H_partition* hap, int8_t *s, mc_gg_status *sa, trans_idx* dis)
{
kvec_t(uint64_t) buf;
kv_init(buf);
uint64_t beg, end, cnt[2];
uint64_t k, i, t_d, r_idx, f_idx, med = (uint64_t)-1;
int beg_status, end_status;
uint32_t is_s;
// int beg_status, end_status;
buf.n = 0;
for (k = 0; k < hits->a.n; ++k)
@@ -9329,25 +9376,28 @@ H_partition* hap, int8_t *s, trans_idx* dis)
t_d = get_hic_distance(&(hits->a.a[k]), link, idx, NULL);
if(t_d == (uint64_t)-1) continue;
if(beg == end)
{
t_d = (t_d << 1);
}
else
{
beg_status = (hap? get_phase_status(hap, beg):s[beg]);
if(beg_status != 1 && beg_status != -1) continue;
end_status = (hap? get_phase_status(hap, end):s[end]);
if(end_status != 1 && end_status != -1) continue;
if(beg_status != end_status)
{
t_d = (t_d << 1) + 1;
}
else
{
t_d = (t_d << 1);
}
}
// if(beg == end)
// {
// t_d = (t_d << 1);
// }
// else
// {
// beg_status = (hap? get_phase_status(hap, beg):s[beg]);
// if(beg_status != 1 && beg_status != -1) continue;
// end_status = (hap? get_phase_status(hap, end):s[end]);
// if(end_status != 1 && end_status != -1) continue;
// if(beg_status != end_status)
// {
// t_d = (t_d << 1) + 1;
// }
// else
// {
// t_d = (t_d << 1);
// }
// }
is_s = is_same_phase(beg, end, hap, s, sa);
if(is_s == (uint32_t)-1) continue;
t_d = (t_d << 1) + 1 - is_s;
kv_push(uint64_t, buf, t_d);
}
@@ -9532,7 +9582,7 @@ void init_hic_advance(ha_ug_index* idx, kvec_pe_hit* hits, hc_links* link, bubbl
if(bub->round_id > 0 && ignore_dis == 0)
{
is_comples_weight = get_trans_rate_function_advance(idx, hits, link, bub, hap, NULL, &dis);
is_comples_weight = get_trans_rate_function_advance(idx, hits, link, bub, hap, NULL, NULL, &dis);
}
@@ -14265,7 +14315,7 @@ void label_unitigs(G_partition* g_p, ma_ug_t* ug)
///fprintf(stderr, "# Mother reads: %lu\n", occ);
}
void label_unitigs_sm(int8_t *s, ma_ug_t* ug)
void label_unitigs_sm(int8_t *s, mc_gg_status *sa, ma_ug_t* ug)
{
memset(R_INF.trio_flag, AMBIGU, R_INF.total_reads * sizeof(uint8_t));
uint32_t i, k, flag = AMBIGU;
@@ -14273,8 +14323,20 @@ void label_unitigs_sm(int8_t *s, ma_ug_t* ug)
for (i = 0; i < ug->g->n_seq; i++)
{
if(ug->g->seq[i].del || s[i] == 0) continue;
flag = (s[i] > 0? FATHER:MOTHER);
if(ug->g->seq[i].del) continue;
flag = 0;
if(s)
{
if(s[i] == 0) continue;
flag = (s[i] > 0? FATHER:MOTHER);
}
if(sa)
{
if(sa[i].s != 1 && sa[i].s != 2) continue;
flag = sa[i].s;
}
u = &ug->u.a[i];
if(u->m == 0) continue;
for (k = 0; k < u->n; k++)
@@ -15186,6 +15248,7 @@ void print_kv_weight(kv_u_trans_t *ta)
{
uint32_t i;
u_trans_t *e = NULL;
fprintf(stderr, "\n[M::%s]\n", __func__);
fprintf(stderr, "*********ta->n: %u\n", (uint32_t)ta->n);
for (i = 0; i < ta->n; i++)
{
@@ -15227,14 +15290,14 @@ void print_debug_hc_links(ha_ug_index* idx, bubble_type* bub, hc_links* lk, kv_u
}
}
void renew_kv_u_trans(kv_u_trans_t *ta, hc_links *lk, kvec_pe_hit* hits, kv_u_trans_t *ref,
ha_ug_index* idx, bubble_type* bub, int8_t *s, uint32_t ignore_dis)
ha_ug_index* idx, bubble_type* bub, int8_t *s, mc_gg_status *sa, uint32_t ignore_dis)
{
uint64_t k, i, m, is_comples_weight = 0;
trans_idx dis;
kv_init(dis);
if(bub->round_id > 0 && ignore_dis == 0)
{
is_comples_weight = get_trans_rate_function_advance(idx, hits, lk, bub, NULL, s, &dis);
is_comples_weight = get_trans_rate_function_advance(idx, hits, lk, bub, NULL, s, sa, &dis);
}
for (i = 0; i < lk->a.n; i++)
@@ -15635,21 +15698,21 @@ int hic_short_align(const enzyme *fn1, const enzyme *fn2, ha_ug_index* idx, ug_o
if((asm_opt.flag & HA_F_VERBOSE_GFA) && load_ps_t(&s, asm_opt.output_file_name))
{
bub.round_id = bub.n_round;
label_unitigs_sm(s->s, idx->ug);
label_unitigs_sm(s->s, NULL, idx->ug);
goto skip_flipping;
}
s = init_ps_t(11, idx->ug->g->n_seq);
for (bub.round_id = 0; bub.round_id < bub.n_round; bub.round_id++)
{
// identify_bubbles(idx->ug, &bub, idx->t_ch->is_r_het, &(idx->t_ch->k_trans));
renew_kv_u_trans(&k_trans, &link, &sl.hits, &(idx->t_ch->k_trans), idx, &bub, s->s, 0);
renew_kv_u_trans(&k_trans, &link, &sl.hits, &(idx->t_ch->k_trans), idx, &bub, s->s, NULL, 0);
// if(bub.round_id == 0) init_phase(idx, &k_trans, &bub, s);
// update_trans_g(idx, &k_trans, &bub);
/*******************************for debug************************************/
mc_solve(NULL, NULL, &k_trans, idx->ug, idx->read_g, 0.8, R_INF.trio_flag,
(bub.round_id == 0? 1 : 0), s->s, 1, /**&bub**/NULL, &(idx->t_ch->k_trans));
/*******************************for debug************************************/
label_unitigs_sm(s->s, idx->ug);
label_unitigs_sm(s->s, NULL, idx->ug);
/*******************************for debug************************************/
// if(bub.round_id == bub.n_round - 1)
@@ -15670,6 +15733,8 @@ int hic_short_align(const enzyme *fn1, const enzyme *fn2, ha_ug_index* idx, ug_o
skip_flipping:
verbose_het_stat(&bub);
print_kv_weight(&k_trans);
// horder_t *ho = init_horder_t(&sl.hits, idx->uID_bits, idx->pos_mode, idx->read_g, idx->ug, &bub, &(idx->t_ch->k_trans), opt, 3);
///print_hc_links(&link, 0, &hap);
@@ -15711,8 +15776,156 @@ int hic_short_align(const enzyme *fn1, const enzyme *fn2, ha_ug_index* idx, ug_o
return 1;
}
int load_psg_t(psg_t **sg, const char *fn)
{
uint64_t flag = 0;
char *buf = (char*)calloc(strlen(fn) + 25, 1);
sprintf(buf, "%s.hic.pst.bin", fn);
FILE* fp = NULL;
fp = fopen(buf, "r");
if(!fp) return 0;
CALLOC(*sg, 1);
flag += fread(&((*sg)->xs), sizeof((*sg)->xs), 1, fp);
flag += fread(&((*sg)->sg.n), sizeof((*sg)->sg.n), 1, fp);
(*sg)->sg.m = (*sg)->sg.n;
MALLOC((*sg)->sg.a, (*sg)->sg.n);
flag += fread((*sg)->sg.a, sizeof(mc_gg_status), (*sg)->sg.n, fp);
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt)
fclose(fp);
free(buf);
return 1;
}
void write_psg_t(psg_t *sg, const char *fn)
{
char *buf = (char*)calloc(strlen(fn) + 25, 1);
sprintf(buf, "%s.hic.pst.bin", fn);
FILE* fp = fopen(buf, "w");
fwrite(&(sg->xs), sizeof(sg->xs), 1, fp);
fwrite(&(sg->sg.n), sizeof(sg->sg.n), 1, fp);
fwrite(sg->sg.a, sizeof(mc_gg_status), sg->sg.n, fp);
fclose(fp);
free(buf);
}
psg_t* init_psg_t(uint64_t seed, ma_ug_t* ug, asg_t* rg, ug_opt_t *opt)
{
psg_t *s = NULL; CALLOC(s, 1);
s->xs = seed;
kv_gg_status *sa = init_mc_gg_status(ug, rg, opt->coverage_cut, opt->sources, opt->ruIndex,
asm_opt.hom_global_coverage_set?asm_opt.hom_global_coverage:((double)asm_opt.hom_global_coverage)/((double)HOM_PEAK_RATE),
asm_opt.polyploidy);
s->sg = *sa;
free(sa);
return s;
}
void destory_psg_t(psg_t **s)
{
free((*s)->sg.a);
free((*s));
}
int hic_short_align_poy(const enzyme *fn1, const enzyme *fn2, ha_ug_index* idx, ug_opt_t *opt)
{
double index_time = yak_realtime();
sldat_t sl;
kvec_hc_edge back_hc_edge;
kv_init(back_hc_edge.a);
sl.idx = idx;
sl.t_ch = idx->t_ch;
sl.chunk_size = 20000000;
sl.n_thread = asm_opt.thread_num;
sl.total_base = sl.total_pair = 0;
idx->hap_cnt = asm_opt.hap_occ;
kv_init(sl.hits.a); kv_init(sl.hits.idx); kv_init(sl.hits.occ);
if(!load_hc_hits(&sl.hits, asm_opt.output_file_name))
{
alignment_worker_pipeline(&sl, fn1, fn2);
write_hc_hits(&sl.hits, asm_opt.output_file_name);
}
hc_links link;
init_hc_links(&link, idx->ug->g->n_seq, idx->t_ch);
///H_partition hap;
bubble_type bub;
kv_u_trans_t k_trans;
kv_init(k_trans); kv_init(k_trans.idx);
psg_t *s = NULL;
mb_nodes_t u;
kv_init(u.bid); kv_init(u.idx); kv_init(u.u);
memset(&bub, 0, sizeof(bubble_type));
bub.round_id = 0; bub.n_round = asm_opt.n_weight;
resolve_tangles_hic(idx, &bub, &sl.hits, &k_trans);
measure_distance(idx, idx->ug, &sl.hits, &link, &bub, &(idx->t_ch->k_trans));
// if((asm_opt.flag & HA_F_VERBOSE_GFA) && load_psg_t(&s, asm_opt.output_file_name))
// {
// bub.round_id = bub.n_round;
// goto skip_flipping;
// }
s = init_psg_t(11, idx->ug, idx->read_g, opt);
for (bub.round_id = 0; bub.round_id < bub.n_round; bub.round_id++)
{
// identify_bubbles(idx->ug, &bub, idx->t_ch->is_r_het, &(idx->t_ch->k_trans));
renew_kv_u_trans(&k_trans, &link, &sl.hits, &(idx->t_ch->k_trans), idx, &bub, NULL, s->sg.a, 0);
// if(bub.round_id == 0) init_phase(idx, &k_trans, &bub, s);
// update_trans_g(idx, &k_trans, &bub);
/*******************************for debug************************************/
mc_solve_general(&k_trans, idx->ug->u.n, &(s->sg), asm_opt.polyploidy, 0, 1);
/*******************************for debug************************************/
/*******************************for debug************************************/
// if(bub.round_id == bub.n_round - 1)
// {
// debug_output_disconnected_hits(idx, &k_trans, &sl.hits, &link, &bub, s->s);
// }
/*******************************for debug************************************/
}
// write_psg_t(s, asm_opt.output_file_name);
skip_flipping:
verbose_het_stat(&bub);
if(asm_opt.polyploidy == 2) label_unitigs_sm(NULL, s->sg.a, idx->ug);
// print_kv_weight(&k_trans);
// horder_t *ho = init_horder_t(&sl.hits, idx->uID_bits, idx->pos_mode, idx->read_g, idx->ug, &bub, &(idx->t_ch->k_trans), opt, 3);
///print_hc_links(&link, 0, &hap);
// print_kv_u_trans(&k_trans, &link, s->s);
///print_bubbles(idx->ug, &bub, sl.hits.a.n?&sl.hits:NULL, idx->link, idx);
///print_hits(idx, &sl.hits, fn1);
///print_debug_bubble_graph(&bub, idx->ug, asm_opt.output_file_name);
// print_bubble_chain(&bub);
// destory_contig_partition(&hap);
// destory_horder_t(&ho);
kv_destroy(back_hc_edge.a);
kv_destroy(sl.hits.a);
kv_destroy(sl.hits.idx);
kv_destroy(sl.hits.occ);
destory_hc_links(&link);
kv_destroy(k_trans);
kv_destroy(k_trans.idx);
destory_psg_t(&s);
kv_destroy(u.bid); kv_destroy(u.idx); kv_destroy(u.u);
fprintf(stderr, "[M::%s::%.3f] processed %lu pairs; %lu bases\n", __func__, yak_realtime()-index_time, sl.total_pair, sl.total_base);
return 1;
}
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, uint32_t is_poy)
{
ug_index = NULL;
int exist = (asm_opt.load_index_from_disk?
@@ -15723,11 +15936,14 @@ void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt)
ug_index->read_g = read_g;
ug_index->t_ch = t_ch;
///test_unitig_index(ug_index, ug);
hic_short_align(asm_opt.hic_reads[0], asm_opt.hic_reads[1], ug_index, opt);
if(!is_poy) hic_short_align(asm_opt.hic_reads[0], asm_opt.hic_reads[1], ug_index, opt);
else hic_short_align_poy(asm_opt.hic_reads[0], asm_opt.hic_reads[1], ug_index, opt);
destory_hc_pt_index(ug_index);
}
void init_ug_idx(ma_ug_t *ug, uint64_t k, uint64_t up_bound, uint64_t low_bound, uint64_t build_idx)
{
ug_index = NULL;
+2 -1
View File
@@ -11,7 +11,7 @@
hc_edge* get_hc_edge(hc_links* link, uint64_t src, uint64_t dest, uint64_t dir);
hc_edge* push_hc_edge(hc_linkeage* x, uint64_t uID, double weight, int dir, uint64_t* d);
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt);
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, uint32_t is_poy);
void hic_benchmark(ma_ug_t *ug, asg_t* read_g);
typedef struct {
@@ -104,5 +104,6 @@ void destory_pdq(pdq* q);
uint32_t check_trans_relation_by_path(uint32_t v, uint32_t w, pdq* pqv, uint32_t* path_v, buf_t *resv,
pdq* pqw, uint32_t* path_w, buf_t *resw, asg_t *sg, uint8_t *dest, uint8_t df, uint32_t df_occ, double rate,
long long *dis);
void set_utg_by_dis(uint32_t v, pdq* pq, asg_t *g, kvec_t_u32_warp *res, uint32_t dis);
#endif
+31
View File
@@ -12,6 +12,37 @@ static void ha_hist_line(int c, int x, int exceed, int64_t cnt)
fprintf(stderr, " %lld\n", (long long)cnt);
}
void print_hist_lines(int n_cnt, int start_cnt, const int64_t *cnt)
{
const int hist_max = 100;
int i, start, low_i, max_i, max;
// determine the start point
assert(n_cnt > start_cnt);
start = cnt[1] > 0? 1 : 2;
// find the low point from the left
low_i = start > start_cnt? start : start_cnt;
for (i = low_i; i < n_cnt; ++i)
if (cnt[i] > cnt[i-1]) break;
low_i = i - 1;
fprintf(stderr, "[M::%s] lowest: count[%d] = %ld\n", __func__, low_i, (long)cnt[low_i]);
// find the highest peak
max_i = start > start_cnt? start : start_cnt, max = cnt[max_i];
for (i = max_i; i < n_cnt; ++i)
if (cnt[i] > max)
max = cnt[i], max_i = i;
fprintf(stderr, "[M::%s] highest: count[%d] = %ld\n", __func__, max_i, (long)cnt[max_i]);
for (i = start; i < n_cnt; ++i) {
int x, exceed = 0;
x = (int)((double)hist_max * cnt[i] / cnt[max_i] + .499);
if (x > hist_max) exceed = 1, x = hist_max; // may happen if cnt[2] is higher
if (i > max_i && x == 0) break;
ha_hist_line(i, x, exceed, cnt[i]);
}
}
int ha_analyze_count(int n_cnt, int start_cnt, const int64_t *cnt, int *peak_het)
{
const int hist_max = 100;
+1
View File
@@ -39,4 +39,5 @@ typedef struct {
horder_t *init_horder_t(kvec_pe_hit *i_hits, uint64_t i_hits_uid_bits, uint64_t i_hits_pos_mode,
asg_t *i_rg, ma_ug_t* i_ug, bubble_type* bub, kv_u_trans_t *ref, ug_opt_t *opt, uint32_t round);
void destory_horder_t(horder_t **h);
void horder_clean_sg_by_utg(asg_t *sg, ma_ug_t *ug);
#endif
+85 -11
View File
@@ -193,10 +193,8 @@ static int ha_ct_insert_list(ha_ct_t *h, int create_new, int n, const uint64_t *
khint_t k;
if ((a[j]&mask) != (a[0]&mask)) continue;
if (create_new) {
///for 0-th counting, g->b = NULL
if (g->b)
ins = (yak_bf_insert(g->b, x) == h->n_hash);
///for 0-th counting, g->b = NULL
///x = the high 52 bits of a[j] + low 12 bits 0
///the low 12 bits are used for counting
if (ins) {
@@ -515,6 +513,7 @@ KSEQ_INIT(gzFile, gzread)
#define HAF_RS_READ 0x10
#define HAF_CREATE_NEW 0x20
#define HAF_SKIP_READ 0x40
#define HAF_UG_READ 0x80
typedef struct { // global data structure for kt_pipeline()
const yak_copt_t *opt;
@@ -527,6 +526,7 @@ typedef struct { // global data structure for kt_pipeline()
ha_pt_t *pt;
const All_reads *rs_in;
All_reads *rs_out;
const ma_utg_v *us_in;
} pl_data_t;
typedef struct { // data structure for each step in kt_pipeline()
@@ -597,6 +597,24 @@ static void *worker_count(void *data, int step, void *in) // callback for kt_pip
if (s->sum_len >= p->opt->chunk_size)
break;
}
} else if(p->us_in) {
ma_utg_t *u;
while (p->n_seq < p->us_in->n) {
u = &(p->us_in->a[p->n_seq]);
if (s->n_seq == s->m_seq) {
s->m_seq = s->m_seq < 16? 16 : s->m_seq + (s->m_seq>>1);
REALLOC(s->len, s->m_seq);
REALLOC(s->seq, s->m_seq);
}
MALLOC(s->seq[s->n_seq], u->len);
memcpy(s->seq[s->n_seq], u->s, u->len);
s->len[s->n_seq++] = u->len;
++p->n_seq;
s->sum_len += u->len;
s->nk += u->len >= p->opt->k? u->len - p->opt->k + 1 : 0;
if (s->sum_len >= p->opt->chunk_size)
break;
}
} else {
while ((ret = kseq_read(p->ks)) >= 0) {
int l = (int)(p->ks->seq.l) - (int)(p->opt->adaLen) - (int)(p->opt->adaLen);
@@ -765,15 +783,18 @@ void debug_adapter(const hifiasm_opt_t *asm_opt, All_reads *rs)
exit(1);
}
static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt_t *p0, ha_ct_t *c0, const void *flt_tab, All_reads *rs, int64_t *n_seq)
static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt_t *p0, ha_ct_t *c0, const void *flt_tab, All_reads *rs, ma_utg_v *us, int64_t *n_seq)
{
///for 0-th counting, flag = HAF_COUNT_ALL|HAF_RS_WRITE_LEN|HAF_CREATE_NEW
int read_rs = (rs && (flag & HAF_RS_READ));
int ug_rs = (us && (flag & HAF_UG_READ));
pl_data_t pl;
gzFile fp = 0;
memset(&pl, 0, sizeof(pl_data_t));
pl.n_seq = *n_seq;
if (read_rs) {
if(ug_rs) {
pl.us_in = us;
} else if (read_rs) {
pl.rs_in = rs;
init_UC_Read(&pl.ucr);
} else {///for 0-th counting, go into here
@@ -804,7 +825,7 @@ static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt
kt_pipeline(3, worker_count, &pl, 3);
if (read_rs) {
destory_UC_Read(&pl.ucr);
} else {
} else if(!read_rs && !ug_rs) {
kseq_destroy(pl.ks);
gzclose(fp);
}
@@ -812,7 +833,7 @@ static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt
return pl.ct;
}
ha_ct_t *ha_count(const hifiasm_opt_t *asm_opt, int flag, ha_pt_t *p0, const void *flt_tab, All_reads *rs)
ha_ct_t *ha_count(const hifiasm_opt_t *asm_opt, int flag, ha_pt_t *p0, const void *flt_tab, All_reads *rs, ma_utg_v *us, int keep_adapter)
{
int i;
int64_t n_seq = 0;
@@ -836,10 +857,10 @@ ha_ct_t *ha_count(const hifiasm_opt_t *asm_opt, int flag, ha_pt_t *p0, const voi
///for ha_pt_gen, shoud be 0
opt.bf_shift = flag & HAF_COUNT_EXACT? 0 : asm_opt->bf_shift;
opt.n_thread = asm_opt->thread_num;
opt.adaLen = asm_opt->adapterLen;
opt.adaLen = (keep_adapter? asm_opt->adapterLen : 0);
///asm_opt->num_reads is the number of fastq files
for (i = 0; i < asm_opt->num_reads; ++i)
h = yak_count(&opt, asm_opt->read_file_names[i], flag|HAF_CREATE_NEW, p0, h, flt_tab, rs, &n_seq);
h = yak_count(&opt, asm_opt->read_file_names[i], flag|HAF_CREATE_NEW, p0, h, flt_tab, rs, us, &n_seq);
if (h && opt.bf_shift > 0)
ha_ct_destroy_bf(h);
return h;
@@ -914,6 +935,59 @@ void debug_ct_index(void* q_ct_idx, void* r_ct_idx)
* High-level interfaces *
*************************/
void *ha_ft_ug_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int hap_n)
{
yak_ft_t *flt_tab;
int64_t cnt[YAK_N_COUNTS];
int cutoff = hap_n + 1;
ha_ct_t *h;
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_UG_READ|HAF_COUNT_EXACT, NULL, NULL, NULL, us, 0);
ha_ct_hist(h, cnt, asm_opt->thread_num);
print_hist_lines(YAK_N_COUNTS, 1, cnt);
ha_ct_shrink(h, cutoff, YAK_MAX_COUNT, asm_opt->thread_num);
flt_tab = gen_hh(h);
ha_ct_destroy(h);
fprintf(stderr, "[M::%s::%.3f*%.2f@%.3fGB] ==> filtered out %ld k-mers occurring %d or more times\n", __func__,
yak_realtime(), yak_cpu_usage(), yak_peakrss_in_gb(), (long)kh_size(flt_tab), cutoff);
return (void*)flt_tab;
}
ha_pt_t *ha_pt_ug_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, ma_utg_v *us, int hap_n)
{
int64_t cnt[YAK_N_COUNTS], tot_cnt;
int i;
ha_ct_t *ct;
ha_pt_t *pt;
///HAF_COUNT_EXACT: no bf
ct = ha_count(asm_opt, HAF_COUNT_EXACT|HAF_UG_READ, NULL, flt_tab, NULL, us, 0);
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> counted %ld distinct minimizer k-mers\n", __func__,
yak_realtime(), yak_cpu_usage(), (long)ct->tot);
ha_ct_hist(ct, cnt, asm_opt->thread_num);
print_hist_lines(YAK_N_COUNTS, 1, cnt);
///here ha_ct_shrink is mostly used to remove k-mer appearing only 1 time
if (flt_tab == 0) {
ha_ct_shrink(ct, 2, hap_n, asm_opt->thread_num);
for (i = 2, tot_cnt = 0; i <= hap_n; ++i) tot_cnt += cnt[i] * i;
} else {
///Note: here is just to remove minimizer appearing YAK_MAX_COUNT times
///minimizer with YAK_MAX_COUNT occ may apper > YAK_MAX_COUNT times, so it may lead to overflow at ha_pt_gen
ha_ct_shrink(ct, 2, YAK_MAX_COUNT - 1, asm_opt->thread_num);
for (i = 2, tot_cnt = 0; i <= YAK_MAX_COUNT - 1; ++i) tot_cnt += cnt[i] * i;
}
pt = ha_pt_gen(ct, asm_opt->thread_num);
ha_count(asm_opt, HAF_COUNT_EXACT|HAF_UG_READ, pt, flt_tab, NULL, us, 0);
assert((uint64_t)tot_cnt == pt->tot_pos);
//ha_pt_sort(pt, asm_opt->thread_num);
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> indexed %ld positions\n", __func__,
yak_realtime(), yak_cpu_usage(), (long)pt->tot_pos);
return pt;
}
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode)
{
yak_ft_t *flt_tab;
@@ -921,7 +995,7 @@ void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int i
int peak_hom, peak_het, cutoff = YAK_MAX_COUNT - 1, ex_flag = 0;
if(is_hp_mode) ex_flag = HAF_RS_READ|HAF_SKIP_READ;
ha_ct_t *h;
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_RS_WRITE_LEN|ex_flag, NULL, NULL, rs);
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_RS_WRITE_LEN|ex_flag, NULL, NULL, rs, NULL, 1);
if((asm_opt->flag & HA_F_VERBOSE_GFA))
{
write_ct_index((void*)h, asm_opt->output_file_name);
@@ -966,7 +1040,7 @@ ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_f
}
if(is_hp_mode) extra_flag1 |= HAF_SKIP_READ, extra_flag2 |= HAF_SKIP_READ;
ct = ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag1, NULL, flt_tab, rs);
ct = ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag1, NULL, flt_tab, rs, NULL, 1);
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> counted %ld distinct minimizer k-mers\n", __func__,
yak_realtime(), yak_cpu_usage(), (long)ct->tot);
ha_ct_hist(ct, cnt, asm_opt->thread_num);
@@ -989,7 +1063,7 @@ ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_f
for (i = 2, tot_cnt = 0; i <= YAK_MAX_COUNT - 1; ++i) tot_cnt += cnt[i] * i;
}
pt = ha_pt_gen(ct, asm_opt->thread_num);
ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag2, pt, flt_tab, rs);
ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag2, pt, flt_tab, rs, NULL, 1);
assert((uint64_t)tot_cnt == pt->tot_pos);
//ha_pt_sort(pt, asm_opt->thread_num);
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> indexed %ld positions\n", __func__,
+38 -1
View File
@@ -18,6 +18,41 @@ typedef struct {
typedef struct { uint32_t n, m; ha_mz1_t *a; } ha_mz1_v;
typedef struct {
uint64_t x; ///x is the hash key
///rid is the read id, pos is the end pos of this minimizer, rev is the direction
///span is the length of this k-mer. For non-HPC k-mer, span may not be equal to k
uint64_t rid:30, pos:34;
uint16_t rev:1, span:15;
} ha_mzl_t;
typedef struct {
uint64_t rid:30, pos:34;
uint16_t rev:1, span:15;
} ha_mzl_idxpos_t;
typedef struct { uint32_t n, m; ha_mzl_t *a; } ha_mzl_v;
typedef struct { // a simplified version of kdq
int front, count;
int a[64];
} tiny_queue_t;
static inline void tq_push(tiny_queue_t *q, int x)
{
q->a[((q->count++) + q->front) & 0x3f] = x;
}
static inline int tq_shift(tiny_queue_t *q)
{
int x;
if (q->count == 0) return -1;
x = q->a[q->front++];
q->front &= 0x3f;
--q->count;
return x;
}
struct ha_pt_s;
typedef struct ha_pt_s ha_pt_t;
@@ -31,11 +66,12 @@ extern void *ha_flt_tab_hp;
extern ha_pt_t *ha_idx_hp;
extern void *ha_ct_table;
void *ha_ft_ug_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int hap_n);
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode);
int ha_ft_isflt(const void *hh, uint64_t y);
void ha_ft_destroy(void *h);
ha_pt_t *ha_pt_ug_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, ma_utg_v *us, int hap_n);
ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_from_store, int is_hp_mode, All_reads *rs, int *hom_cov, int *het_cov);
void ha_pt_destroy(ha_pt_t *h);
const ha_idxpos_t *ha_pt_get(const ha_pt_t *h, uint64_t hash, int *n);
@@ -62,6 +98,7 @@ void ha_triobin(const hifiasm_opt_t *opt);
void ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mz1_v *p, const void *hf);
void ha_sketch_query(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mz1_v *p, const void *hf, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct);
int ha_analyze_count(int n_cnt, int start_cnt, const int64_t *cnt, int *peak_het);
void print_hist_lines(int n_cnt, int start_cnt, const int64_t *cnt);
void debug_adapter(const hifiasm_opt_t *asm_opt, All_reads *rs);
static inline uint64_t yak_hash64(uint64_t key, uint64_t mask) // invertible integer hash function
+882 -1
View File
@@ -23,6 +23,9 @@ KRADIX_SORT_INIT(mc32, uint32_t, mc_generic_key, 4)
#define ma_x(z) (((z).x>>32))
#define ma_y(z) (((uint32_t)((z).x)))
#define mcb_pat(x, id) (&((x).m.a[(id)-1]))
#define mcp_de(x, id, m) ((x).z[((id)<<(x).hapN)+(m)])
uint8_t bit_filed[8] = {1, 2, 4, 8, 16, 32, 64, 128};
#define is_bit_set(id, a) ((a)[(id)>>3]&bit_filed[(id)&7])
@@ -2103,6 +2106,24 @@ void print_sc(const mc_opt_t *opt, const mc_g_t *mg, mc_svaux_t *b, t_w_t sc_opt
fprintf(stderr, "# iter: %u, sc_opt: %f, sc-local: %f, sc-global: %f\n", n_iter, sc_opt, w, mc_score_all_advance(mg->e, mg->s.a));
}
void print_mc_node(const mc_match_t *ma, mc_svaux_t *b, uint32_t id)
{
fprintf(stderr, "[M::%s::utg%.6ul-hap%u]\n", __func__, id, b->s[id]>0?1:b->s[id]<0?2:0);
w_t w[128], z[4];
uint32_t o, n, i, hn = 4;
int8_t s;
for (i = 0; i < hn; i++) w[i] = 0;
o = ma->idx.a[id] >> 32;
n = (uint32_t)ma->idx.a[id];
for (i = 0; i < n; ++i)
{
s = b->s[ma_y(ma->ma.a[o + i])];
w[s>0?1:s<0?2:0] += ma->ma.a[o + i].w;
}
z[0] = z[3] = 0; z[1] = b->z[id].z[0]; z[2] = b->z[id].z[1];
for (i = 0; i < hn; i++) fprintf(stderr, "w[%u]-%f, z[%u]-%f\n", i, w[i], i, z[i]);
}
uint32_t mc_solve_cc(const mc_opt_t *opt, const mc_g_t *mg, mc_svaux_t *b, uint32_t cc_off, uint32_t cc_size)
{
uint32_t j, k, n_iter = 0, flush = opt->max_iter * 50;
@@ -2130,6 +2151,11 @@ uint32_t mc_solve_cc(const mc_opt_t *opt, const mc_g_t *mg, mc_svaux_t *b, uint3
b->z[b->cc_node[j]] = b->z_opt[b->cc_node[j]];
}
}
fprintf(stderr, "\nBeg-[M::%s::score->%f]\n", __func__, mc_score(mg->e, b));
print_mc_node(mg->e, b, 3838);
print_mc_node(mg->e, b, 36880);
// print_sc(opt, mg, b, sc_opt, n_iter);
// mc_reset_z_debug(mg->e, b);
// print_sc(opt, mg->e, b, sc_opt, n_iter);
@@ -2163,6 +2189,10 @@ uint32_t mc_solve_cc(const mc_opt_t *opt, const mc_g_t *mg, mc_svaux_t *b, uint3
}
sc_opt = sc;
}
fprintf(stderr, "\n");
print_mc_node(mg->e, b, 3838);
print_mc_node(mg->e, b, 36880);
}
for (j = 0; j < b->cc_size; ++j)
@@ -2170,7 +2200,8 @@ uint32_t mc_solve_cc(const mc_opt_t *opt, const mc_g_t *mg, mc_svaux_t *b, uint3
b->s[b->cc_node[j]] = b->s_opt[b->cc_node[j]];
b->z[b->cc_node[j]] = b->z_opt[b->cc_node[j]];
}
fprintf(stderr, "End-[M::%s::score->%f]\n", __func__, mc_score(mg->e, b));
return n_iter;
}
@@ -2808,6 +2839,19 @@ void debug_mc_g_t(const char* name)
exit(1);
}
void print_hap_s(int8_t *s, uint32_t sn)
{
fprintf(stderr, "\n[M::%s]\n", __func__);
uint32_t i;
for (i = 0; i < sn; i++)
{
fprintf(stderr,"utg%.6ul\t", i + 1);
if(s[i] > 0) fprintf(stderr, "h%u\n", 1);
else if(s[i] < 0) fprintf(stderr, "h%u\n", 2);
else fprintf(stderr,"\n");
}
}
void mc_solve(hap_overlaps_list* ovlp, trans_chain* t_ch, kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g, double f_rate, uint8_t* trio_flag, uint32_t renew_s, int8_t *s, uint32_t is_sys, bubble_type* bub, kv_u_trans_t *ref)
{
mc_opt_t opt;
@@ -2829,5 +2873,842 @@ void mc_solve(hap_overlaps_list* ovlp, trans_chain* t_ch, kv_u_trans_t *ta, ma_u
if(ovlp) clean_ovlp_by_mc(mg, ovlp);
print_hap_s(s, ug->u.n);
destory_mc_g_t(&mg);
}
void comp(int m, int N, int M, mcb_t *p, int *c)
{
if (m == M + 1)
{
int i;
mcg_node_t x = 0;
for (i = 0; i < M; i++) x |= ((mcg_node_t)1<<(c[i+1]-1));
kv_push(mcg_node_t, *p, x);
}
else
{
for (c[m] = c[m - 1] + 1; c[m] <= N - M + m; c[m]++)
{
comp(m + 1, N, M, p, c);
}
}
}
void get_mcb(uint32_t n, uint32_t m, mcb_t *p, int *c)
{
memset(c, 0, sizeof(int)*n+1);
p->n = 0;
comp(1, n, m, p, c);
}
mc_gg_t *init_mc_gg_t(uint32_t un, kv_gg_status *s, uint16_t hapN)
{
uint32_t i;
int *c = NULL; CALLOC(c, hapN+1);
mc_gg_t *p = NULL; CALLOC(p, 1);
// p->ug = ug; p->rg = read_g;
p->un = un; p->s = s; p->hN = hapN;
CALLOC(p->m.a, p->hN); p->m.n = p->m.m = hapN;
for (i = 0; i < p->hN; i++) get_mcb(hapN, i+1, &(p->m.a[i]), c);
p->mask = (1<<hapN); p->mask--;
free(c);
return p;
}
void destory_mc_gg_t(mc_gg_t **p)
{
uint32_t i;
if(!p || !(*p)) return;
for (i = 0; i < (*p)->m.m; i++)
{
free((*p)->m.a[i].a);
}
free((*p)->m.a);
if((*p)->e)
{
kv_destroy((*p)->e->idx);
kv_destroy((*p)->e->ma);
free((*p)->e->cc);
free((*p)->e);
}
free((*p));
}
kv_gg_status *init_mc_gg_status(ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut,
ma_hit_t_alloc* sources, R_to_U* ruIndex, uint64_t t_cov, uint16_t hapN)
{
fprintf(stderr, "t_cov-%lu\n", t_cov);
uint64_t *covs = NULL, i, k, k_i, c = t_cov/hapN, c_min, c_max;
uint32_t len[2];
uint8_t *vis = NULL; CALLOC(vis, read_g->n_seq);
kv_gg_status *p = NULL;
CALLOC(covs, hapN);
for (i = 0; i < hapN; i++)
{
c_min = ((i+1)*c) - (0.6*c);
c_max = ((i+1)*c) + (0.6*c);
if(i == 0) c_min = 0;
if(i == hapN) c_max = (uint32_t)-1;
covs[i] = (c_max<<32)|c_min;
}
CALLOC(p, 1);
p->n = p->m = ug->u.n;
CALLOC(p->a, p->n);
for (i = 0; i < p->n; i++)
{
c = get_utg_cov(ug, i, read_g, coverage_cut, sources, ruIndex, vis);
p->a[i].h[0] = p->a[i].h[1] = (uint16_t)-1; p->a[i].s = 0; k_i = 0;
p->a[i].hw[0] = 1; p->a[i].hw[1] = 0; p->a[i].hc = 0;
for (k = 0; k < hapN; k++)
{
c_min = (uint32_t)covs[k]; c_max = covs[k]>>32;
if(c < c_min || c >= c_max) continue;
p->a[i].h[k_i++] = k + 1;
}
if(p->a[i].h[1] == (uint16_t)-1) continue;
len[0] = (c >= (p->a[i].h[0]*(t_cov/hapN))?
c - (p->a[i].h[0]*(t_cov/hapN)) : (p->a[i].h[0]*(t_cov/hapN)) - c);
len[1] = (c >= (p->a[i].h[1]*(t_cov/hapN))?
c - (p->a[i].h[1]*(t_cov/hapN)) : (p->a[i].h[1]*(t_cov/hapN)) - c);
if(len[0] > len[1])
{
k_i = p->a[i].h[0];
p->a[i].h[0] = p->a[i].h[1];
p->a[i].h[1] = k_i;
k_i = len[0];
len[0] = len[1];
len[1] = k_i;
}
p->a[i].hw[0] = (double)len[1]/(double)(len[0]+len[1]);
p->a[i].hw[1] = (double)len[0]/(double)(len[0]+len[1]);
}
free(covs); free(vis);
return p;
}
void update_mc_edges_general(mc_gg_t *mg, kv_u_trans_t *ta, uint16_t hapN)
{
uint32_t i, k;
mc_edge_t *ma = NULL;
CALLOC(mg->e, 1);
mg->e->n_seq = mg->un;
kv_init(mg->e->idx); kv_init(mg->e->ma);
for (i = 0; i < ta->n; ++i)
{
if(ta->a[i].del) continue;
if(mg->s->a[ta->a[i].qn].h[0] >= hapN && mg->s->a[ta->a[i].qn].h[1] >= hapN) continue;
if(mg->s->a[ta->a[i].tn].h[0] >= hapN && mg->s->a[ta->a[i].tn].h[1] >= hapN) continue;
kv_pushp(mc_edge_t, mg->e->ma, &ma);
ma->x = (uint64_t)ta->a[i].qn << 32 | ta->a[i].tn;
ma->w = w_cast((ta->a[i].nw));
}
for (i = k = 0; i < mg->e->ma.n; i++)
{
if(mg->e->ma.a[i].w == 0) continue;
mg->e->ma.a[k] = mg->e->ma.a[i];
k++;
}
mg->e->ma.n = k;
radix_sort_mce(mg->e->ma.a, mg->e->ma.a + mg->e->ma.n);
mc_edges_idx(mg->e);
mc_edges_symm(mg->e);
}
typedef struct {
t_w_t *z;
uint32_t hapN;
} mc_poy_t;
typedef struct {
uint64_t x; // RNG
uint32_t cc_off, cc_size;
kvec_t(uint64_t) cc_edge;
uint32_t *cc_node;
uint32_t *bfs, *bfs_mark;
mc_poy_t *z, *z_opt;///keep scores to nodes(1) and nodes(-1)
mc_gg_status *s, *s_opt;
mcb_t *m;
mcg_node_t mask;
uint32_t hapN;
} mcgg_svaux_t;
mc_poy_t *init_mc_poy_t(uint32_t un, uint32_t hapN)
{
uint32_t i;
mc_poy_t *z = NULL; CALLOC(z, 1);
MALLOC(z->z, ((uint32_t)un<<hapN));
for (i = 0; i < ((uint32_t)un<<hapN); i++) z->z[i] = 0;
z->hapN = hapN;
return z;
}
void destroy_mc_poy_t(mc_poy_t **p)
{
if(!p || !(*p)) return;
free((*p)->z); free(*p);
}
mcgg_svaux_t *mcgg_svaux_init(const mc_gg_t *mg, uint64_t x, uint32_t hapN)
{
uint32_t st, i, max_cc = 0;
mc_match_t *ma = mg->e;
mcgg_svaux_t *b;
CALLOC(b, 1);
b->x = x;
for (st = 0, i = 1; i <= ma->n_seq; ++i)
if (i == ma->n_seq || ma->cc[st]>>32 != ma->cc[i]>>32)
max_cc = max_cc > i - st? max_cc : i - st, st = i;
kv_init(b->cc_edge);
MALLOC(b->cc_node, max_cc);
b->s = mg->s->a;
CALLOC(b->s_opt, ma->n_seq);
MALLOC(b->bfs, ma->n_seq);
MALLOC(b->bfs_mark, ma->n_seq);
memset(b->bfs_mark, -1, ma->n_seq*sizeof(uint32_t));
b->z = init_mc_poy_t(ma->n_seq, hapN);
b->z_opt = init_mc_poy_t(ma->n_seq, hapN);
b->m = mg->m.a;
b->mask = ((mcg_node_t)1)<<hapN; b->mask--;
b->hapN = hapN;
return b;
}
void mcgg_svaux_destroy(mcgg_svaux_t *b)
{
b->s = NULL;
kv_destroy(b->cc_edge); free(b->cc_node);
free(b->s); free(b->s_opt);
destroy_mc_poy_t(&(b->z));
destroy_mc_poy_t(&(b->z_opt));
free(b->bfs); free(b->bfs_mark);
free(b);
}
static inline mcg_node_t kr_drand_node(uint64_t id, mcgg_svaux_t *b, uint16_t *hc)
{
uint16_t hn = b->s[id].h[0]-1;
if(hc) (*hc) = 0;
b->x = kr_splitmix64(b->x);
if(b->s[id].h[1] != (uint16_t)-1)
{
union { uint64_t i; double d; } u;
u.i = 0x3FFULL << 52 | (b->x) >> 12;
if((u.d - 1.0) > b->s[id].hw[0])
{
hn = b->s[id].h[1]-1;
if(hc) (*hc) = 1;
}
}
return b->m[hn].a[(b->x)%b->m[hn].n];
}
static inline mcg_node_t kr_drand_node_ref(uint64_t id, mcgg_svaux_t *b, uint16_t *hc, mcg_node_t ref, uint64_t rev)
{
uint64_t i, k, m, mm, mn, mi, cn, hn = b->s[id].h[0];
mcg_node_t t;
if(hc) (*hc) = 0;
b->x = kr_splitmix64(b->x);
if(b->s[id].h[1] != (uint16_t)-1)
{
union { uint64_t i; double d; } u;
u.i = 0x3FFULL << 52 | (b->x) >> 12;
if((u.d - 1.0) > b->s[id].hw[0])
{
hn = b->s[id].h[1];
if(hc) (*hc) = 1;
}
}
if(rev) ref ^= (mcg_node_t)-1;
ref &= b->mask;
t = ref; cn = 0;
while (t)
{
cn += t&1;
t >>= 1;
}
if(cn == hn) return ref;
if(cn>=hn) mm=cn, m=1, mn=cn-hn;///1->0
else mm=b->hapN-cn, m=0, mn=hn-cn;///0->1
for (i = 0; i < mn; i++)
{
mi = (b->x%mm);
for (k = 0; k < b->hapN; k++)
{
if(((ref>>k)&1)!=m) continue;
if(mi == 0)
{
ref ^= ((mcg_node_t)1<<k);
break;
}
mi--;
}
mm--;
}
return ref;
}
void debug_hapM(mc_gg_status *s, const char* cmd)
{
uint32_t i, cn, hn = s->h[s->hc];
mcg_node_t ref = s->s;
for (i = cn = 0; i < 32; i++) cn += ((ref>>i)&1);
if (cn != hn) fprintf(stderr, "%s-ERROR-cn, cn-%u, hn-%u\n", cmd, cn, hn);
else fprintf(stderr, "%s-pass-cn, cn-%u, hn-%u\n", cmd, cn, hn);
}
void mcgg_reset_z(const mc_match_t *ma, mcgg_svaux_t *b)
{
uint32_t i;
for (i = 0; i < b->cc_size; ++i) {
uint32_t k = (uint32_t)ma->cc[b->cc_off + i];///uid
uint32_t o = ma->idx.a[k] >> 32;
uint32_t j, n = (1<<b->z->hapN);
for (j = 0; j < n; j++) mcp_de(*(b->z), k, j) = 0;
n = (uint32_t)ma->idx.a[k];
for (j = 0; j < n; ++j) {
const mc_edge_t *e = &ma->ma.a[o + j];
uint32_t t = ma_y(*e);
mcp_de(*(b->z), k, b->s[t].s) += e->w;
}
}
}
t_w_t mcgg_score(const mc_match_t *ma, mcgg_svaux_t *b)
{
uint32_t i, j, n = (1<<b->z->hapN);
t_w_t z = 0;
for (i = 0; i < b->cc_size; ++i) {
uint32_t k = (uint32_t)ma->cc[b->cc_off + i];///uid
for (j = 0; j < n; j++)
{
z += ((b->s[k].s&((mcg_node_t)j))?-mcp_de(*(b->z), k, j):mcp_de(*(b->z), k, j));
}
}
return z;
}
t_w_t mcgg_init_spin(const mc_match_t *ma, mcgg_svaux_t *b)
{
uint32_t i;
b->cc_edge.n = 0;
for (i = 0; i < b->cc_size; ++i) {///how many nodes
uint32_t k = (uint32_t)ma->cc[b->cc_off + i];///node id
b->cc_node[i] = k;
if(b->s[k].s == 0) break;
}
if(i >= b->cc_size) goto passed;
for (i = 0; i < b->cc_size; ++i) {///how many nodes
uint32_t k = (uint32_t)ma->cc[b->cc_off + i];///node id
uint32_t o = ma->idx.a[k] >> 32;///cc group id
uint32_t n = (uint32_t)ma->idx.a[k], j;
b->cc_node[i] = k;
for (j = 0; j < n; ++j) {
w_t w = ma->ma.a[o + j].w;
w = w > 0? w : -w;
kv_push(uint64_t, b->cc_edge, (uint64_t)((uint32_t)-1 - ((uint32_t)w)) << 32 | (o + j));
}
}
radix_sort_mc64(b->cc_edge.a, b->cc_edge.a + b->cc_edge.n);
for (i = 0; i < b->cc_edge.n; ++i) { // from the strongest edge to the weakest
const mc_edge_t *e = &ma->ma.a[(uint32_t)b->cc_edge.a[i]];
uint32_t n1 = ma_x(*e), n2 = ma_y(*e);
if (b->s[n1].s == 0 && b->s[n2].s == 0) {
b->s[n1].s = kr_drand_node(n1, b, &(b->s[n1].hc));
debug_hapM(&(b->s[n1]), "s0");
b->s[n2].s = kr_drand_node_ref(n2, b, &(b->s[n2].hc), b->s[n1].s, e->w>0?1:0);
debug_hapM(&(b->s[n2]), "s1");
}
else if(b->s[n1].s == 0)
{
b->s[n1].s = kr_drand_node_ref(n1, b, &(b->s[n1].hc), b->s[n2].s, e->w>0?1:0);
debug_hapM(&(b->s[n1]), "s2");
}
else if(b->s[n2].s == 0)
{
b->s[n2].s = kr_drand_node_ref(n2, b, &(b->s[n2].hc), b->s[n1].s, e->w>0?1:0);
debug_hapM(&(b->s[n2]), "s3");
}
}
passed:
mcgg_reset_z(ma, b);
return mcgg_score(ma, b);
}
static mcg_node_t get_max_m(uint64_t id, mcgg_svaux_t *b)
{
uint32_t hn = b->s[id].h[0]-1, k, j, n = (1<<b->z->hapN);
mcg_node_t m, *p = NULL;
t_w_t z, max_z = -(1<<30);
for (k = 0; k < b->m[hn].n; k++)
{
m = b->m[hn].a[k]; z = 0;
for (j = 0; j < n; j++)
{
z += ((m&((mcg_node_t)j))?-mcp_de(*(b->z), id, j):mcp_de(*(b->z), id, j));
}
if(!p || max_z < z || (max_z == z && m == b->s[id].s)) p = &(b->m[hn].a[k]), max_z = z;
}
if(b->s[id].h[1] != (uint16_t)-1)
{
hn = b->s[id].h[1]-1;
for (k = 0; k < b->m[hn].n; k++)
{
m = b->m[hn].a[k]; z = 0;
for (j = 0; j < n; j++)
{
z += ((m&((mcg_node_t)j))?-mcp_de(*(b->z), id, j):mcp_de(*(b->z), id, j));
}
if(!p || max_z < z || (max_z == z && m == b->s[id].s)) p = &(b->m[hn].a[k]), max_z = z;
}
}
return (*p);
}
///k is uid
static void mcgg_set_spin(const mc_match_t *ma, mcgg_svaux_t *b, uint32_t k, mcg_node_t s, const char* cmd)
{
uint32_t o, j, n;
mcg_node_t s0 = b->s[k].s;
/*******************************for debug************************************/
// mcg_node_t t = s;
// o = 0;
// while (t) {
// o += (t&1); t>>=1;
// }
// if(o != b->s[k].h[0] && o != b->s[k].h[1]) fprintf(stderr, "cmd-%s, ERROR-mcgg-1\n", cmd);
// if(s0 == s) fprintf(stderr, "cmd-%s, ERROR-mcgg-2\n", cmd);
/*******************************for debug************************************/
if (s0 == s) return;
o = ma->idx.a[k] >> 32;
n = (uint32_t)ma->idx.a[k];
for (j = 0; j < n; ++j) {
const mc_edge_t *e = &ma->ma.a[o + j];
uint32_t t = ma_y(*e);///1->z[0]; (-1)->z[1];
mcp_de(*(b->z), t, s0) -= e->w;
mcp_de(*(b->z), t, s) += e->w;
}
b->s[k].s = s;
}
static t_w_t mcgg_optimize_local(const mc_opt_t *opt, const mc_match_t *ma, mcgg_svaux_t *b, uint32_t *n_iter)
{
uint32_t i, n_flip = 0;
int32_t n_iter_local = 0;
mcg_node_t ms;
while (n_iter_local < opt->max_iter) {
++(*n_iter);
ks_shuffle_uint32_t(b->cc_size, b->cc_node, &b->x);
for (i = n_flip = 0; i < b->cc_size; ++i) {
uint32_t k = b->cc_node[i];///uid
ms = get_max_m(k, b);
if(ms != b->s[k].s)
{
mcgg_set_spin(ma, b, k, ms, __func__);///no need to change the score of k itself
// debug_hapM(&(b->s[k]), "s4");
++n_flip;
}
}
++n_iter_local;
if (n_flip == 0) break;
}
return mcgg_score(ma, b);
}
void inline back_status(mcgg_svaux_t *b, uint32_t id, uint32_t to_opt)
{
if(to_opt)
{
b->s_opt[id] = b->s[id];
memcpy(b->z_opt->z+(id<<b->z->hapN), b->z->z+(id<<b->z->hapN), (1<<b->z->hapN)*sizeof(t_w_t));
}
else
{
b->s[id] = b->s_opt[id];
memcpy(b->z->z+(id<<b->z->hapN), b->z_opt->z+(id<<b->z->hapN), (1<<b->z->hapN)*sizeof(t_w_t));
}
}
static inline mcg_node_t kr_drand_node_new(uint64_t id, mcgg_svaux_t *b)
{
uint32_t hn = b->s[id].h[0]-1, is_old = 1, k;
if((b->s[id].h[1] != (uint16_t)-1) && (kr_drand_r(&b->x) > b->s[id].hw[0]))
{
hn = b->s[id].h[1]-1; is_old = 0;
}
b->x = kr_splitmix64(b->x);
k = b->x%(b->m[hn].n-is_old);
if(b->m[hn].a[k] == b->s[id].s) k = b->m[hn].n-1;
return b->m[hn].a[k];
}
static void mcgg_perturb(const mc_opt_t *opt, const mc_match_t *ma, mcgg_svaux_t *b)
{
uint32_t i;
for (i = 0; i < b->cc_size; ++i) {
uint32_t k = (uint32_t)ma->cc[b->cc_off + i];///node id
double y;
y = kr_drand_r(&b->x);
if (y < opt->f_perturb)
mcgg_set_spin(ma, b, k, kr_drand_node_new(k, b), __func__);
}
}
static uint32_t mcgg_bfs(const mc_match_t *ma, mcgg_svaux_t *b, uint32_t k0, uint32_t bfs_round, uint32_t max_size)
{
uint32_t i, n_bfs = 0, st, en, r;
b->bfs[n_bfs++] = k0, b->bfs_mark[k0] = k0;
st = 0, en = n_bfs;
for (r = 0; r < bfs_round; ++r) {
for (i = st; i < en; ++i) {
uint32_t k = b->bfs[i];
uint32_t o = ma->idx.a[k] >> 32;
uint32_t n = (uint32_t)ma->idx.a[k], j;
for (j = 0; j < n; ++j) {
uint32_t t = (uint32_t)ma->ma.a[o + j].x;
if (b->bfs_mark[t] != k0)
b->bfs[n_bfs++] = t, b->bfs_mark[t] = k0;
}
}
st = en, en = n_bfs;
if (max_size > 0 && n_bfs > max_size) break;
}
return n_bfs;
}
///bfs_round is 3
static void mcgg_perturb_node(const mc_opt_t *opt, const mc_match_t *ma, mcgg_svaux_t *b, int32_t bfs_round)
{
uint32_t i, k, n_bfs = 0;
k = (uint32_t)(kr_drand_r(&b->x) * b->cc_size + .499);
if(k >= b->cc_size) k = b->cc_size - 1;
k = (uint32_t)ma->cc[b->cc_off + k];///node id
n_bfs = mcgg_bfs(ma, b, k, bfs_round, (int32_t)(b->cc_size * opt->f_perturb));
for (i = 0; i < n_bfs; ++i)
mcgg_set_spin(ma, b, b->bfs[i], kr_drand_node_new(b->bfs[i], b), __func__);
}
void print_mcgg_node(const mc_match_t *ma, mcgg_svaux_t *b, uint32_t id)
{
fprintf(stderr, "[M::%s::utg%.6ul-hap%u]\n", __func__, id, b->s[id].s);
w_t w[128];
uint32_t o, n, i, hn = (1<<b->z->hapN);
for (i = 0; i < hn; i++) w[i] = 0;
o = ma->idx.a[id] >> 32;
n = (uint32_t)ma->idx.a[id];
for (i = 0; i < n; ++i) w[b->s[ma_y(ma->ma.a[o + i])].s] += ma->ma.a[o + i].w;
for (i = 0; i < hn; i++) fprintf(stderr, "w[%u]-%f, z[%u]-%f\n", i, w[i], i, mcp_de(*(b->z), id, i));
}
uint32_t mcgg_solve_cc(const mc_opt_t *opt, const mc_gg_t *mg, mcgg_svaux_t *b, uint32_t cc_off, uint32_t cc_size)
{
// double t0, t1, tt0, tt1;
uint32_t j, k, n_iter = 0, flush = opt->max_iter * 50;
t_w_t sc_opt = -(1<<30), sc;///problem-w
b->cc_off = cc_off, b->cc_size = cc_size;
if (b->cc_size < 2) return 0;
sc_opt = mcgg_init_spin(mg->e, b);
if (b->cc_size == 2) return 0;
for (j = 0; j < b->cc_size; ++j) back_status(b, b->cc_node[j], 1);
sc = mcgg_optimize_local(opt, mg->e, b, &n_iter);
if (sc > sc_opt)
{
for (j = 0; j < b->cc_size; ++j) back_status(b, b->cc_node[j], 1);
sc_opt = sc;
}
else
{
for (j = 0; j < b->cc_size; ++j) back_status(b, b->cc_node[j], 0);
}
fprintf(stderr, "\nBeg-[M::%s::score->%f]\n", __func__, mcgg_score(mg->e, b));
print_mcgg_node(mg->e, b, 3838);
print_mcgg_node(mg->e, b, 36880);
// tt0 = tt1 = 0;
for (k = 0; k < (uint32_t)opt->n_perturb; ++k) {
// t0 = yak_realtime();
if (k&1) mcgg_perturb(opt, mg->e, b);
else mcgg_perturb_node(opt, mg->e, b, 3);
// tt0 += yak_realtime()-t0;
// fprintf(stderr, "\n++(%u) sc_pre: %f\n", k, mcgg_score(mg->e, b));
// t1 = yak_realtime();
sc = mcgg_optimize_local(opt, mg->e, b, &n_iter);
// tt1 += yak_realtime()-t1;
// fprintf(stderr, "++(%u) sc_after: %f\n", k, mcgg_score(mg->e, b));
if (sc > sc_opt) {
for (j = 0; j < b->cc_size; ++j) back_status(b, b->cc_node[j], 1);
sc_opt = sc;
} else {
for (j = 0; j < b->cc_size; ++j) back_status(b, b->cc_node[j], 0);
}
if((n_iter%flush) == 0)
{
mcgg_reset_z(mg->e, b);
sc = mcgg_score(mg->e, b);
for (j = 0; j < b->cc_size; ++j) back_status(b, b->cc_node[j], 1);
sc_opt = sc;
}
fprintf(stderr, "\n");
print_mcgg_node(mg->e, b, 3838);
print_mcgg_node(mg->e, b, 36880);
// if((k&31)==0)fprintf(stderr, "+++(%u) sc: %f, sc_opt: %f, tt0: %.3f, tt1: %.3f\n", k, sc, sc_opt, tt0, tt1);
}
for (j = 0; j < b->cc_size; ++j) back_status(b, b->cc_node[j], 0);
fprintf(stderr, "End-[M::%s::score->%f]\n", __func__, mcgg_score(mg->e, b));
return n_iter;
}
void mc_solve_core_genral(const mc_opt_t *opt, mc_gg_t *mg, uint32_t hapN)
{
double index_time = yak_realtime();
uint32_t st, i;
mcgg_svaux_t *b;
mc_g_cc(mg->e);
b = mcgg_svaux_init(mg, opt->seed, hapN);
// if(VERBOSE_CUT)
// {
// fprintf(stderr, "\n\n\n\n\n*************beg-[M::%s::score->%f] ==> Partition\n", __func__, mc_score_all_advance(mg->e, mg->s.a));
// }
for (st = 0, i = 1; i <= mg->e->n_seq; ++i) {
if (i == mg->e->n_seq || mg->e->cc[st]>>32 != mg->e->cc[i]>>32) {
mcgg_solve_cc(opt, mg, b, st, i - st);
st = i;
}
}
// if(VERBOSE_CUT)
// {
// fprintf(stderr, "##############end-[---M::%s::score->%f] ==> Partition\n", __func__, mc_score_all(mg->e, b));
// }
///mc_write_info(g, b);
mcgg_svaux_destroy(b);
fprintf(stderr, "[M::%s::%.3f] ==> Partition\n", __func__, yak_realtime()-index_time);
}
void print_mcb(mc_gg_t *mg)
{
uint32_t i, k, m;
mcg_node_t t;
mcb_t *p;
for (i = 0; i < mg->m.n; i++)
{
p = mcb_pat(*mg, i + 1);
fprintf(stderr, "# haplotypes: %u, # combination: %u\n", i+1, (uint32_t)p->n);
for (k = 0; k < p->n; k++)
{
t = p->a[k];
for (m = 0; m < 32; m++)
{
if((t>>m)&1) fprintf(stderr, "%u\t", m);
}
fprintf(stderr, "\n");
}
}
}
void print_hap_p(kv_gg_status *s)
{
fprintf(stderr, "\n[M::%s]\n", __func__);
uint32_t i, h;
mcg_node_t m;
for (i = 0; i < s->n; i++)
{
fprintf(stderr,"utg%.6ul\t", i + 1);
m = s->a[i].s; h = 0;
while (m) {
h++;
if(m&1) fprintf(stderr, "h%u\t", h);
m>>=1;
}
fprintf(stderr,"\n");
}
}
void write_mc_gg_dump(kv_u_trans_t *ta, uint32_t un, kv_gg_status *s, uint16_t hapN, const char* fn)
{
fprintf(stderr, "\n[M::%s]\n", __func__);
char *buf = (char*)calloc(strlen(fn) + 50, 1);
sprintf(buf, "%s.hic.dbg.dump.bin", fn);
FILE* fp = fopen(buf, "w");
fwrite(&ta->n, sizeof(ta->n), 1, fp);
fwrite(ta->a, sizeof(u_trans_t), ta->n, fp);
fwrite(&ta->idx.n, sizeof(ta->idx.n), 1, fp);
fwrite(ta->idx.a, sizeof(uint64_t), ta->idx.n, fp);
fwrite(&un, sizeof(un), 1, fp);
fwrite(&(s->n), sizeof(s->n), 1, fp);
fwrite(s->a, sizeof(mc_gg_status), s->n, fp);
fwrite(&hapN, sizeof(hapN), 1, fp);
fclose(fp);
free(buf);
}
uint32_t load_mc_gg_dump(kv_u_trans_t **rta, uint32_t *un, kv_gg_status **rs, uint16_t *hapN, const char* fn)
{
fprintf(stderr, "\n[M::%s]\n", __func__);
kv_u_trans_t *ta = NULL;
kv_gg_status *s = NULL;
uint64_t flag = 0;
char *buf = (char*)calloc(strlen(fn) + 25, 1);
sprintf(buf, "%s.hic.dbg.dump.bin", fn);
FILE* fp = NULL;
fp = fopen(buf, "r");
if(!fp)
{
free(buf);
return 0;
}
CALLOC(ta, 1);
flag += fread(&ta->n, sizeof(ta->n), 1, fp);
ta->m = ta->n; MALLOC(ta->a, ta->n);
flag += fread(ta->a, sizeof(u_trans_t), ta->n, fp);
flag += fread(&ta->idx.n, sizeof(ta->idx.n), 1, fp);
ta->idx.m = ta->idx.n; MALLOC(ta->idx.a, ta->idx.n);
flag += fread(ta->idx.a, sizeof(uint64_t), ta->idx.n, fp);
flag += fread(un, sizeof(*un), 1, fp);
CALLOC(s, 1);
flag += fread(&(s->n), sizeof(s->n), 1, fp);
s->m = s->n; MALLOC(s->a, s->n);
flag += fread(s->a, sizeof(mc_gg_status), s->n, fp);
flag += fread(hapN, sizeof(*hapN), 1, fp);
*rta = ta; *rs = s;
fclose(fp);
free(buf);
return 1;
}
mc_g_t* to_mc_g_t(kv_u_trans_t *ta, kv_gg_status *s, uint32_t un)
{
fprintf(stderr, "[M::%s]\n", __func__);
uint32_t i, k;
mc_edge_t *ma = NULL;
mc_g_t *mg = NULL; CALLOC(mg, 1);
CALLOC(mg->e, 1);
mg->e->n_seq = un;
kv_init(mg->e->idx); kv_init(mg->e->ma);
// double sc = get_w_scale(ta);
// fprintf(stderr, "sc: %f\n", sc);
for (i = 0; i < ta->n; ++i)
{
if(ta->a[i].del) continue;
kv_pushp(mc_edge_t, mg->e->ma, &ma);
ma->x = (uint64_t)ta->a[i].qn << 32 | ta->a[i].tn;
// ma->w = w_cast((ta->a[i].nw*sc));
ma->w = w_cast((ta->a[i].nw));
}
for (i = k = 0; i < mg->e->ma.n; i++)
{
if(mg->e->ma.a[i].w == 0) continue;
mg->e->ma.a[k] = mg->e->ma.a[i];
k++;
}
mg->e->ma.n = k;
radix_sort_mce(mg->e->ma.a, mg->e->ma.a + mg->e->ma.n);
mc_merge_dup(mg);
mc_edges_idx(mg->e);
mc_edges_symm(mg->e);
kv_init(mg->s);
mg->s.m = mg->s.n = s->n; CALLOC(mg->s.a, mg->s.n);
for (i = 0; i < mg->s.n; ++i)
{
if(s->a[i].s != 1 && s->a[i].s != 2) continue;
mg->s.a[i] = (s->a[i].s == 1? 1:-1);
}
return mg;
}
void debug_mc_gg_t(const char* fn, uint32_t update_ta, uint32_t convert_mc_g_t)
{
kv_u_trans_t *ta = NULL;
kv_gg_status *s = NULL;
uint32_t un;
uint16_t hapN;
if(load_mc_gg_dump(&ta, &un, &s, &hapN, fn))
{
if(convert_mc_g_t)
{
mc_opt_t opt;
mc_opt_init(&opt, asm_opt.n_perturb, asm_opt.f_perturb, asm_opt.seed);
mc_g_t *mg = NULL;
mg = to_mc_g_t(ta, s, un);
mc_solve_core(&opt, mg, NULL);
}
else
{
mc_solve_general(ta, un, s, hapN, update_ta, 0);
}
}
exit(1);
}
void clean_solve_general_ovlp(kv_u_trans_t *ta, uint32_t un, kv_gg_status *s)
{
uint32_t i;
for (i = 0; i < ta->n; i++) ta->a[i].del = !!(s->a[ta->a[i].qn].s&s->a[ta->a[i].tn].s);
kt_u_trans_t_simple_symm(ta, un, 0);
}
void mc_solve_general(kv_u_trans_t *ta, uint32_t un, kv_gg_status *s, uint16_t hapN, uint16_t update_ta, uint16_t write_dump)
{
mc_opt_t opt;
mc_opt_init(&opt, asm_opt.n_perturb, asm_opt.f_perturb, asm_opt.seed);
mc_gg_t *mg = init_mc_gg_t(un, s, hapN);
// print_mcb(mg);
update_mc_edges_general(mg, ta, hapN);
mc_solve_core_genral(&opt, mg, hapN);
if(update_ta) clean_solve_general_ovlp(ta, un, s);
// print_hap_p(s);
destory_mc_gg_t(&mg);
if(write_dump) write_mc_gg_dump(ta, un, s, hapN, MC_NAME);
}
+35
View File
@@ -15,6 +15,7 @@ typedef struct {
}mc_interval_t;
#define mc_node_t int8_t
#define mcg_node_t uint32_t
// #define w_t int64_t
// #define t_w_t int64_t
// #define w_cast(x) ((t_w_t)((x) < 0 ? (x) - 0.5 : (x) + 0.5))
@@ -72,6 +73,35 @@ typedef struct {
mb_match_t* e;
}mb_g_t;
typedef struct {
mcg_node_t s;
uint16_t h[2], hc;
double hw[2];
}mc_gg_status;
typedef struct {
mc_gg_status *a;
size_t n, m;
}kv_gg_status;
typedef struct {
mcg_node_t *a;
size_t n, m;
}mcb_t;
typedef struct {
kv_gg_status *s;
// ma_ug_t *ug;
// asg_t *rg;
uint32_t un;
mc_match_t* e;
kvec_t(mcb_t) m;
mcg_node_t mask;
uint16_t hN;
}mc_gg_t;
static inline uint64_t kr_splitmix64(uint64_t x)
{
uint64_t z = (x += 0x9E3779B97F4A7C15ULL);
@@ -90,4 +120,9 @@ static inline double kr_drand_r(uint64_t *x)
void mc_solve(hap_overlaps_list* ovlp, trans_chain* t_ch, kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g, double f_rate, uint8_t* trio_flag, uint32_t renew_s, int8_t *s, uint32_t is_sys, bubble_type* bub, kv_u_trans_t *ref);
void debug_mc_g_t(const char* name);
void mc_solve_general(kv_u_trans_t *ta, uint32_t un, kv_gg_status *s, uint16_t hapN, uint16_t update_ta, uint16_t write_dump);
kv_gg_status *init_mc_gg_status(ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut,
ma_hit_t_alloc* sources, R_to_U* ruIndex, uint64_t t_cov, uint16_t hapN);
void destory_mc_gg_t(mc_gg_t **p);
void debug_mc_gg_t(const char* fn, uint32_t update_ta, uint32_t convert_mc_g_t);
#endif
-21
View File
@@ -5,26 +5,6 @@
#include "kvec.h"
#include "htab.h"
typedef struct { // a simplified version of kdq
int front, count;
int a[64];
} tiny_queue_t;
static inline void tq_push(tiny_queue_t *q, int x)
{
q->a[((q->count++) + q->front) & 0x3f] = x;
}
static inline int tq_shift(tiny_queue_t *q)
{
int x;
if (q->count == 0) return -1;
x = q->a[q->front++];
q->front &= 0x3f;
--q->count;
return x;
}
/**
* Find symmetric (w,k)-minimizers on a DNA sequence
*
@@ -160,7 +140,6 @@ kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct)
memset(k_flag->a.a, 0, k_flag->a.n);
}
assert(len > 0 && len < 1<<27 && rid < 1<<28 && (w > 0 && w < 256) && (k > 0 && k <= 63));
///sizeof(ha_mz1_t) = 16
memset(buf, 0xff, w * 16);
+508 -2
View File
@@ -489,8 +489,8 @@ ma_ug_t *ug, uint32_t flag, double score, const char* cmd)
if(kh->tue <= kh->tus) continue;
kv_pushp(u_trans_t, o->k_trans, &kt);
kt->f = flag; kt->rev = ((kh->qn ^ kh->tn) & 1); kt->del = 0;
kt->qn = kh->qn>>1; kt->qs = kh->qus; kt->qe = kh->que; kt->qo = kh->qn&1;
kt->tn = kh->tn>>1; kt->ts = kh->tus; kt->te = kh->tue; kt->to = kh->tn&1;
kt->qn = kh->qn>>1; kt->qs = kh->qus; kt->qe = kh->que; ///kt->qo = kh->qn&1;
kt->tn = kh->tn>>1; kt->ts = kh->tus; kt->te = kh->tue; ///kt->to = kh->tn&1;
if(score < 0)
{
kt->nw = (MIN((kt->qe - kt->qs), (kt->te - kt->ts)))*CHAIN_MATCH;
@@ -1438,4 +1438,510 @@ R_to_U* ruIndex, utg_trans_t *o)
free(path_p);
free(path_q);
return n_reduced;
}
typedef struct {
uint64_t *idx;
kvec_t(uint64_t) pos;
} mz_ds_t;
typedef struct {
uint64_t x, y;
} pt128_t;
typedef struct {
uint64_t x;
uint64_t rid:32, span:32;
uint64_t pos:63, rev:1;
} pt_mz1_t;
typedef struct {
///cnt1: how many unique minimizers
///cnt2: how many non-unique minimizers
uint32_t cnt2, cnt1;
uint32_t m[2];
int8_t s;
} pt_uinfo_t;
typedef struct {
uint32_t n_seq; // number of segments; same as gfa_t::n_seg
pt_uinfo_t *info; // of size n_seg
kv_u_trans_t *ma;
} pt_match_t;
#define mz_key(z) ((z).x)
KRADIX_SORT_INIT(mz, pt_mz1_t, mz_key, 8)
#define pt128x_key(z) ((z).x)
KRADIX_SORT_INIT(pt128x, pt128_t, pt128x_key, 8)
#define generic_key(x) (x)
KRADIX_SORT_INIT(tb64, uint64_t, generic_key, 8)
typedef struct { uint32_t n, m; pt128_t *a; } pt128_v;
typedef struct { uint32_t n, m; pt_mz1_t *a; } pt_mz1_v;
static inline int mzcmp(const pt_mz1_t *a, const pt_mz1_t *b)
{
return (a->x > b->x) - (a->x < b->x);
}
void pt_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, pt_mz1_v *p)
{
static const pt_mz1_t dummy = { UINT64_MAX, (1<<28) - 1, 0, 0 };
uint64_t shift1 = k - 1, mask = (1ULL<<k) - 1, kmer[4] = {0,0,0,0};
int i, j, l, buf_pos, min_pos, kmer_span = 0;
pt_mz1_t buf[256], min = dummy;
tiny_queue_t tq;
assert(len > 0 && rid < (uint64_t)1<<32 && (w > 0 && w < 256) && (k > 0 && k <= 63));
memset(buf, 0xff, w * sizeof(pt_mz1_t));
memset(&tq, 0, sizeof(tiny_queue_t));
kv_resize(pt_mz1_t, *p, p->n + len/w);
for (i = l = buf_pos = min_pos = 0; i < len; ++i) {
int c = seq_nt4_table[(uint8_t)str[i]];
pt_mz1_t info = dummy;
if (c < 4) { // not an ambiguous base
int z;
if (is_hpc) {
int skip_len = 1;
if (i + 1 < len && seq_nt4_table[(uint8_t)str[i + 1]] == c) {
for (skip_len = 2; i + skip_len < len; ++skip_len)
if (seq_nt4_table[(uint8_t)str[i + skip_len]] != c)
break;
i += skip_len - 1; // put $i at the end of the current homopolymer run
}
tq_push(&tq, skip_len);
kmer_span += skip_len;
if (tq.count > k) kmer_span -= tq_shift(&tq);
} else kmer_span = l + 1 < k? l + 1 : k;
kmer[0] = (kmer[0] << 1 | (c&1)) & mask; // forward k-mer
kmer[1] = (kmer[1] << 1 | (c>>1)) & mask;
kmer[2] = kmer[2] >> 1 | (uint64_t)(1 - (c&1)) << shift1; // reverse k-mer
kmer[3] = kmer[3] >> 1 | (uint64_t)(1 - (c>>1)) << shift1;
if (kmer[1] == kmer[3]) continue; // skip "symmetric k-mers" as we don't know its strand
z = kmer[1] < kmer[3]? 0 : 1; // strand
++l;
if (l >= k && kmer_span < 256) {
uint64_t y;
y = yak_hash64_64(kmer[z<<1|0]) + yak_hash64_64(kmer[z<<1|1]);
info.x = y, info.rid = rid, info.pos = i, info.rev = z, info.span = kmer_span; // initially pt_mz1_t::rid keeps the k-mer count
}
} else l = 0, tq.count = tq.front = 0, kmer_span = 0;
buf[buf_pos] = info; // need to do this here as appropriate buf_pos and buf[buf_pos] are needed below
if (l == w + k - 1 && min.x != UINT64_MAX) { // special case for the first window - because identical k-mers are not stored yet
for (j = buf_pos + 1; j < w; ++j)
if (mzcmp(&min, &buf[j]) == 0 && buf[j].pos != min.pos) kv_push(pt_mz1_t, *p, buf[j]);
for (j = 0; j < buf_pos; ++j)
if (mzcmp(&min, &buf[j]) == 0 && buf[j].pos != min.pos) kv_push(pt_mz1_t, *p, buf[j]);
}
///three cases: 1.
if (info.x <= min.x) { // a new minimum; then write the old min
if (l >= w + k && min.x != UINT64_MAX) kv_push(pt_mz1_t, *p, min);
min = info, min_pos = buf_pos;
} else if (buf_pos == min_pos) { // old min has moved outside the window
if (l >= w + k - 1 && min.x != UINT64_MAX) kv_push(pt_mz1_t, *p, min);
for (j = buf_pos + 1, min.x = UINT64_MAX; j < w; ++j) // the two loops are necessary when there are identical k-mers
if (mzcmp(&min, &buf[j]) >= 0) min = buf[j], min_pos = j; // >= is important s.t. min is always the closest k-mer
for (j = 0; j <= buf_pos; ++j)
if (mzcmp(&min, &buf[j]) >= 0) min = buf[j], min_pos = j;
if (l >= w + k - 1 && min.x != UINT64_MAX) { // write identical k-mers
for (j = buf_pos + 1; j < w; ++j) // these two loops make sure the output is sorted
if (mzcmp(&min, &buf[j]) == 0 && min.pos != buf[j].pos) kv_push(pt_mz1_t, *p, buf[j]);
for (j = 0; j <= buf_pos; ++j)
if (mzcmp(&min, &buf[j]) == 0 && min.pos != buf[j].pos) kv_push(pt_mz1_t, *p, buf[j]);
}
}
if (++buf_pos == w) buf_pos = 0;
}
if (min.x != UINT64_MAX)
kv_push(pt_mz1_t, *p, min);
}
pt_mz1_v *pt_collect_minimizers(ma_ug_t *ug, asg_t *read_g, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources,
kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp)
{
uint32_t i;
pt_mz1_v *mz = NULL; CALLOC(mz, 1);
for (i = 0; i < ug->u.n; i++)
{
if(!ug->u.a[i].len) continue;
pt_sketch(ug->u.a[i].s, ug->u.a[i].len, asm_opt.mz_win, asm_opt.k_mer_length, i, 0, mz);
}
radix_sort_mz(mz->a, mz->a + mz->n);
return mz;
}
pt128_v *pt_collect_anchors(ma_ug_t *ug, pt_mz1_v *mz, mz_ds_t *mz_idx, uint32_t max_occ)
{
uint32_t st, j;
pt128_v *pa = NULL; CALLOC(pa, 1);
pt128_t *p = NULL;
mz_idx->pos.n = 0;
///all minimizers
for (j = 1, st = 0; j <= mz->n; ++j) {
if (j == mz->n || mz->a[j].x != mz->a[st].x) {
uint32_t k, l;
// if (j - st == 1) ++info[mz->a[st].rid].cnt1; ///of size n_seg
///max_occ is the frquency threshold of minimizer
///if (j - st) == 1, means minimizer only occurs in one read, it is not useful
if (j - st == 1 || j - st > max_occ) goto end_anchor;
for (k = st; k < j; ++k) {
// ++info[mz->a[k].rid].cnt2;
kv_push(uint64_t, mz_idx->pos, (uint64_t)(mz->a[k].rid)<<32|(uint64_t)(mz->a[k].pos));
for (l = k + 1; l < j; ++l) {
///k is current minimizer
uint32_t span, rev = (mz->a[k].rev != mz->a[l].rev);
int32_t lk = ug->u.a[mz->a[k].rid].len, ll = ug->u.a[mz->a[l].rid].len;
kv_pushp(pt128_t, *pa, &p);
span = mz->a[l].span;
p->x = (uint64_t)mz->a[k].rid << 33 | mz->a[l].rid << 1 | rev;
p->y = (uint64_t)mz->a[k].pos << 32 | (rev? ll - (mz->a[l].pos + 1 - span) - 1 : mz->a[l].pos);
kv_pushp(pt128_t, *pa, &p);
span = mz->a[k].span;
p->x = (uint64_t)mz->a[l].rid << 33 | mz->a[k].rid << 1 | rev;
p->y = (uint64_t)mz->a[l].pos << 32 | (rev? lk - (mz->a[k].pos + 1 - span) - 1 : mz->a[k].pos);
}
}
end_anchor: st = j;
}
}
radix_sort_pt128x(pa->a, pa->a + pa->n);
radix_sort_tb64(mz_idx->pos.a, mz_idx->pos.a + mz_idx->pos.n);
CALLOC(mz_idx->idx, ug->u.n);
for (st = 0, j = 1; j <= mz_idx->pos.n; ++j)
{
if (j == mz_idx->pos.n || (mz_idx->pos.a[j]>>32) != (mz_idx->pos.a[st]>>32))
{
mz_idx->idx[mz_idx->pos.a[st]>>32] = (uint64_t)st << 32 | (j - st), st = j;
}
}
return pa;
}
int32_t pt_lis_64(int32_t n, const uint64_t *a_idx, int32_t *b, int32_t *M)
{
int32_t i, k, L = 0, *P = b;
// MALLOC(M, n+1);
for (i = 0; i < n; ++i) {
int32_t lo = 1, hi = L, newL;
while (lo <= hi) {
int32_t mid = (lo + hi + 1) >> 1;
if ((uint32_t)a_idx[M[mid]] < (uint32_t)a_idx[i]) lo = mid + 1;
else hi = mid - 1;
}
newL = lo, P[i] = M[newL - 1], M[newL] = i;
if (newL > L) L = newL;
}
k = M[L];
memcpy(M, P, n * sizeof(int32_t));
for (i = L - 1; i >= 0; --i) b[i] = k, k = M[k];
// free(M);
return L;
}
uint32_t debug_lis_64(const uint64_t *a, int32_t *b, uint64_t n)
{
uint32_t i;
if(n <= 1) return 1;
for (i = 0; i+1 < n; i++)
{
if(((a[b[i]]>>32) > (a[b[i+1]]>>32)) || (((uint32_t)a[b[i]]) > ((uint32_t)a[b[i+1]])))
{
i = (uint32_t)-1;
break;
}
}
if(i == (uint32_t)-1)
{
fprintf(stderr, "\nERROR-chain\n");
for (i = 0; i < n; i++)
{
fprintf(stderr, "x-%lu, y-%lu\n", (a[b[i]]>>32), (uint64_t)((uint32_t)a[b[i]]));
}
}
return 1;
}
void update_mz_ovlp(uint32_t* n_x_beg, uint32_t* n_x_end, int64_t xLen,
uint32_t* n_y_beg, uint32_t* n_y_end, int64_t yLen, uint32_t rev)
{
int64_t x_beg = (*n_x_beg), x_end = (*n_x_end);
int64_t y_beg = (*n_y_beg), y_end = (*n_y_end);
if(x_beg <= y_beg)
{
y_beg = y_beg - x_beg;
x_beg = 0;
}
else
{
x_beg = x_beg - y_beg;
y_beg = 0;
}
long long x_right_length = xLen - x_end - 1;
long long y_right_length = yLen - y_end - 1;
if(x_right_length <= y_right_length)
{
x_end = xLen - 1;
y_end = y_end + x_right_length;
}
else
{
x_end = x_end + y_right_length;
y_end = yLen - 1;
}
if(rev == 0)
{
(*n_y_beg) = y_beg;
(*n_y_end) = y_end + 1;
}
else
{
(*n_y_beg) = yLen - y_end - 1;
(*n_y_end) = yLen - y_beg - 1 + 1;
}
(*n_x_beg) = x_beg;
(*n_x_end) = x_end + 1;
}
int64_t get_insert_pos(uint64_t *a, int64_t n, int64_t target)
{
int64_t left, right, ans, mid;
left = 0; right = n - 1; ans = n;
while (left <= right) {
mid = ((right - left) >> 1) + left;
if (target <= (uint32_t)a[mid]) {
ans = mid;
right = mid - 1;
} else {
left = mid + 1;
}
}
return ans;
}
uint32_t get_mz_occ(mz_ds_t *mz_idx, uint64_t uid, uint64_t s, uint64_t e)
{
uint64_t *a = mz_idx->pos.a + (mz_idx->idx[uid]>>32), n = (uint32_t)(mz_idx->idx[uid]);
if(n == 0) return 0;
int64_t sid, eid;
sid = get_insert_pos(a, n, s);
eid = get_insert_pos(a, n, e);
return eid + 1 - sid;
}
kv_u_trans_t *pt_cal_sim(pt128_v *pa, mz_ds_t *mz_idx, ma_ug_t *ug, uint32_t min_cnt, double min_sim)
{
int64_t st, i, j;
kvec_t(uint64_t) a; kv_init(a);
kvec_t(int32_t) b; kv_init(b);
kvec_t(int32_t) M; kv_init(M);
kv_u_trans_t *ma = NULL; CALLOC(ma, 1);
u_trans_t m;
///x = (uint64_t)mz[l].rid << 33 | mz[k].rid << 1 | rev;
for (st = 0, i = 1; i <= pa->n; ++i) {
if (i == pa->n || pa->a[i].x != pa->a[st].x) {///minimizers between a pair of unitigs
if((pa->a[st].x>>33) == (((uint32_t)pa->a[st].x)>>1)) goto end_chain;
uint32_t nn[2], nn_min;
a.n = 0; memset(&m, 0, sizeof(m));
if (i - st < min_cnt) goto end_chain;
//(uint64_t)mz[l].pos << 32 | (rev? lk - (mz[k].pos + 1 - span) - 1 : mz[k].pos);
for (j = st; j < i; ++j) kv_push(uint64_t, a, pa->a[j].y);
radix_sort_tb64(a.a, a.a + a.n);///sort by query pos + target pos
// for (l = 0; l < a.n; ++l) a.a[l] = (uint32_t)a.a[l];///only need target pos to do LIS
kv_resize(int32_t, b, a.n); kv_resize(int32_t, M, a.n+1);
m.occ = pt_lis_64(a.n, a.a, b.a, M.a);
/*******************************for debug************************************/
// debug_lis_64(a.a, b.a, m.occ);
/*******************************for debug************************************/
if (m.occ == 0 || m.occ < min_cnt) goto end_chain;//chain occ
m.qn = pa->a[st].x >> 33;///query id
m.tn = ((uint32_t)pa->a[st].x) >> 1;///target id
if (m.qn == m.tn) goto end_chain;
m.qs = a.a[b.a[0]]>>32; m.qe = a.a[b.a[m.occ-1]]>>32;
m.ts = (uint32_t)(a.a[b.a[0]]); m.te = (uint32_t)(a.a[b.a[m.occ-1]]);
m.rev = (pa->a[st].x>>32&1) ^ (pa->a[st].x&1);
update_mz_ovlp(&(m.qs), &(m.qe), ug->u.a[m.qn].len, &(m.ts), &(m.te), ug->u.a[m.tn].len, m.rev);
nn[0] = get_mz_occ(mz_idx, m.qn, m.qs, m.qe-1);
nn[1] = get_mz_occ(mz_idx, m.tn, m.ts, m.te-1);
nn_min = MIN(nn[0], nn[1]);
nn_min = MAX(nn_min, m.occ);
// m.sim = pow(2.0 * m.m / (nn[0] + nn[1]), 1.0 / k);
if(m.occ >= nn_min*min_sim){
m.nw = (double)(m.occ) - (double)(nn_min-m.occ)*0.2;
if(m.nw > 0) kv_push(u_trans_t, *ma, m);
}
end_chain: st = i;
}
}
kv_destroy(b); kv_destroy(a); kv_destroy(M);
return ma;
}
void clean_mz_ovlp(kv_u_trans_t *ta, ma_ug_t *ug)
{
u_trans_t *a = NULL;
asg_arc_t *as = NULL;
uint32_t k, i, n, v, ns;
pdq pq; init_pdq(&pq, ug->g->n_seq<<1);
uint8_t *vis = NULL; CALLOC(vis, ug->g->n_seq);
kvec_t_u32_warp p; kv_init(p.a);
for (k = 0; k < ta->idx.n; k++)
{
p.a.n = 0;
a = u_trans_a(*ta, k);
n = u_trans_n(*ta, k);
if(n == 0) continue;
v = k<<1;
as = asg_arc_a(ug->g, v); ns = asg_arc_n(ug->g, v);
if(ns > 0)
{
for (i = 0; i < ns; i++)
{
if(as[i].del) continue;
kv_push(uint32_t, p.a, as[i].v);
}
set_utg_by_dis(v, &pq, ug->g, &p, ug->g->seq[v>>1].len);
}
v = (k<<1) + 1;
as = asg_arc_a(ug->g, v); ns = asg_arc_n(ug->g, v);
if(ns > 0)
{
for (i = 0; i < ns; i++)
{
if(as[i].del) continue;
kv_push(uint32_t, p.a, as[i].v);
}
set_utg_by_dis(v, &pq, ug->g, &p, ug->g->seq[v>>1].len);
}
for (i = 0; i < p.a.n; i++) vis[p.a.a[i]>>1] = 1;
for (i = 0; i < n; i++)
{
if(vis[a[i].tn]) a[i].del = 1;
}
for (i = 0; i < p.a.n; i++) vis[p.a.a[i]>>1] = 0;
}
destory_pdq(&pq);
kv_destroy(p.a);
free(vis);
for (k = n = 0; k < ta->n; ++k)
{
if(ta->a[k].del) continue;
ta->a[n] = ta->a[k];
n++;
}
ta->n = n;
kt_u_trans_t_idx(ta, ug->g->n_seq);
kt_u_trans_t_simple_symm(ta, ug->g->n_seq, 0);
}
int cmp_u_trans_nw(const void * a, const void * b)
{
if((*(u_trans_t*)a).nw == (*(u_trans_t*)b).nw) return 0;
return (*(u_trans_t*)a).nw < (*(u_trans_t*)b).nw ? 1 : -1;
}
/**
void flat_mz_ovlp(kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut,
ma_hit_t_alloc* sources, R_to_U* ruIndex, uint32_t min_cnt, uint32_t cov_thres,
double cov_pass_rate, double nw_pass_rate)
{
u_trans_t *a = NULL, *p = NULL;
uint8_t *vis = NULL; CALLOC(vis, ug->g->n_seq);
uint32_t *cov = NULL; CALLOC(cov, ug->g->n_seq);
kvec_t(uint32_t) tc; kv_init(tc);
uint32_t k, i, j, n, v, ns, qs, qe, occ, hapN, pass;
double w;
for (i = 0; i < ug->u.n; i++)
{
cov[i] = get_utg_cov(ug, i, read_g, coverage_cut, sources, ruIndex, vis);
}
for (k = 0; k < ta->idx.n; k++)
{
a = u_trans_a(*ta, k);
n = u_trans_n(*ta, k);
if(n == 0) continue;
qsort(a, n, sizeof(u_trans_t), cmp_u_trans_nw);
kv_resize(uint32_t, tc, ug->u.a[k].len);
tc.n = ug->u.a[k].len;
memset(tc.a, 0, tc.n*sizeof(uint32_t));
for (i = 0, p = NULL; i < n; i++)
{
qs = a[i].qs; qe = a[i].qe; w = a[i].nw; occ = a[i].occ; pass = 0;
for (j = qs; j < qe; j++)
{
if(tc.a[j] + cov[k] >= cov_thres) pass++;
}
if(pass > cov_pass_rate*(qe - qs))
{
if(p && nw_pass_rate*) a[i].del = 1;
}
else
{
for (j = qs; j < qe; j++) tc.a[j] += cov[k];
}
}
}
free(vis); free(cov); kv_destroy(tc);
}
**/
void print_u_trans(kv_u_trans_t *ta)
{
uint32_t i;
u_trans_t *p = NULL;
for (i = 0; i < ta->n; i++)
{
p = &(ta->a[i]);
fprintf(stderr, "q-utg%.6ul\tqs(%u)\tqe(%u)\tt-utg%.6ul\tts(%u)\tte(%u)\trev(%u)\tw(%f)\tf(%u)\n",
p->qn+1, p->qs, p->qe, p->tn+1, p->ts, p->te, p->rev, p->nw, p->f);
}
fprintf(stderr, "[M::%s::] \n", __func__);
}
kv_u_trans_t *pt_pdist(ma_ug_t *ug, asg_t *read_g, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources,
kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp, uint32_t min_chain_cnt)
{
pt_mz1_v *mz = NULL;
mz_ds_t mz_idx; memset(&mz_idx, 0, sizeof(mz_idx));
pt128_v *an = NULL;
kv_u_trans_t *ma = NULL; CALLOC(ma, 1);
mz = pt_collect_minimizers(ug, read_g, coverage_cut, sources, edge, max_hang, min_ovlp);
an = pt_collect_anchors(ug, mz, &mz_idx, asm_opt.polyploidy*10);
kv_destroy(*mz); free(mz);
ma = pt_cal_sim(an, &mz_idx, ug, min_chain_cnt, asm_opt.purge_simi_thres);
kv_destroy(*an); free(an);
kv_destroy(mz_idx.pos); free(mz_idx.idx);
kt_u_trans_t_idx(ma, ug->g->n_seq);
kt_u_trans_t_simple_symm(ma, ug->g->n_seq, 1);
clean_mz_ovlp(ma, ug);
// print_u_trans(ma);
// exit(1);
return ma;
}
+2
View File
@@ -12,4 +12,6 @@ int asg_arc_decompress(asg_t *g, ma_ug_t *ug, asg_t *read_sg, ma_hit_t_alloc* re
R_to_U* ruIndex, utg_trans_t *o);
int asg_arc_decompress_mul(asg_t *g, ma_ug_t *ug, asg_t *read_sg, uint32_t positive_flag, uint32_t negative_flag,
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, utg_trans_t *o);
kv_u_trans_t *pt_pdist(ma_ug_t *ug, asg_t *read_g, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources,
kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp, uint32_t min_chain_cnt);
#endif