mirror of
https://github.com/chhylp123/hifiasm.git
synced 2026-10-12 07:10:57 +08:00
Compare commits
96
Commits
0.18.7
...
f5078f7b23
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f5078f7b23 | ||
|
|
6b026f0caa | ||
|
|
ba8627fb23 | ||
|
|
815d709fb9 | ||
|
|
a137a6b06f | ||
|
|
e53fc786bd | ||
|
|
aa4d12b1c6 | ||
|
|
9a2884f518 | ||
|
|
5e5a1568ed | ||
|
|
c0478830e6 | ||
|
|
2fa4ef224f | ||
|
|
86311effb9 | ||
|
|
53e1b150c3 | ||
|
|
67877deed3 | ||
|
|
d6ba102d42 | ||
|
|
ec9a8b222d | ||
|
|
b3b18ab1d0 | ||
|
|
a96191560e | ||
|
|
79387d081f | ||
|
|
4733ef5011 | ||
|
|
2c77b3c87e | ||
|
|
3067771783 | ||
|
|
4889f1c6d8 | ||
|
|
dbdef7ff63 | ||
|
|
184eb0a9fe | ||
|
|
676385cf8e | ||
|
|
80fa5ed436 | ||
|
|
6de4e14782 | ||
|
|
ade800990e | ||
|
|
fc98214321 | ||
|
|
382adb89c5 | ||
|
|
6e33d972d0 | ||
|
|
39a30a8d55 | ||
|
|
b01614fd71 | ||
|
|
63d51c3441 | ||
|
|
49dd83df2a | ||
|
|
e8b18560a5 | ||
|
|
d5f8a8a6c0 | ||
|
|
70fd9a0b1f | ||
|
|
73dd5eeb5b | ||
|
|
358b090e20 | ||
|
|
1ac574adc7 | ||
|
|
61ceb3a4e1 | ||
|
|
7f6d36b2c4 | ||
|
|
4ca8951f9d | ||
|
|
40e2a48706 | ||
|
|
281ae6ba89 | ||
|
|
e974b22ac0 | ||
|
|
c7685e6f19 | ||
|
|
94a284b430 | ||
|
|
858b95a3bb | ||
|
|
30d2ee065e | ||
|
|
64091a76a5 | ||
|
|
d91fc50058 | ||
|
|
adeb2c603d | ||
|
|
abff5fae04 | ||
|
|
44475e2665 | ||
|
|
2cc990ed22 | ||
|
|
2c18ec30da | ||
|
|
5bddae4ae9 | ||
|
|
18b649aadb | ||
|
|
fbfcf72eb3 | ||
|
|
0efcc14499 | ||
|
|
84e642d20d | ||
|
|
b49b4a6cc7 | ||
|
|
7e855ba5e2 | ||
|
|
5df5892f33 | ||
|
|
00959dfdd9 | ||
|
|
503cd0f7cc | ||
|
|
00ad7458c3 | ||
|
|
86b7dd424a | ||
|
|
b763e1ff76 | ||
|
|
d4773a497f | ||
|
|
7bb366688e | ||
|
|
f67faa9c34 | ||
|
|
3407c7029e | ||
|
|
f69166ee3b | ||
|
|
2218bac8aa | ||
|
|
b533067835 | ||
|
|
fa663a9680 | ||
|
|
384b721d84 | ||
|
|
8b79a3f22e | ||
|
|
af4be8ca1a | ||
|
|
85c643e85d | ||
|
|
8a00c6c5f4 | ||
|
|
a0f2f70d91 | ||
|
|
ad4ff5550b | ||
|
|
04f660fd8c | ||
|
|
dd5f368fe6 | ||
|
|
2b4b4f29e0 | ||
|
|
8ea2d5a622 | ||
|
|
27f77dca29 | ||
|
|
7d5c793109 | ||
|
|
3abd3793fb | ||
|
|
b55e4396d2 | ||
|
|
fc598d71f6 |
+439
-26
@@ -12,6 +12,7 @@
|
||||
#include "kthread.h"
|
||||
#include "rcut.h"
|
||||
#include "kalloc.h"
|
||||
#include "ecovlp.h"
|
||||
|
||||
void ha_get_candidates_interface(ha_abuf_t *ab, int64_t rid, UC_Read *ucr, overlap_region_alloc *overlap_list, overlap_region_alloc *overlap_list_hp, Candidates_list *cl, double bw_thres,
|
||||
int max_n_chain, int keep_whole_chain, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* chain_idx, ma_hit_t_alloc* paf, ma_hit_t_alloc* rev_paf, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct, st_mt_t *sp);
|
||||
@@ -23,6 +24,7 @@ All_reads R_INF;
|
||||
Debug_reads R_INF_FLAG;
|
||||
all_ul_t UL_INF, ULG_INF;
|
||||
uint32_t *het_cnt = NULL;
|
||||
// uint32_t debug_out = 0;
|
||||
|
||||
void get_corrected_read_from_cigar(Cigar_record* cigar, char* pre_read, int pre_length, char* new_read, int* new_length)
|
||||
{
|
||||
@@ -594,9 +596,9 @@ static void worker_ovec(void *data, long i, int tid)
|
||||
{
|
||||
ha_ovec_buf_t *b = ((ha_ovec_buf_t**)data)[tid];
|
||||
int fully_cov, abnormal;
|
||||
// if(i != 33) return;
|
||||
// if(i != 12578) return;
|
||||
// fprintf(stderr, "[M::%s-beg] rid->%ld\n", __func__, i);
|
||||
// if (memcmp("m64012_190920_173625/88015004/ccs", Get_NAME((R_INF), i), Get_NAME_LENGTH((R_INF),i)) == 0) {
|
||||
// if (memcmp("7897e875-76e5-42c8-bc37-94b370c4cc8d", Get_NAME((R_INF), i), Get_NAME_LENGTH((R_INF),i)) == 0) {
|
||||
// fprintf(stderr, "[M::%s-beg] rid->%ld\n", __func__, i);
|
||||
// } else {
|
||||
// return;
|
||||
@@ -605,6 +607,9 @@ static void worker_ovec(void *data, long i, int tid)
|
||||
ha_get_candidates_interface(b->ab, i, &b->self_read, &b->olist, &b->olist_hp, &b->clist,
|
||||
0.02, asm_opt.max_n_chain, 1, NULL/**&(b->k_flag)**/, &b->r_buf, &(R_INF.paf[i]), &(R_INF.reverse_paf[i]), &(b->tmp_region), NULL, &(b->sp));
|
||||
|
||||
// prt_chain(&b->olist);
|
||||
// return;
|
||||
|
||||
clear_Cigar_record(&b->cigar1);
|
||||
clear_Round2_alignment(&b->round2);
|
||||
|
||||
@@ -644,6 +649,44 @@ static void worker_ovec(void *data, long i, int tid)
|
||||
}
|
||||
|
||||
|
||||
static void worker_ovec_cal0(void *data, long i, int tid)
|
||||
{
|
||||
ha_ovec_buf_t *b = ((ha_ovec_buf_t**)data)[tid];
|
||||
int fully_cov, abnormal;
|
||||
|
||||
ha_get_candidates_interface(b->ab, i, &b->self_read, &b->olist, &b->olist_hp, &b->clist,
|
||||
0.02, asm_opt.max_n_chain, 1, NULL/**&(b->k_flag)**/, &b->r_buf, &(R_INF.paf[i]), &(R_INF.reverse_paf[i]), &(b->tmp_region), NULL, &(b->sp));
|
||||
|
||||
clear_Cigar_record(&b->cigar1);
|
||||
clear_Round2_alignment(&b->round2);
|
||||
|
||||
correct_overlap(&b->olist, &R_INF, &b->self_read, &b->correct, &b->ovlp_read, &b->POA_Graph, &b->DAGCon,
|
||||
&b->cigar1, &b->hap, &b->round2, &b->r_buf, &(b->tmp_region.w_list), 0, 0/**1***/, &fully_cov, &abnormal);
|
||||
|
||||
b->num_read_base += b->self_read.length;
|
||||
b->num_correct_base += b->correct.corrected_base;
|
||||
b->num_recorrect_base += b->round2.dumy.corrected_base;
|
||||
|
||||
|
||||
// R_INF.paf[i].is_fully_corrected = 0;
|
||||
// R_INF.paf[i].is_abnormal = abnormal;
|
||||
R_INF.trio_flag[i] = AMBIGU;
|
||||
|
||||
// R_INF.paf[i].is_fully_corrected = 0;
|
||||
// if (fully_cov) {
|
||||
// if (get_cigar_errors(&b->cigar1) == 0 && get_cigar_errors(&b->round2.cigar) == 0)
|
||||
// R_INF.paf[i].is_fully_corrected = 1;
|
||||
// }
|
||||
// R_INF.paf[i].is_abnormal = abnormal;
|
||||
// R_INF.trio_flag[i] = AMBIGU;
|
||||
|
||||
push_final_overlaps(&(R_INF.paf[i]), R_INF.reverse_paf, &b->olist, 1);
|
||||
push_final_overlaps(&(R_INF.reverse_paf[i]), R_INF.reverse_paf, &b->olist, 2);
|
||||
|
||||
if(het_cnt) het_cnt[i] = get_het_cnt(&b->hap);
|
||||
}
|
||||
|
||||
|
||||
static void worker_ovec_related_reads(void *data, long i, int tid)
|
||||
{
|
||||
ha_ovec_buf_t *b = ((ha_ovec_buf_t**)data)[tid];
|
||||
@@ -661,6 +704,7 @@ static void worker_ovec_related_reads(void *data, long i, int tid)
|
||||
|
||||
if(k < R_INF_FLAG.query_num)
|
||||
{
|
||||
R_INF_FLAG.read_id[k] = i;
|
||||
int fully_cov, abnormal, q_idx = k;
|
||||
|
||||
ha_get_candidates_interface(b->ab, i, &b->self_read, &b->olist, &b->olist_hp, &b->clist,
|
||||
@@ -727,6 +771,7 @@ static void worker_ovec_related_reads(void *data, long i, int tid)
|
||||
Get_READ_LENGTH(R_INF, b->olist.list[k].y_id));
|
||||
}
|
||||
|
||||
/**
|
||||
fprintf(R_INF_FLAG.fp, "***************************unmatched ovlp***************************\n");
|
||||
for (k = 0; k < b->olist.length; k++)
|
||||
{
|
||||
@@ -738,6 +783,7 @@ static void worker_ovec_related_reads(void *data, long i, int tid)
|
||||
b->olist.list[k].y_pos_strand, b->olist.list[k].strong, b->olist.list[k].without_large_indel,
|
||||
Get_READ_LENGTH(R_INF, b->olist.list[k].y_id));
|
||||
}
|
||||
**/
|
||||
|
||||
R_INF.trio_flag[i] = AMBIGU;
|
||||
|
||||
@@ -837,16 +883,14 @@ static void worker_ec_save(void *data, long i, int tid)
|
||||
|
||||
void Output_corrected_reads()
|
||||
{
|
||||
long long i;
|
||||
UC_Read g_read;
|
||||
uint64_t i; UC_Read g_read;
|
||||
init_UC_Read(&g_read);
|
||||
char* gfa_name = (char*)malloc(strlen(asm_opt.output_file_name)+35);
|
||||
sprintf(gfa_name, "%s.ec.fa", asm_opt.output_file_name);
|
||||
FILE *output_file = fopen(gfa_name, "w");
|
||||
free(gfa_name);
|
||||
|
||||
for (i = 0; i < (long long)R_INF.total_reads; i++)
|
||||
{
|
||||
for (i = 0; i < R_INF.total_reads; i++) {
|
||||
recover_UC_Read(&g_read, &R_INF, i);
|
||||
fwrite(">", 1, 1, output_file);
|
||||
fwrite(Get_NAME(R_INF, i), 1, Get_NAME_LENGTH(R_INF, i), output_file);
|
||||
@@ -858,6 +902,40 @@ void Output_corrected_reads()
|
||||
fclose(output_file);
|
||||
}
|
||||
|
||||
void Output_corrected_fastq()
|
||||
{
|
||||
uint64_t i, k;
|
||||
UC_Read g_read; asg8_v dv;
|
||||
init_UC_Read(&g_read); kv_init(dv);
|
||||
char* gfa_name = (char*)malloc(strlen(asm_opt.output_file_name)+35);
|
||||
sprintf(gfa_name, "%s.ec.fq", asm_opt.output_file_name);
|
||||
FILE* fp = fopen(gfa_name, "w");
|
||||
free(gfa_name);
|
||||
|
||||
for (i = 0; i < R_INF.tqn; i++) {
|
||||
recover_UC_Read(&g_read, &R_INF, i);
|
||||
fprintf(fp, "@%.*s\n", (int32_t)Get_NAME_LENGTH(R_INF, i), Get_NAME(R_INF, i));
|
||||
fprintf(fp, "%.*s\n", (int32_t)g_read.length, g_read.seq);
|
||||
fprintf(fp, "+\n");
|
||||
retrive_bqual(&dv, NULL, i, -1, -1, 0, sc_bn);
|
||||
for (k = 0; k < dv.n; k++) fprintf(fp, "%c", (char)(sc_tb[dv.a[k]] + 33 - 1));
|
||||
fprintf(fp, "\n");
|
||||
}
|
||||
|
||||
for (; i < R_INF.total_reads; i++) {
|
||||
recover_UC_Read(&g_read, &R_INF, i);
|
||||
fprintf(fp, "@%.*s\n", (int32_t)Get_NAME_LENGTH(R_INF, i), Get_NAME(R_INF, i));
|
||||
fprintf(fp, "%.*s\n", (int32_t)g_read.length, g_read.seq);
|
||||
fprintf(fp, "+\n");
|
||||
// retrive_bqual(&dv, NULL, i, -1, -1, 0, sc_bn);
|
||||
// for (k = 0; k < dv.n; k++) fprintf(fp, "%c", (char)(sc_tb[dv.a[k]] + 33 - 1));
|
||||
for (k = 0; k < (uint64_t)g_read.length; k++) fprintf(fp, "%c", (char)(3 + 33 - 1));
|
||||
fprintf(fp, "\n");
|
||||
}
|
||||
destory_UC_Read(&g_read); kv_destroy(dv);
|
||||
fclose(fp);
|
||||
}
|
||||
|
||||
void debug_print_pob_regions()
|
||||
{
|
||||
uint64_t i, total = 0;
|
||||
@@ -878,7 +956,7 @@ void rescue_hp_reads(ha_ovec_buf_t **b)
|
||||
int hom_cov, het_cov;
|
||||
ha_flt_tab_hp = ha_idx_hp = NULL;
|
||||
if (!(asm_opt.flag & HA_F_NO_KMER_FLT)) {
|
||||
ha_flt_tab_hp = ha_ft_gen(&asm_opt, &R_INF, &hom_cov, 1);
|
||||
ha_flt_tab_hp = ha_ft_gen(&asm_opt, &R_INF, &hom_cov, 1, 0);
|
||||
}
|
||||
ha_idx_hp = ha_pt_gen(&asm_opt, ha_flt_tab, 1, 1, &R_INF, &hom_cov, &het_cov);
|
||||
|
||||
@@ -911,6 +989,94 @@ void print_het_cnt_log(uint32_t *het_cnt)
|
||||
fclose(output_file);
|
||||
}
|
||||
|
||||
void prt_dbg_rs(FILE *fp, Debug_reads* x, uint64_t round)
|
||||
{
|
||||
uint64_t k, id; UC_Read g_read; init_UC_Read(&g_read);
|
||||
for (k = 0; k < R_INF_FLAG.query_num; k++) {
|
||||
id = x->read_id[k];
|
||||
if(id == ((uint64_t)-1)) continue;
|
||||
recover_UC_Read(&g_read, &R_INF, id);
|
||||
fprintf(fp, ">%.*s_r%lu\n", (int)Get_NAME_LENGTH((R_INF), id), Get_NAME((R_INF), id), round);
|
||||
fprintf(fp, "%.*s\n", (int)g_read.length, g_read.seq);
|
||||
}
|
||||
destory_UC_Read(&g_read);
|
||||
}
|
||||
|
||||
void ha_ec(int64_t round, int num_pround, int des_idx, uint64_t *tot_b, uint64_t *tot_e, uint64_t w_tmp)
|
||||
{
|
||||
int hom_cov, het_cov, r_out = 0;
|
||||
ha_flt_tab_hp = ha_idx_hp = NULL; (*tot_b) = (*tot_e) = 0;
|
||||
|
||||
if((ha_idx == NULL)&&(asm_opt.flag & HA_F_VERBOSE_GFA)&&(round == asm_opt.number_of_round - 1)) r_out = 1;
|
||||
|
||||
if(asm_opt.required_read_name) init_Debug_reads(&R_INF_FLAG, asm_opt.required_read_name); // for debugging only
|
||||
|
||||
if(ha_idx) hom_cov = asm_opt.hom_cov;
|
||||
if(ha_idx == NULL) {
|
||||
ha_idx = ha_pt_gen(&asm_opt, ha_flt_tab, round == 0? 0 : 1, 0, &R_INF, &hom_cov, &het_cov); // build the index
|
||||
asm_opt.hom_cov = hom_cov; asm_opt.het_cov = het_cov;
|
||||
}
|
||||
///debug_adapter(&asm_opt, &R_INF);
|
||||
if (round == 0 && ha_flt_tab == 0) // then asm_opt.hom_cov hasn't been updated
|
||||
ha_opt_update_cov(&asm_opt, hom_cov);
|
||||
het_cnt = NULL;
|
||||
if(round == asm_opt.number_of_round-1 && asm_opt.is_dbg_het_cnt) CALLOC(het_cnt, R_INF.total_reads);
|
||||
|
||||
if (r_out) {
|
||||
write_pt_index(ha_flt_tab, ha_idx, &R_INF, &asm_opt, asm_opt.output_file_name);
|
||||
if((asm_opt.flag & HA_F_VERBOSE_GFA) && (asm_opt.bin_only == 1)) exit(1);///just for debug
|
||||
}
|
||||
if (w_tmp) tmp_pt_pro(&ha_flt_tab, &ha_idx, &R_INF, &asm_opt, asm_opt.output_file_name, round, asm_opt.number_of_round, 0);
|
||||
|
||||
// Output_corrected_fastq();
|
||||
|
||||
|
||||
cal_ec_r(asm_opt.thread_num, round, num_pround, R_INF.total_reads, (round == (asm_opt.number_of_round-1))?1:0, tot_b, tot_e);
|
||||
|
||||
// exit(1);
|
||||
|
||||
// if (r_out) write_pt_index(ha_flt_tab, ha_idx, &R_INF, &asm_opt, asm_opt.output_file_name);
|
||||
if(des_idx) {
|
||||
ha_pt_destroy(ha_idx); ha_idx = NULL;
|
||||
}
|
||||
|
||||
|
||||
if(het_cnt) {
|
||||
print_het_cnt_log(het_cnt); free(het_cnt); het_cnt = NULL;
|
||||
}
|
||||
|
||||
// exit(1);
|
||||
|
||||
|
||||
if (asm_opt.required_read_name) prt_dbg_rs(R_INF_FLAG.fp_r0, &R_INF_FLAG, 0); // for debugging only
|
||||
|
||||
// save corrected reads to R_INF
|
||||
// sl_ec_r(asm_opt.thread_num, R_INF.total_reads);
|
||||
|
||||
if (asm_opt.required_read_name) prt_dbg_rs(R_INF_FLAG.fp_r1, &R_INF_FLAG, 1); // for debugging only
|
||||
if (asm_opt.required_read_name) destory_Debug_reads(&R_INF_FLAG), exit(0); // for debugging only
|
||||
///debug_print_pob_regions();
|
||||
|
||||
// Output_corrected_reads();
|
||||
|
||||
// exit(1);
|
||||
}
|
||||
|
||||
|
||||
int ha_ec_dbg(void)
|
||||
{
|
||||
int hom_cov, het_cov;
|
||||
|
||||
ha_idx = ha_pt_gen(&asm_opt, 0, 0, 0, &R_INF, &hom_cov, &het_cov); // build the index
|
||||
asm_opt.hom_cov = hom_cov; asm_opt.het_cov = het_cov;
|
||||
ha_opt_update_cov(&asm_opt, hom_cov);
|
||||
|
||||
cal_ec_r_dbg(asm_opt.thread_num, R_INF.total_reads);
|
||||
|
||||
ha_pt_destroy(ha_idx); ha_idx = NULL;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
void ha_overlap_and_correct(int round)
|
||||
{
|
||||
@@ -937,11 +1103,14 @@ void ha_overlap_and_correct(int round)
|
||||
het_cnt = NULL;
|
||||
if(round == asm_opt.number_of_round-1 && asm_opt.is_dbg_het_cnt) CALLOC(het_cnt, R_INF.total_reads);
|
||||
// fprintf(stderr, "[M::%s-start]\n", __func__);
|
||||
// double tt0 = yak_realtime_0();
|
||||
if (asm_opt.required_read_name)
|
||||
kt_for(asm_opt.thread_num, worker_ovec_related_reads, b, R_INF.total_reads);
|
||||
else
|
||||
kt_for(asm_opt.thread_num, worker_ovec, b, R_INF.total_reads);///debug_for_fix
|
||||
// fprintf(stderr, "[M::%s-end]\n", __func__);
|
||||
// fprintf(stderr, "[M::%s::%.3f] ==> chaining\n", __func__, yak_realtime_0()-tt0);
|
||||
// exit(1);
|
||||
|
||||
if (r_out) write_pt_index(ha_flt_tab, ha_idx, &R_INF, &asm_opt, asm_opt.output_file_name);
|
||||
ha_pt_destroy(ha_idx);
|
||||
@@ -960,8 +1129,7 @@ void ha_overlap_and_correct(int round)
|
||||
ha_ovec_destroy(b[i]);
|
||||
}
|
||||
free(b);
|
||||
|
||||
if (asm_opt.required_read_name) destory_Debug_reads(&R_INF_FLAG), exit(0); // for debugging only
|
||||
if (asm_opt.required_read_name) prt_dbg_rs(R_INF_FLAG.fp_r0, &R_INF_FLAG, 0); // for debugging only
|
||||
|
||||
// save corrected reads to R_INF
|
||||
CALLOC(e, asm_opt.thread_num);
|
||||
@@ -978,9 +1146,53 @@ void ha_overlap_and_correct(int round)
|
||||
free(e[i].second_round_read);
|
||||
}
|
||||
free(e);
|
||||
|
||||
if (asm_opt.required_read_name) prt_dbg_rs(R_INF_FLAG.fp_r1, &R_INF_FLAG, 1); // for debugging only
|
||||
if (asm_opt.required_read_name) destory_Debug_reads(&R_INF_FLAG), exit(0); // for debugging only
|
||||
///debug_print_pob_regions();
|
||||
}
|
||||
|
||||
void ha_overlap_cal(int round, int read_from_store)
|
||||
{
|
||||
int i, hom_cov, het_cov;
|
||||
ha_ovec_buf_t **b;
|
||||
ha_flt_tab_hp = ha_idx_hp = NULL;
|
||||
|
||||
// overlap and correct reads
|
||||
CALLOC(b, asm_opt.thread_num);
|
||||
for (i = 0; i < asm_opt.thread_num; ++i)
|
||||
b[i] = ha_ovec_init(0, (round == asm_opt.number_of_round - 1),0);
|
||||
if(ha_idx) hom_cov = asm_opt.hom_cov;
|
||||
if(ha_idx == NULL) ha_idx = ha_pt_gen(&asm_opt, ha_flt_tab, ((round == 0)&&(read_from_store == 0))?0:1, 0, &R_INF, &hom_cov, &het_cov); // build the index
|
||||
///debug_adapter(&asm_opt, &R_INF);
|
||||
if (/**round == 0 &&**/ ha_flt_tab == 0) // then asm_opt.hom_cov hasn't been updated
|
||||
ha_opt_update_cov(&asm_opt, hom_cov);
|
||||
het_cnt = NULL;
|
||||
if(round == asm_opt.number_of_round-1 && asm_opt.is_dbg_het_cnt) CALLOC(het_cnt, R_INF.total_reads);
|
||||
// fprintf(stderr, "[M::%s-start]\n", __func__);
|
||||
kt_for(asm_opt.thread_num, worker_ovec_cal0, b, R_INF.total_reads);///debug_for_fix
|
||||
// fprintf(stderr, "[M::%s-end]\n", __func__);
|
||||
|
||||
ha_pt_destroy(ha_idx);
|
||||
ha_idx = NULL;
|
||||
|
||||
if(het_cnt) {
|
||||
print_het_cnt_log(het_cnt); free(het_cnt); het_cnt = NULL;
|
||||
}
|
||||
|
||||
// collect statistics
|
||||
for (i = 0; i < asm_opt.thread_num; ++i) {
|
||||
asm_opt.num_bases += b[i]->num_read_base;
|
||||
asm_opt.num_corrected_bases += b[i]->num_correct_base;
|
||||
asm_opt.num_recorrected_bases += b[i]->num_recorrect_base;
|
||||
asm_opt.mem_buf += ha_ovec_mem(b[i], NULL);
|
||||
ha_ovec_destroy(b[i]);
|
||||
}
|
||||
free(b);
|
||||
asm_opt.hom_cov = hom_cov;
|
||||
asm_opt.het_cov = het_cov;
|
||||
}
|
||||
|
||||
|
||||
void update_overlaps(overlap_region_alloc* overlap_list, ma_hit_t_alloc* paf,
|
||||
UC_Read* g_read, UC_Read* overlap_read, int is_match, int is_exact)
|
||||
@@ -1517,6 +1729,52 @@ void Output_PAF()
|
||||
fprintf(stderr, "PAF has been written.\n");
|
||||
}
|
||||
|
||||
|
||||
void Output_PAF0(ma_hit_t_alloc* sources, const char *prefix)
|
||||
{
|
||||
fprintf(stderr, "Writing PAF to disk ...... \n");
|
||||
char* paf_name = (char*)malloc(strlen(asm_opt.output_file_name)+strlen(prefix)+50);
|
||||
sprintf(paf_name, "%s.%s.ovlp.paf", asm_opt.output_file_name, prefix);
|
||||
FILE* output_file = fopen(paf_name, "w");
|
||||
uint64_t i, j;
|
||||
|
||||
for (i = 0; i < R_INF.total_reads; i++)
|
||||
{
|
||||
for (j = 0; j < sources[i].length; j++)
|
||||
{
|
||||
fwrite(Get_NAME(R_INF, Get_qn(sources[i].buffer[j])), 1,
|
||||
Get_NAME_LENGTH(R_INF, Get_qn(sources[i].buffer[j])), output_file);
|
||||
fwrite("\t", 1, 1, output_file);
|
||||
fprintf(output_file, "%lu\t", (unsigned long)Get_READ_LENGTH(R_INF, Get_qn(sources[i].buffer[j])));
|
||||
fprintf(output_file, "%d\t", Get_qs(sources[i].buffer[j]));
|
||||
fprintf(output_file, "%d\t", Get_qe(sources[i].buffer[j]));
|
||||
if(sources[i].buffer[j].rev)
|
||||
{
|
||||
fprintf(output_file, "-\t");
|
||||
}
|
||||
else
|
||||
{
|
||||
fprintf(output_file, "+\t");
|
||||
}
|
||||
fwrite(Get_NAME(R_INF, Get_tn(sources[i].buffer[j])), 1,
|
||||
Get_NAME_LENGTH(R_INF, Get_tn(sources[i].buffer[j])), output_file);
|
||||
fwrite("\t", 1, 1, output_file);
|
||||
fprintf(output_file, "%lu\t", (unsigned long)Get_READ_LENGTH(R_INF, Get_tn(sources[i].buffer[j])));
|
||||
fprintf(output_file, "%d\t", Get_ts(sources[i].buffer[j]));
|
||||
fprintf(output_file, "%d\t", Get_te(sources[i].buffer[j]));
|
||||
fprintf(output_file, "%d\t", sources[i].buffer[j].ml);
|
||||
fprintf(output_file, "%d\t", sources[i].buffer[j].bl);
|
||||
fprintf(output_file, "255\n");
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
free(paf_name);
|
||||
fclose(output_file);
|
||||
|
||||
fprintf(stderr, "PAF has been written.\n");
|
||||
}
|
||||
|
||||
int check_cluster(uint64_t* list, long long listLen, ma_hit_t_alloc* paf, float threshold)
|
||||
{
|
||||
long long i, k;
|
||||
@@ -1645,7 +1903,7 @@ void hap_recalculate_peaks(char* output_file_name)
|
||||
int hom_cov, het_cov;
|
||||
// construct hash table for high occurrence k-mers
|
||||
if (!(asm_opt.flag & HA_F_NO_KMER_FLT)) {
|
||||
ha_flt_tab = ha_ft_gen(&asm_opt, &R_INF, &hom_cov, 0);
|
||||
ha_flt_tab = ha_ft_gen(&asm_opt, &R_INF, &hom_cov, 0, 0);
|
||||
ha_opt_update_cov(&asm_opt, hom_cov);
|
||||
}
|
||||
free(R_INF.read_length);
|
||||
@@ -1694,6 +1952,30 @@ void ha_overlap_final(void)
|
||||
asm_opt.het_cov = het_cov;
|
||||
}
|
||||
|
||||
void ha_ec_ff(int renew_idx)
|
||||
{
|
||||
int hom_cov, het_cov;
|
||||
ha_flt_tab_hp = ha_idx_hp = NULL;
|
||||
|
||||
if(ha_idx && renew_idx) {
|
||||
ha_pt_destroy(ha_idx); ha_idx = NULL;
|
||||
}
|
||||
|
||||
if(!ha_idx) {
|
||||
ha_idx = ha_pt_gen(&asm_opt, ha_flt_tab, 1, 0, &R_INF, &hom_cov, &het_cov); // build the index
|
||||
asm_opt.hom_cov = hom_cov; asm_opt.het_cov = het_cov;
|
||||
}
|
||||
|
||||
cal_ov_r(asm_opt.thread_num, R_INF.total_reads, renew_idx);
|
||||
|
||||
if(asm_opt.write_pos_idx) {
|
||||
refresh_pt_idx(&ha_flt_tab, &ha_idx, NULL, &asm_opt, asm_opt.output_file_name, 1);
|
||||
// write_pt_index(ha_flt_tab, ha_idx, NULL, &asm_opt, asm_opt.output_file_name);
|
||||
} else {
|
||||
ha_pt_destroy(ha_idx); ha_idx = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
static void worker_ov_utg(void *data, long i, int tid)
|
||||
{
|
||||
ha_ovec_buf_t *b = ((ha_ovec_buf_t**)data)[tid];
|
||||
@@ -1746,13 +2028,137 @@ void ug_idx_build(ma_ug_t *ug, int hap_n)
|
||||
exit(1);
|
||||
}
|
||||
|
||||
int ha_assemble_ovec(void)
|
||||
{
|
||||
extern void ha_extract_print_list(const All_reads *rs, int n_rounds, const char *o);
|
||||
int r = 0, hom_cov = -1;
|
||||
|
||||
ha_flt_tab = ha_idx = NULL;
|
||||
|
||||
// construct hash table for high occurrence k-mers
|
||||
if (!(asm_opt.flag & HA_F_NO_KMER_FLT) && ha_flt_tab == NULL) {
|
||||
ha_flt_tab = ha_ft_gen(&asm_opt, &R_INF, &hom_cov, 0, 0);
|
||||
ha_opt_update_cov(&asm_opt, hom_cov);
|
||||
}
|
||||
// error correction
|
||||
assert(asm_opt.number_of_round > 0);
|
||||
ha_opt_reset_to_round(&asm_opt, r); // this update asm_opt.roundID and a few other fields
|
||||
ha_overlap_cal(r, 0);
|
||||
fprintf(stderr, "[M::%s] size of buffer: %.3fGB\n", __func__, asm_opt.mem_buf / 1073741824.0);
|
||||
|
||||
|
||||
ha_print_ovlp_stat(R_INF.paf, R_INF.reverse_paf, R_INF.total_reads);
|
||||
ha_ft_destroy(ha_flt_tab);
|
||||
|
||||
Output_PAF0(R_INF.paf, "0");
|
||||
Output_PAF0(R_INF.reverse_paf, "1");
|
||||
if (asm_opt.flag & HA_F_WRITE_PAF) Output_PAF();
|
||||
|
||||
destory_All_reads(&R_INF);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int ha_assemble_ovec_cc(void)
|
||||
{
|
||||
ha_idx = NULL;
|
||||
|
||||
ha_opt_reset_to_round(&asm_opt, 0); // this update asm_opt.roundID and a few other fields
|
||||
ha_ec_dbg();
|
||||
|
||||
// Output_PAF0(R_INF.paf, "0");
|
||||
destory_All_reads(&R_INF);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int ha_assemble(void)
|
||||
{
|
||||
// debug_mc_g_t(MC_NAME);
|
||||
// debug_mc_gg_t(MC_NAME, 0, 0);
|
||||
// quick_debug_phasing(MC_NAME);
|
||||
extern void ha_extract_print_list(const All_reads *rs, int n_rounds, const char *o);
|
||||
int r, hom_cov = -1, ovlp_loaded = 0;
|
||||
int r, r0 = -1, hom_cov = -1, ovlp_loaded = 0; uint64_t tot_b, tot_e;
|
||||
if (asm_opt.load_index_from_disk && load_all_data_from_disk(&R_INF.paf, &R_INF.reverse_paf, asm_opt.output_file_name)) {
|
||||
ovlp_loaded = 1;
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> loaded corrected reads and overlaps from disk\n", __func__, yak_realtime(), yak_cpu_usage());
|
||||
if (asm_opt.extract_list) {
|
||||
ha_extract_print_list(&R_INF, asm_opt.extract_iter, asm_opt.extract_list);
|
||||
exit(0);
|
||||
}
|
||||
if (asm_opt.flag & HA_F_WRITE_EC) {
|
||||
if(asm_opt.is_sc) Output_corrected_fastq();
|
||||
else Output_corrected_reads();
|
||||
}
|
||||
if (asm_opt.flag & HA_F_WRITE_PAF) Output_PAF();
|
||||
if (asm_opt.het_cov == -1024) hap_recalculate_peaks(asm_opt.output_file_name), ovlp_loaded = 2;
|
||||
}
|
||||
if (!ovlp_loaded) {
|
||||
ha_flt_tab = ha_idx = NULL;
|
||||
if((asm_opt.flag & HA_F_VERBOSE_GFA)) load_pt_index(&ha_flt_tab, &ha_idx, &R_INF, &asm_opt, asm_opt.output_file_name), load_ct_index(&ha_ct_table, asm_opt.output_file_name);
|
||||
r = ha_idx?asm_opt.number_of_round-1:0;
|
||||
if((!ha_idx) && (asm_opt.restart)) {
|
||||
for (r = asm_opt.number_of_round - 1; r >= 0; --r) {
|
||||
if(tmp_pt_pro(&ha_flt_tab, &ha_idx, &R_INF, &asm_opt, asm_opt.output_file_name, r, asm_opt.number_of_round, 1)) {
|
||||
load_ct_index(&ha_ct_table, asm_opt.output_file_name); r0 = r;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if(r < 0) r = 0;
|
||||
}
|
||||
|
||||
// construct hash table for high occurrence k-mers
|
||||
if (!(asm_opt.flag & HA_F_NO_KMER_FLT) && ha_flt_tab == NULL) {
|
||||
ha_flt_tab = ha_ft_gen(&asm_opt, &R_INF, &hom_cov, 0, 0);
|
||||
ha_opt_update_cov(&asm_opt, hom_cov);
|
||||
}
|
||||
// error correction
|
||||
assert(asm_opt.number_of_round > 0);
|
||||
for (; r < asm_opt.number_of_round; ++r) {
|
||||
ha_opt_reset_to_round(&asm_opt, r); // this update asm_opt.roundID and a few other fields
|
||||
tot_b = tot_e = 0;
|
||||
// ha_overlap_and_correct(r);
|
||||
ha_ec(r, asm_opt.number_of_pround, (r<asm_opt.number_of_round-1)?1:0, &tot_b, &tot_e, ((r > r0) && (asm_opt.restart))?1:0);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f@%.3fGB] ==> corrected reads for round %d\n", __func__, yak_realtime(),
|
||||
yak_cpu_usage(), yak_peakrss_in_gb(), r + 1);
|
||||
fprintf(stderr, "[M::%s] # bases: %lu; # corrected bases: %lu\n", __func__, tot_b, tot_e);
|
||||
// fprintf(stderr, "[M::%s] # bases: %lld; # corrected bases: %lld; # recorrected bases: %lld\n", __func__,
|
||||
// asm_opt.num_bases, asm_opt.num_corrected_bases, asm_opt.num_recorrected_bases);
|
||||
// fprintf(stderr, "[M::%s] size of buffer: %.3fGB\n", __func__, asm_opt.mem_buf / 1073741824.0);
|
||||
}
|
||||
if (asm_opt.flag & HA_F_WRITE_EC) {
|
||||
if(asm_opt.is_sc) Output_corrected_fastq();
|
||||
else Output_corrected_reads();
|
||||
}
|
||||
// overlap between corrected reads
|
||||
ha_opt_reset_to_round(&asm_opt, asm_opt.number_of_round);
|
||||
// ha_overlap_final();
|
||||
ha_ec_ff(1/**0**/);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f@%.3fGB] ==> found overlaps for the final round\n", __func__, yak_realtime(), yak_cpu_usage(), yak_peakrss_in_gb());
|
||||
// fprintf(stderr, "\n[M::%s::%.3f*%.2f@%.3fGB] ==> found overlaps for the final round\n", __func__, yak_realtime(), yak_cpu_usage(), yak_peakrss_in_gb());
|
||||
// ha_print_ovlp_stat(R_INF.paf, R_INF.reverse_paf, R_INF.total_reads);
|
||||
if(!(asm_opt.write_pos_idx)) {
|
||||
ha_ft_destroy(ha_flt_tab); ha_flt_tab = NULL;
|
||||
}
|
||||
if (asm_opt.flag & HA_F_WRITE_PAF) Output_PAF();
|
||||
ha_triobin(&asm_opt);
|
||||
|
||||
// exit(1);
|
||||
}
|
||||
if(ovlp_loaded == 2) ovlp_loaded = 0;
|
||||
ha_opt_update_cov_min(&asm_opt, asm_opt.hom_cov, MIN_N_CHAIN);
|
||||
|
||||
build_string_graph_without_clean(asm_opt.min_overlap_coverage, R_INF.paf, R_INF.reverse_paf,
|
||||
R_INF.total_reads, R_INF.read_length, asm_opt.min_overlap_Len, asm_opt.max_hang_Len, asm_opt.clean_round,
|
||||
asm_opt.gap_fuzz, asm_opt.min_drop_rate, asm_opt.max_drop_rate, asm_opt.output_file_name, asm_opt.large_pop_bubble_size, 0, !ovlp_loaded);
|
||||
destory_All_reads(&R_INF); if(asm_opt.dbg_bam) destroy_cc_v(&scb);
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
int ha_assemble_pair(void)
|
||||
{
|
||||
extern void ha_extract_print_list(const All_reads *rs, int n_rounds, const char *o);
|
||||
int r = 0, hom_cov = -1, ovlp_loaded = 0; memset((&R_INF), 0, sizeof(R_INF));
|
||||
|
||||
if (asm_opt.load_index_from_disk && load_all_data_from_disk(&R_INF.paf, &R_INF.reverse_paf, asm_opt.output_file_name)) {
|
||||
ovlp_loaded = 1;
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> loaded corrected reads and overlaps from disk\n", __func__, yak_realtime(), yak_cpu_usage());
|
||||
@@ -1764,37 +2170,44 @@ int ha_assemble(void)
|
||||
if (asm_opt.flag & HA_F_WRITE_PAF) Output_PAF();
|
||||
if (asm_opt.het_cov == -1024) hap_recalculate_peaks(asm_opt.output_file_name), ovlp_loaded = 2;
|
||||
}
|
||||
|
||||
|
||||
if (!ovlp_loaded) {
|
||||
ha_flt_tab = ha_idx = NULL;
|
||||
|
||||
if(!append_All_reads(&R_INF, asm_opt.output_file_name, 0)) {
|
||||
fprintf(stderr, "[M::%s::] Cannot load %s.0\n", __func__, asm_opt.output_file_name);
|
||||
exit(1);
|
||||
}
|
||||
|
||||
if(!append_All_reads(&R_INF, asm_opt.output_file_name, 1)) {
|
||||
fprintf(stderr, "[M::%s::] Cannot load %s.1\n", __func__, asm_opt.output_file_name);
|
||||
exit(1);
|
||||
}
|
||||
// Output_corrected_reads(); exit(0);
|
||||
|
||||
ha_flt_tab = ha_idx = NULL; r = asm_opt.number_of_round - 1;
|
||||
if((asm_opt.flag & HA_F_VERBOSE_GFA)) load_pt_index(&ha_flt_tab, &ha_idx, &R_INF, &asm_opt, asm_opt.output_file_name), load_ct_index(&ha_ct_table, asm_opt.output_file_name);
|
||||
|
||||
// construct hash table for high occurrence k-mers
|
||||
if (!(asm_opt.flag & HA_F_NO_KMER_FLT) && ha_flt_tab == NULL)
|
||||
{
|
||||
ha_flt_tab = ha_ft_gen(&asm_opt, &R_INF, &hom_cov, 0);
|
||||
if (!(asm_opt.flag & HA_F_NO_KMER_FLT) && ha_flt_tab == NULL) {
|
||||
ha_flt_tab = ha_ft_gen(&asm_opt, &R_INF, &hom_cov, 0, 1);
|
||||
ha_opt_update_cov(&asm_opt, hom_cov);
|
||||
}
|
||||
// error correction
|
||||
assert(asm_opt.number_of_round > 0);
|
||||
for (r = ha_idx?asm_opt.number_of_round-1:0; r < asm_opt.number_of_round; ++r) {
|
||||
ha_opt_reset_to_round(&asm_opt, r); // this update asm_opt.roundID and a few other fields
|
||||
ha_overlap_and_correct(r);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f@%.3fGB] ==> corrected reads for round %d\n", __func__, yak_realtime(),
|
||||
yak_cpu_usage(), yak_peakrss_in_gb(), r + 1);
|
||||
fprintf(stderr, "[M::%s] # bases: %lld; # corrected bases: %lld; # recorrected bases: %lld\n", __func__,
|
||||
asm_opt.num_bases, asm_opt.num_corrected_bases, asm_opt.num_recorrected_bases);
|
||||
ha_overlap_cal(r, 1);
|
||||
fprintf(stderr, "[M::%s] size of buffer: %.3fGB\n", __func__, asm_opt.mem_buf / 1073741824.0);
|
||||
}
|
||||
if (asm_opt.flag & HA_F_WRITE_EC) Output_corrected_reads();
|
||||
// overlap between corrected reads
|
||||
ha_opt_reset_to_round(&asm_opt, asm_opt.number_of_round);
|
||||
ha_overlap_final();
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f@%.3fGB] ==> found overlaps for the final round\n", __func__, yak_realtime(),
|
||||
yak_cpu_usage(), yak_peakrss_in_gb());
|
||||
|
||||
|
||||
if (asm_opt.flag & HA_F_WRITE_EC) Output_corrected_reads();
|
||||
ha_print_ovlp_stat(R_INF.paf, R_INF.reverse_paf, R_INF.total_reads);
|
||||
ha_ft_destroy(ha_flt_tab);
|
||||
if (asm_opt.flag & HA_F_WRITE_PAF) Output_PAF();
|
||||
ha_triobin(&asm_opt);
|
||||
|
||||
}
|
||||
if(ovlp_loaded == 2) ovlp_loaded = 0;
|
||||
ha_opt_update_cov_min(&asm_opt, asm_opt.hom_cov, MIN_N_CHAIN);
|
||||
|
||||
+4
-2
@@ -44,11 +44,13 @@ typedef struct {
|
||||
} ha_ovec_buf_t;
|
||||
|
||||
int ha_assemble(void);
|
||||
int ha_assemble_pair(void);
|
||||
void ug_idx_build(ma_ug_t *ug, int hap_n);
|
||||
ha_ovec_buf_t *ha_ovec_init(int is_final, int save_ov, int is_ug);
|
||||
ha_ovec_buf_t *ha_ovec_buf_init(void *km, int is_final, int save_ov, int is_ug);
|
||||
void ha_ovec_destroy(ha_ovec_buf_t *b);
|
||||
int64_t ha_ovec_mem(const ha_ovec_buf_t *b, int64_t *mem_a)
|
||||
;
|
||||
int64_t ha_ovec_mem(const ha_ovec_buf_t *b, int64_t *mem_a);
|
||||
int ha_assemble_ovec(void);
|
||||
int ha_ec_dbg(void);
|
||||
|
||||
#endif
|
||||
|
||||
+359
-26
@@ -7,6 +7,9 @@
|
||||
#include <sys/time.h>
|
||||
#include "CommandLines.h"
|
||||
#include "ketopt.h"
|
||||
#include "kseq.h"
|
||||
|
||||
KSEQ_INIT(gzFile, gzread)
|
||||
|
||||
#define DEFAULT_OUTPUT "hifiasm.asm"
|
||||
|
||||
@@ -18,8 +21,8 @@ static ko_longopt_t long_options[] = {
|
||||
{ "write-paf", ko_no_argument, 302 },
|
||||
{ "write-ec", ko_no_argument, 303 },
|
||||
{ "skip-triobin", ko_no_argument, 304 },
|
||||
{ "max-od-ec", ko_no_argument, 305 },
|
||||
{ "max-od-final", ko_no_argument, 306 },
|
||||
{ "max-od-ec", ko_required_argument, 305 },
|
||||
{ "max-od-final", ko_required_argument, 306 },
|
||||
{ "ex-list", ko_required_argument, 307 },
|
||||
{ "ex-iter", ko_required_argument, 308 },
|
||||
{ "hom-cov", ko_required_argument, 309 },
|
||||
@@ -56,6 +59,42 @@ static ko_longopt_t long_options[] = {
|
||||
{ "bin-only", ko_no_argument, 341},
|
||||
{ "ul-round", ko_required_argument, 342},
|
||||
{ "prt-raw", ko_no_argument, 343},
|
||||
{ "integer-correct", ko_required_argument, 344},
|
||||
{ "dbg-ovec", ko_no_argument, 345},
|
||||
{ "path-max", ko_required_argument, 346},
|
||||
{ "path-min", ko_required_argument, 347},
|
||||
{ "trio-dual", ko_no_argument, 348},
|
||||
{ "ul-cut", ko_required_argument, 349},
|
||||
{ "dual-scaf", ko_no_argument, 350},
|
||||
{ "scaf-gap", ko_required_argument, 351},
|
||||
{ "sec-in", ko_required_argument, 352},
|
||||
{ "somatic-cov", ko_required_argument, 353},
|
||||
{ "telo-m", ko_required_argument, 354},
|
||||
{ "telo-p", ko_required_argument, 355},
|
||||
{ "telo-d", ko_required_argument, 356},
|
||||
{ "telo-s", ko_required_argument, 357},
|
||||
{ "ctg-n", ko_required_argument, 358},
|
||||
{ "ont", ko_no_argument, 359},
|
||||
// { "sc-n", ko_no_argument, 360},
|
||||
{ "chem-c", ko_required_argument, 361},
|
||||
{ "chem-f", ko_required_argument, 362},
|
||||
{ "ul-m", ko_required_argument, 363},
|
||||
{ "rl-cut", ko_required_argument, 364},
|
||||
{ "sc-cut", ko_required_argument, 365},
|
||||
{ "hf", ko_required_argument, 366},
|
||||
{ "cb", ko_required_argument, 367},
|
||||
{ "gpath", ko_no_argument, 368},
|
||||
{ "het-cov", ko_required_argument, 369},
|
||||
{ "resume", ko_no_argument, 370},
|
||||
{ "flt-kocc", ko_required_argument, 371},
|
||||
{ "chn-occ", ko_required_argument, 372},
|
||||
{ "dbg-in1", ko_required_argument, 373},
|
||||
{ "dbg-in2", ko_required_argument, 374},
|
||||
{ "ec-only", ko_no_argument, 375},
|
||||
{ "hyb-syn", ko_required_argument, 376},
|
||||
{ "simd-m", ko_required_argument, 377},
|
||||
{ "del-hf", ko_no_argument, 378},
|
||||
// { "path-round", ko_required_argument, 348},
|
||||
{ 0, 0, 0 }
|
||||
};
|
||||
|
||||
@@ -75,11 +114,15 @@ void Print_H(hifiasm_opt_t* asm_opt)
|
||||
fprintf(stderr, " -t INT number of threads [%d]\n", asm_opt->thread_num);
|
||||
fprintf(stderr, " -h show help information\n");
|
||||
fprintf(stderr, " --version show version number\n");
|
||||
fprintf(stderr, " Preset options:\n");
|
||||
fprintf(stderr, " --ont assemble Oxford Nanopore reads\n");
|
||||
fprintf(stderr, " Overlap/Error correction:\n");
|
||||
fprintf(stderr, " -k INT k-mer length (must be <64) [%d]\n", asm_opt->k_mer_length);
|
||||
fprintf(stderr, " -w INT minimizer window size [%d]\n", asm_opt->mz_win);
|
||||
fprintf(stderr, " -f INT number of bits for bloom filter; 0 to disable [%d]\n", asm_opt->bf_shift);
|
||||
fprintf(stderr, " -D FLOAT drop k-mers occurring >FLOAT*coverage times [%.1f]\n", asm_opt->high_factor);
|
||||
fprintf(stderr, " -D FLOAT drop k-mers occurring >FLOAT*coverage times [%.1f]; work with --flt-kocc or -N\n", asm_opt->high_factor);
|
||||
fprintf(stderr, " --flt-kocc INT\n");
|
||||
fprintf(stderr, " drop k-mers occurring >max(-D*coverage,--flt-kocc) times [%ld]\n", asm_opt->hf_cutoff);
|
||||
fprintf(stderr, " -N INT consider up to max(-D*coverage,-N) overlaps for each oriented read [%d]\n", asm_opt->max_n_chain);
|
||||
fprintf(stderr, " -r INT round of correction [%d]\n", asm_opt->number_of_round);
|
||||
fprintf(stderr, " -z INT length of adapters that should be removed [%d]\n", asm_opt->adapterLen);
|
||||
@@ -87,6 +130,16 @@ void Print_H(hifiasm_opt_t* asm_opt)
|
||||
fprintf(stderr, " employ k-mers occurring <INT times to rescue repetitive overlaps [%d]\n", asm_opt->max_kmer_cnt);
|
||||
fprintf(stderr, " --hg-size INT(k, m or g)\n");
|
||||
fprintf(stderr, " estimated haploid genome size used for inferring read coverage [auto]\n");
|
||||
fprintf(stderr, " --resume resume from the previously incomplete assembly [%ld]\n", asm_opt->restart);
|
||||
fprintf(stderr, " --het-cov INT\n");
|
||||
fprintf(stderr, " heterozygous read coverage [auto]; used for error correction and assembly; manual value overrides auto\n");
|
||||
fprintf(stderr, " --hom-cov INT\n");
|
||||
fprintf(stderr, " homozygous read coverage [auto]; used for error correction and assembly; manual value overrides auto\n");
|
||||
fprintf(stderr, " --chn-occ INT\n");
|
||||
fprintf(stderr, " discard overlaps supported by <INT minimizers [%ld]\n", asm_opt->chn_occ);
|
||||
fprintf(stderr, " --ec-only error correction only; disable overlapping and assembly\n");
|
||||
fprintf(stderr, " --simd-m use SIMD acceleration when supported: AVX-512 (2), AVX2 (1), or non-SIMD (0)\n");
|
||||
|
||||
fprintf(stderr, " Assembly:\n");
|
||||
fprintf(stderr, " -a INT round of assembly cleaning [%d]\n", asm_opt->clean_round);
|
||||
fprintf(stderr, " -m INT pop bubbles of <INT in size in contig graphs [%lld]\n", asm_opt->large_pop_bubble_size);
|
||||
@@ -95,9 +148,9 @@ void Print_H(hifiasm_opt_t* asm_opt)
|
||||
fprintf(stderr, " -x FLOAT max overlap drop ratio [%.2g]\n", asm_opt->max_drop_rate);
|
||||
fprintf(stderr, " -y FLOAT min overlap drop ratio [%.2g]\n", asm_opt->min_drop_rate);
|
||||
fprintf(stderr, " -i ignore saved read correction and overlaps\n");
|
||||
fprintf(stderr, " -u disable post-join step for contigs which may improve N50\n");
|
||||
fprintf(stderr, " --hom-cov INT\n");
|
||||
fprintf(stderr, " homozygous read coverage [auto]\n");
|
||||
fprintf(stderr, " -u post-join step for contigs which may improve N50; 0 to disable; 1 to enable\n");
|
||||
fprintf(stderr, " [%u] and [%u] in default for the UL+HiFi assembly and the HiFi assembly, respectively\n",
|
||||
asm_opt->ul_pst_join, asm_opt->hifi_pst_join);
|
||||
fprintf(stderr, " --lowQ INT\n");
|
||||
fprintf(stderr, " output contig regions with >=INT%% inconsistency in BED format; 0 to disable [%d]\n", asm_opt->bed_inconsist_rate);
|
||||
fprintf(stderr, " --b-cov INT\n");
|
||||
@@ -108,6 +161,10 @@ void Print_H(hifiasm_opt_t* asm_opt)
|
||||
fprintf(stderr, " break contigs at positions with <=FLOAT*coverage exact overlaps;\n");
|
||||
fprintf(stderr, " only work with '--b-cov' or '--h-cov'[%.2f]\n", asm_opt->m_rate);
|
||||
fprintf(stderr, " --primary output a primary assembly and an alternate assembly\n");
|
||||
fprintf(stderr, " --ctg-n INT\n");
|
||||
fprintf(stderr, " remove tip contigs composed of <=INT reads [%d]\n", asm_opt->max_contig_tip);
|
||||
fprintf(stderr, " --gpath output the corresponding path of each contig (p_ctg) within the assembly graph (d_utg.noseq.gfa)\n");
|
||||
|
||||
|
||||
// fprintf(stderr, " --pri-range INT1[,INT2]\n");
|
||||
// fprintf(stderr, " keep contigs with coverage in this range in p_ctg.gfa; -1 to disable [auto,inf]\n");
|
||||
@@ -122,6 +179,8 @@ void Print_H(hifiasm_opt_t* asm_opt)
|
||||
fprintf(stderr, " --t-occ INT\n");
|
||||
fprintf(stderr, " forcedly remove unitigs with >INT unexpected haplotype-specific reads;\n");
|
||||
fprintf(stderr, " ignore graph topology; [%d]\n", asm_opt->trio_flag_occ_thres);
|
||||
fprintf(stderr, " --trio-dual utilize homology information to correct trio phasing errors\n");
|
||||
|
||||
|
||||
fprintf(stderr, " Purge-dups:\n");
|
||||
fprintf(stderr, " -l INT purge level. 0: no purging; 1: light; 2/3: aggressive [0 for trio; 3 for unzip]\n");
|
||||
@@ -152,14 +211,63 @@ void Print_H(hifiasm_opt_t* asm_opt)
|
||||
fprintf(stderr, " --l-msjoin INT\n");
|
||||
fprintf(stderr, " detect misjoined unitigs of >=INT in size; 0 to disable [%lu]\n", asm_opt->misjoin_len);
|
||||
|
||||
fprintf(stderr, " Ultra-Long-integration (beta):\n");
|
||||
fprintf(stderr, " Ultra-Long-integration:\n");
|
||||
fprintf(stderr, " --ul FILEs file names of Ultra-Long reads [r1.fq,r2.fq,...]\n");
|
||||
///pending for integration
|
||||
/**
|
||||
fprintf(stderr, " --ul-m INT\n");
|
||||
fprintf(stderr, " hybrid assembly mode. 0: fast and memory efficent; 1: may produce better assembly with ONT R10 [%d]\n", asm_opt->ul_mod);
|
||||
**/
|
||||
fprintf(stderr, " --ul-rate FLOAT\n");
|
||||
fprintf(stderr, " error rate of Ultra-Long reads [%.3g]\n", asm_opt->ul_error_rate);
|
||||
fprintf(stderr, " --ul-tip INT\n");
|
||||
fprintf(stderr, " remove tip unitigs composed of <=INT reads for the UL assembly [%d]\n", asm_opt->max_short_ul_tip);
|
||||
fprintf(stderr, " --path-max FLOAT\n");
|
||||
fprintf(stderr, " max path drop ratio [%.2g]; higher number may make the assembly cleaner\n", asm_opt->max_path_drop_rate);
|
||||
fprintf(stderr, " but may lead to more misassemblies\n");
|
||||
fprintf(stderr, " --path-min FLOAT\n");
|
||||
fprintf(stderr, " min path drop ratio [%.2g]; higher number may make the assembly cleaner\n", asm_opt->min_path_drop_rate);
|
||||
fprintf(stderr, " but may lead to more misassemblies\n");
|
||||
fprintf(stderr, " --ul-cut INT\n");
|
||||
fprintf(stderr, " filter out <INT UL reads during the UL assembly [%d]\n", asm_opt->ul_min_base);
|
||||
// fprintf(stderr, " --low-het enable it for genomes with very low het heterozygosity rate (<0.0001%%)\n");
|
||||
|
||||
fprintf(stderr, " Dual-Scaffolding:\n");
|
||||
fprintf(stderr, " --dual-scaf output scaffolding\n");
|
||||
fprintf(stderr, " --scaf-gap INT\n");
|
||||
fprintf(stderr, " max gap size for scaffolding [%ld]\n", asm_opt->self_scaf_gap_max);
|
||||
|
||||
fprintf(stderr, " Telomere-identification:\n");
|
||||
fprintf(stderr, " --telo-m STR\n");
|
||||
fprintf(stderr, " telomere motif at 5'-end; CCCTAA for human [%s]\n", ((asm_opt->telo_motif)?(asm_opt->telo_motif):("NULL")));///5'-end, check CCCTAA
|
||||
fprintf(stderr, " --telo-p INT\n");
|
||||
fprintf(stderr, " non-telomeric penalty [%ld]\n", asm_opt->telo_pen);
|
||||
fprintf(stderr, " --telo-d INT\n");
|
||||
fprintf(stderr, " max drop [%ld]\n", asm_opt->telo_drop);
|
||||
fprintf(stderr, " --telo-s INT\n");
|
||||
fprintf(stderr, " min score for telomere reads [%ld]\n", asm_opt->telo_mic_sc);
|
||||
|
||||
fprintf(stderr, " ONT Simplex assembly (beta):\n");
|
||||
fprintf(stderr, " --ont assemble ONT Simplex reads in fastq format\n");
|
||||
// fprintf(stderr, " --sc-n consider base qual value for assembly\n");
|
||||
fprintf(stderr, " --chem-c INT\n");
|
||||
// fprintf(stderr, " detect chimeric reads with <=INT other reads support [%lu]\n", asm_opt->chemical_cov);
|
||||
fprintf(stderr, " detect chimeric reads with <=INT other reads support [auto]\n");
|
||||
fprintf(stderr, " --chem-f INT\n");
|
||||
fprintf(stderr, " length of flanking regions for chimeric read detection [%lu]\n", asm_opt->chemical_flank);
|
||||
fprintf(stderr, " --rl-cut INT\n");
|
||||
fprintf(stderr, " filter out ONT Simplex reads shorter than <INT> for assembly [%ld]\n", asm_opt->rl_cut);
|
||||
fprintf(stderr, " --sc-cut INT\n");
|
||||
fprintf(stderr, " filter out ONT Simplex reads with a mean base quality score below <INT> [%ld]\n", asm_opt->sc_cut);
|
||||
fprintf(stderr, " --hf FILEs HiFi read file(s)\n");
|
||||
fprintf(stderr, " --hyb-syn INT\n");
|
||||
fprintf(stderr, " hybrid correction mode (requires --hf) [%d]:\n", asm_opt->hyb_syn);
|
||||
fprintf(stderr, " 1: all-vs-all (ONT<-all, HiFi<-all)\n");
|
||||
fprintf(stderr, " 2: ONT<-all, HiFi<-HiFi\n");
|
||||
fprintf(stderr, " 3: mode 2 + ONT/HiFi sync to reduce bias\n");
|
||||
|
||||
|
||||
|
||||
fprintf(stderr, "Example: ./hifiasm -o NA12878.asm -t 32 NA12878.fq.gz\n");
|
||||
fprintf(stderr, "See `https://hifiasm.readthedocs.io/en/latest/' or `man ./hifiasm.1' for complete documentation.\n");
|
||||
}
|
||||
@@ -178,6 +286,7 @@ void init_opt(hifiasm_opt_t* asm_opt)
|
||||
asm_opt->hic_reads[0] = NULL;
|
||||
asm_opt->hic_reads[1] = NULL;
|
||||
asm_opt->fn_bin_poy = NULL;
|
||||
asm_opt->fn_chr_bin = NULL;
|
||||
asm_opt->ar = NULL;
|
||||
asm_opt->thread_num = 1;
|
||||
asm_opt->k_mer_length = 51;
|
||||
@@ -194,6 +303,7 @@ void init_opt(hifiasm_opt_t* asm_opt)
|
||||
asm_opt->max_kmer_cnt = 2000;
|
||||
asm_opt->high_factor = 5.0;
|
||||
asm_opt->max_ov_diff_ec = 0.04;
|
||||
asm_opt->max_ov_diff_ec_sec = 0.04;
|
||||
asm_opt->max_ov_diff_final = 0.03;
|
||||
asm_opt->hom_cov = 20;
|
||||
asm_opt->het_cov = -1024;
|
||||
@@ -202,6 +312,7 @@ void init_opt(hifiasm_opt_t* asm_opt)
|
||||
asm_opt->load_index_from_disk = 1;
|
||||
asm_opt->write_index_to_disk = 1;
|
||||
asm_opt->number_of_round = 3;
|
||||
asm_opt->number_of_pround = 0/**3**/;
|
||||
asm_opt->adapterLen = 0;
|
||||
asm_opt->clean_round = 4;
|
||||
///asm_opt->small_pop_bubble_size = 100000;
|
||||
@@ -216,6 +327,7 @@ void init_opt(hifiasm_opt_t* asm_opt)
|
||||
asm_opt->min_overlap_coverage = 0;
|
||||
asm_opt->max_short_tip = 3;
|
||||
asm_opt->max_short_ul_tip = 6;
|
||||
asm_opt->max_contig_tip = 3;
|
||||
asm_opt->min_cnt = 2;
|
||||
asm_opt->mid_cnt = 5;
|
||||
asm_opt->purge_level_primary = 3;
|
||||
@@ -267,6 +379,71 @@ void init_opt(hifiasm_opt_t* asm_opt)
|
||||
asm_opt->bin_only = 0;
|
||||
asm_opt->ul_clean_round = 1;
|
||||
asm_opt->prt_dbg_gfa = 0;
|
||||
asm_opt->integer_correct_round = 0;
|
||||
asm_opt->dbg_ovec_cal = 0;
|
||||
asm_opt->min_path_drop_rate = 0.2;
|
||||
asm_opt->max_path_drop_rate = 0.6;
|
||||
asm_opt->hifi_pst_join = 1;
|
||||
asm_opt->ul_pst_join = 1;
|
||||
asm_opt->trio_cov_het_ovlp = -1;
|
||||
asm_opt->ul_min_base = 0;
|
||||
asm_opt->self_scaf = 0;
|
||||
asm_opt->self_scaf_min = 250000;
|
||||
asm_opt->self_scaf_reliable_min = 5000000;
|
||||
asm_opt->self_scaf_gap_max = 3000000;
|
||||
asm_opt->sec_in = NULL;
|
||||
asm_opt->somatic_cov = -1;
|
||||
|
||||
asm_opt->telo_motif = NULL;
|
||||
asm_opt->telo_pen = 1;
|
||||
asm_opt->telo_drop = 2000;
|
||||
asm_opt->telo_mic_sc = 500;
|
||||
|
||||
asm_opt->is_ont = 0;
|
||||
asm_opt->is_sc = 0;
|
||||
asm_opt->chemical_cov = -1/**1**/;
|
||||
asm_opt->chemical_flank = 256;
|
||||
asm_opt->ul_mod = 0;
|
||||
|
||||
asm_opt->rl_cut = 1000;
|
||||
asm_opt->sc_cut = 10;
|
||||
|
||||
asm_opt->hf = NULL;
|
||||
|
||||
asm_opt->gpath = 0;
|
||||
|
||||
asm_opt->hf_rate = 4;
|
||||
asm_opt->ont_rate = 1;///must be 1 or 0
|
||||
asm_opt->hf_rate_max = 4;
|
||||
|
||||
asm_opt->het_cov_set = -1;
|
||||
asm_opt->restart = 0;
|
||||
|
||||
asm_opt->hf_cutoff = -1;
|
||||
|
||||
asm_opt->write_pos_idx = 0/**1**/;
|
||||
|
||||
asm_opt->hom_cov_0 = -1;
|
||||
asm_opt->het_cov_0 = -1;
|
||||
asm_opt->max_n_chain_0 = -1;
|
||||
|
||||
|
||||
asm_opt->hmo_cov_ss = -1;
|
||||
asm_opt->het_cov_ss = -1;
|
||||
asm_opt->chn_occ = 2;
|
||||
|
||||
asm_opt->dbg_run_1 = NULL;
|
||||
asm_opt->dbg_run_2 = NULL;
|
||||
|
||||
asm_opt->ec_only = 0;
|
||||
asm_opt->hyb_syn = 1;
|
||||
asm_opt->step_rd = -1/**128**/;
|
||||
|
||||
asm_opt->dbg_bam = 0;
|
||||
|
||||
asm_opt->simd_mm = -1;
|
||||
|
||||
asm_opt->del_hf = 0;
|
||||
}
|
||||
|
||||
void destory_enzyme(enzyme* f)
|
||||
@@ -340,14 +517,43 @@ static int check_file(char* name, const char* opt)
|
||||
|
||||
static int check_hic_reads(enzyme* f, const char* opt)
|
||||
{
|
||||
int i;
|
||||
for (i = 0; i < f->n; i++)
|
||||
{
|
||||
int32_t i;
|
||||
for (i = 0; i < f->n; i++) {
|
||||
if(check_file(f->a[i], opt) == 0) return 0;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
static int check_fq_files(enzyme* f, const char* opt, int32_t is_fq)
|
||||
{
|
||||
int32_t i, ret; gzFile dfp; kseq_t *ks = NULL;
|
||||
for (i = 0; i < f->n; i++) {
|
||||
if(!(f->a[i])) {
|
||||
fprintf(stderr, "[ERROR] input file does not exist (%s)\n", opt);
|
||||
return 0;
|
||||
}
|
||||
|
||||
dfp = gzopen(f->a[i], "r");
|
||||
if (dfp == 0) {
|
||||
fprintf(stderr, "[ERROR] Cannot find the input file: %s (%s)\n", f->a[i], opt);
|
||||
return 0;
|
||||
} else if(is_fq){
|
||||
ks = kseq_init(dfp);
|
||||
while (((ret = kseq_read(ks)) >= 0)) {
|
||||
if((ks->qual.l == 0) || (ks->qual.s == NULL)) {
|
||||
fprintf(stderr, "[ERROR] %s is in fasta format rather than fastq format (%s)\n", f->a[i], opt);
|
||||
fprintf(stderr, "[ERROR] set --ul-m 0 for fasta files\n");
|
||||
return 0;
|
||||
}
|
||||
break;
|
||||
}
|
||||
kseq_destroy(ks); ks = NULL;
|
||||
}
|
||||
gzclose(dfp);
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
int check_option(hifiasm_opt_t* asm_opt)
|
||||
{
|
||||
if(asm_opt->read_file_names == NULL || asm_opt->num_reads == 0)
|
||||
@@ -547,7 +753,7 @@ int check_option(hifiasm_opt_t* asm_opt)
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->ar != NULL && check_hic_reads(asm_opt->ar, "UL") == 0) return 0;
|
||||
if(asm_opt->ar != NULL && check_fq_files(asm_opt->ar, "--ul", asm_opt->ul_mod) == 0) return 0;
|
||||
if(asm_opt->ar != NULL && asm_opt->ar->n == 0)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] wrong UL reads (--ul)\n");
|
||||
@@ -596,30 +802,70 @@ int check_option(hifiasm_opt_t* asm_opt)
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->ul_mod != 0 && asm_opt->ul_mod != 1) {
|
||||
fprintf(stderr, "[ERROR] must be 0 or 1 (--ul-m)\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->telo_motif) {
|
||||
uint64_t k, tlen = strlen((asm_opt->telo_motif)); char c;
|
||||
if(tlen > 32) {
|
||||
fprintf(stderr, "[ERROR] [--telo-m] must be no longer than 32\n");
|
||||
return 0;
|
||||
}
|
||||
for (k = 0; k < tlen; k++) {
|
||||
c = asm_opt->telo_motif[k];
|
||||
if(c != 'A' && c != 'C' && c != 'G' && c != 'T' &&
|
||||
c != 'a' && c != 'c' && c != 'g' && c != 't') {
|
||||
fprintf(stderr, "[ERROR] [--telo-m] must be A/C/G/T\n");
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
if((asm_opt->hf) && (!(asm_opt->is_ont))) {
|
||||
fprintf(stderr, "[ERROR] [--hf] must work with [--ont]\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
if((asm_opt->hyb_syn != 1) && (asm_opt->hyb_syn != 2) && (asm_opt->hyb_syn != 3)) {
|
||||
fprintf(stderr, "[ERROR] [--hyb-syn] must be 1/2/3\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
void get_queries(int argc, char *argv[], ketopt_t* opt, hifiasm_opt_t* asm_opt)
|
||||
{
|
||||
if(opt->ind == argc)
|
||||
{
|
||||
if(opt->ind == argc) {
|
||||
return;
|
||||
}
|
||||
|
||||
asm_opt->num_reads = argc - opt->ind;
|
||||
asm_opt->read_file_names = (char**)malloc(sizeof(char*)*asm_opt->num_reads);
|
||||
|
||||
long long i;
|
||||
gzFile dfp;
|
||||
for (i = 0; i < asm_opt->num_reads; i++)
|
||||
{
|
||||
long long i; int ret;
|
||||
gzFile dfp; kseq_t *ks = NULL;
|
||||
for (i = 0; i < asm_opt->num_reads; i++) {
|
||||
asm_opt->read_file_names[i] = argv[i + opt->ind];
|
||||
dfp = gzopen(asm_opt->read_file_names[i], "r");
|
||||
if (dfp == 0)
|
||||
{
|
||||
if (dfp == 0) {
|
||||
fprintf(stderr, "[ERROR] Cannot find the input read file: %s\n",
|
||||
asm_opt->read_file_names[i]);
|
||||
exit(0);
|
||||
} else if(asm_opt->is_sc){
|
||||
ks = kseq_init(dfp);
|
||||
while (((ret = kseq_read(ks)) >= 0)) {
|
||||
if((ks->qual.l == 0) || (ks->qual.s == NULL)) {
|
||||
fprintf(stderr, "[ERROR] %s is in fasta format rather than fastq format\n", asm_opt->read_file_names[i]);
|
||||
asm_opt->is_sc = 0;
|
||||
exit(0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
kseq_destroy(ks); ks = NULL;
|
||||
}
|
||||
gzclose(dfp);
|
||||
}
|
||||
@@ -704,7 +950,7 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
|
||||
int c;
|
||||
|
||||
while ((c = ketopt(&opt, argc, argv, 1, "hvt:o:k:w:m:n:r:a:b:z:x:y:p:c:d:M:P:if:D:FN:1:2:3:4:5:l:s:O:eu", long_options)) >= 0) {
|
||||
while ((c = ketopt(&opt, argc, argv, 1, "hvt:o:k:w:m:n:r:a:b:z:x:y:p:c:d:M:P:if:D:FN:1:2:3:4:5:l:s:O:eu:", long_options)) >= 0) {
|
||||
if (c == 'h')
|
||||
{
|
||||
Print_H(asm_opt);
|
||||
@@ -741,7 +987,13 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
else if (c == 'm') asm_opt->large_pop_bubble_size = atoll(opt.arg);
|
||||
else if (c == 'n') asm_opt->max_short_tip = atoll(opt.arg);
|
||||
else if (c == 'e') asm_opt->flag |= HA_F_BAN_ASSEMBLY;
|
||||
else if (c == 'u') asm_opt->flag |= HA_F_BAN_POST_JOIN;
|
||||
else if (c == 'u') {
|
||||
if(atoll(opt.arg)) {
|
||||
asm_opt->hifi_pst_join = asm_opt->ul_pst_join = 1;
|
||||
} else {
|
||||
asm_opt->hifi_pst_join = asm_opt->ul_pst_join = 0;
|
||||
}
|
||||
}
|
||||
else if (c == 301) asm_opt->flag |= HA_F_VERBOSE_GFA;
|
||||
else if (c == 302) asm_opt->flag |= HA_F_WRITE_PAF;
|
||||
else if (c == 303) asm_opt->flag |= HA_F_WRITE_EC;
|
||||
@@ -750,12 +1002,14 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
else if (c == 306) asm_opt->max_ov_diff_final = atof(opt.arg);
|
||||
else if (c == 307) asm_opt->extract_list = opt.arg;
|
||||
else if (c == 308) asm_opt->extract_iter = atoi(opt.arg);
|
||||
else if (c == 309)
|
||||
{
|
||||
asm_opt->hom_global_coverage = atoi(opt.arg);
|
||||
else if (c == 309) {
|
||||
asm_opt->hom_global_coverage = asm_opt->hmo_cov_ss = atoi(opt.arg);
|
||||
asm_opt->hom_global_coverage_set = 1;
|
||||
if(asm_opt->hmo_cov_ss <= 0) {
|
||||
fprintf(stderr, "[ERROR] homozygous read coverage should be > 0 (--hom-cov)");
|
||||
return 1;
|
||||
}
|
||||
else if (c == 310)
|
||||
} else if (c == 310)
|
||||
{
|
||||
char* s = NULL;
|
||||
asm_opt->recover_atg_cov_min = strtol(opt.arg, &s, 10);
|
||||
@@ -801,7 +1055,73 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
else if (c == 341) asm_opt->bin_only = 1;
|
||||
else if (c == 342) asm_opt->ul_clean_round = atol(opt.arg);
|
||||
else if (c == 343) asm_opt->prt_dbg_gfa = 1;
|
||||
else if (c == 'l') { ///0: disable purge_dup; 1: purge containment; 2: purge overlap
|
||||
else if (c == 344) asm_opt->integer_correct_round = atol(opt.arg);
|
||||
else if (c == 345) asm_opt->dbg_ovec_cal = 1;
|
||||
else if (c == 346) asm_opt->max_path_drop_rate = atof(opt.arg);
|
||||
else if (c == 347) asm_opt->min_path_drop_rate = atof(opt.arg);
|
||||
else if (c == 348) asm_opt->trio_cov_het_ovlp = 1;
|
||||
else if (c == 349) asm_opt->ul_min_base = atol(opt.arg);
|
||||
else if (c == 350) asm_opt->self_scaf = 1;
|
||||
else if (c == 351) asm_opt->self_scaf_gap_max = atol(opt.arg);
|
||||
else if (c == 352) get_hic_enzymes(opt.arg, &(asm_opt->sec_in), 0);
|
||||
else if (c == 353) asm_opt->somatic_cov = atol(opt.arg);
|
||||
else if (c == 354) asm_opt->telo_motif = opt.arg;
|
||||
else if (c == 355) asm_opt->telo_pen = atol(opt.arg);
|
||||
else if (c == 356) asm_opt->telo_drop = atol(opt.arg);
|
||||
else if (c == 357) asm_opt->telo_mic_sc = atol(opt.arg);
|
||||
else if (c == 358) asm_opt->max_contig_tip = atol(opt.arg);
|
||||
else if (c == 359) {
|
||||
asm_opt->is_ont = 1; asm_opt->max_ov_diff_ec = 0.07; asm_opt->is_sc = 1; ///asm_opt->mz_win = 37; asm_opt->k_mer_length = 37;
|
||||
} /**else if (c == 360) {
|
||||
asm_opt->is_sc = 1;
|
||||
}**/ else if (c == 361) {
|
||||
asm_opt->chemical_cov = atol(opt.arg);
|
||||
} else if (c == 362) {
|
||||
asm_opt->chemical_flank = atol(opt.arg);
|
||||
///pending for integration
|
||||
/**
|
||||
} else if (c == 363) {
|
||||
asm_opt->ul_mod = atol(opt.arg);
|
||||
**/
|
||||
} else if (c == 364) {
|
||||
asm_opt->rl_cut = atol(opt.arg);
|
||||
} else if (c == 365) {
|
||||
asm_opt->sc_cut = atol(opt.arg);
|
||||
} else if (c == 366) {
|
||||
get_hic_enzymes(opt.arg, &(asm_opt->hf), 0);
|
||||
} else if (c == 367) {
|
||||
asm_opt->fn_chr_bin = opt.arg;
|
||||
} else if (c == 368) {
|
||||
asm_opt->gpath = 1;
|
||||
} else if (c == 369) {
|
||||
asm_opt->het_cov_set = asm_opt->het_cov_ss = atoi(opt.arg);
|
||||
if(asm_opt->het_cov_ss <= 0) {
|
||||
fprintf(stderr, "[ERROR] heterozygous read coverage should be > 0 (--het-cov)");
|
||||
return 1;
|
||||
}
|
||||
} else if (c == 370) {
|
||||
asm_opt->restart = 1;
|
||||
} else if (c == 371) {
|
||||
asm_opt->hf_cutoff = atoi(opt.arg);
|
||||
} else if (c == 372) {
|
||||
asm_opt->chn_occ = atoi(opt.arg);
|
||||
if(asm_opt->chn_occ <= 0) {
|
||||
fprintf(stderr, "[ERROR] chain cutoff should be > 0 (--chn-occ)");
|
||||
return 1;
|
||||
}
|
||||
} else if (c == 373) {
|
||||
asm_opt->dbg_run_1 = opt.arg;
|
||||
} else if (c == 374) {
|
||||
asm_opt->dbg_run_2 = opt.arg;
|
||||
} else if (c == 375) {
|
||||
asm_opt->ec_only = 1;
|
||||
} else if (c == 376) {
|
||||
asm_opt->hyb_syn = atoi(opt.arg);
|
||||
} else if (c == 377) {
|
||||
asm_opt->simd_mm = atoi(opt.arg);
|
||||
} else if (c == 378) {
|
||||
asm_opt->del_hf = 1;
|
||||
} else if (c == 'l') { ///0: disable purge_dup; 1: purge containment; 2: purge overlap
|
||||
asm_opt->purge_level_primary = asm_opt->purge_level_trio = atoi(opt.arg);
|
||||
}
|
||||
else if (c == 's') asm_opt->purge_simi_rate_l2 = asm_opt->purge_simi_rate_l3 = atof(opt.arg);
|
||||
@@ -829,5 +1149,18 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
|
||||
get_queries(argc, argv, &opt, asm_opt);
|
||||
|
||||
c = ((asm_opt->ar)?(asm_opt->ul_pst_join):(asm_opt->hifi_pst_join));
|
||||
if(c) {
|
||||
if((asm_opt->flag&HA_F_BAN_POST_JOIN)) asm_opt->flag-=HA_F_BAN_POST_JOIN;
|
||||
} else {
|
||||
asm_opt->flag |= HA_F_BAN_POST_JOIN;
|
||||
}
|
||||
|
||||
// fprintf(stderr, "[M::%s::] post join::%u\n", __func__, (uint32_t)(!(asm_opt->flag & HA_F_BAN_POST_JOIN)));
|
||||
// exit(1);
|
||||
if(!(asm_opt->is_ont)) {
|
||||
asm_opt->rl_cut = -1; asm_opt->sc_cut = 1;
|
||||
}
|
||||
|
||||
return check_option(asm_opt);
|
||||
}
|
||||
|
||||
+66
-2
@@ -5,7 +5,7 @@
|
||||
#include <pthread.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#define HA_VERSION "0.18.6-r513"
|
||||
#define HA_VERSION "0.25.1-r910"
|
||||
|
||||
#define VERBOSE 0
|
||||
|
||||
@@ -41,10 +41,13 @@ typedef struct {
|
||||
char *fn_bin_yak[2];
|
||||
char *fn_bin_list[2];
|
||||
char *fn_bin_poy;
|
||||
char *fn_chr_bin;
|
||||
char *extract_list;
|
||||
enzyme *hic_reads[2];
|
||||
enzyme *hic_enzymes;
|
||||
enzyme *ar;
|
||||
enzyme *hf;
|
||||
enzyme *sec_in;
|
||||
int extract_iter;
|
||||
int thread_num;
|
||||
int k_mer_length;
|
||||
@@ -62,6 +65,7 @@ typedef struct {
|
||||
int max_kmer_cnt;
|
||||
double high_factor; // coverage cutoff set to high_factor*hom_cov
|
||||
double max_ov_diff_ec;
|
||||
double max_ov_diff_ec_sec;
|
||||
double max_ov_diff_final;
|
||||
int hom_cov;
|
||||
int het_cov;
|
||||
@@ -73,6 +77,7 @@ typedef struct {
|
||||
int load_index_from_disk;
|
||||
int write_index_to_disk;
|
||||
int number_of_round;
|
||||
int number_of_pround;
|
||||
int adapterLen;
|
||||
int clean_round;
|
||||
int roundID;
|
||||
@@ -82,6 +87,7 @@ typedef struct {
|
||||
int min_overlap_coverage;
|
||||
int max_short_tip;
|
||||
int max_short_ul_tip;
|
||||
int max_contig_tip;
|
||||
int min_cnt;
|
||||
int mid_cnt;
|
||||
int purge_level_primary;
|
||||
@@ -104,6 +110,9 @@ typedef struct {
|
||||
float purge_simi_thres;
|
||||
float trans_base_rate;
|
||||
float trans_base_rate_sec;
|
||||
float min_path_drop_rate;
|
||||
float max_path_drop_rate;
|
||||
// uint64_t path_clean_round;
|
||||
|
||||
///float purge_simi_rate_hic;
|
||||
|
||||
@@ -128,9 +137,10 @@ typedef struct {
|
||||
float dp_e;
|
||||
int64_t hg_size;
|
||||
float kpt_rate;
|
||||
int64_t infor_cov, s_hap_cov;
|
||||
int64_t infor_cov, s_hap_cov, trio_cov_het_ovlp;
|
||||
double ul_error_rate, ul_error_rate_low, ul_error_rate_hpc;
|
||||
int32_t ul_ec_round;
|
||||
int32_t ul_mod;
|
||||
uint8_t is_dbg_het_cnt;
|
||||
uint8_t is_low_het_ul;
|
||||
uint8_t is_base_trans;
|
||||
@@ -138,8 +148,62 @@ typedef struct {
|
||||
uint8_t is_topo_trans;
|
||||
uint8_t is_bub_trans;
|
||||
uint8_t bin_only;
|
||||
uint8_t ec_only;
|
||||
int32_t ul_clean_round;
|
||||
int32_t prt_dbg_gfa;
|
||||
int32_t integer_correct_round;
|
||||
uint8_t dbg_ovec_cal;
|
||||
uint8_t hifi_pst_join, ul_pst_join;
|
||||
uint32_t ul_min_base;
|
||||
uint8_t self_scaf;
|
||||
uint64_t self_scaf_min;
|
||||
uint64_t self_scaf_reliable_min;
|
||||
int64_t self_scaf_gap_max;
|
||||
int64_t somatic_cov;
|
||||
|
||||
char *telo_motif;
|
||||
int64_t telo_pen;
|
||||
int64_t telo_drop;
|
||||
int64_t telo_mic_sc;
|
||||
|
||||
uint64_t is_ont;
|
||||
uint64_t is_sc;
|
||||
int64_t chemical_cov;
|
||||
uint64_t chemical_flank;
|
||||
|
||||
int64_t rl_cut;
|
||||
int64_t sc_cut;
|
||||
uint8_t gpath;
|
||||
|
||||
uint64_t hf_rate;///cannot be larger than 128?
|
||||
uint64_t hf_rate_max;///cannot be larger than 128?
|
||||
uint64_t ont_rate;///cannot be 0, should be 1 in anyway
|
||||
|
||||
int64_t het_cov_set;
|
||||
int64_t restart;
|
||||
|
||||
int64_t hf_cutoff;
|
||||
|
||||
uint8_t write_pos_idx;
|
||||
|
||||
int hom_cov_0;
|
||||
int het_cov_0;
|
||||
int max_n_chain_0; // fall-back max number of chains to consider
|
||||
|
||||
int64_t hmo_cov_ss;
|
||||
int64_t het_cov_ss;
|
||||
int64_t chn_occ;
|
||||
|
||||
char *dbg_run_1, *dbg_run_2;
|
||||
|
||||
int32_t hyb_syn;
|
||||
|
||||
int64_t step_rd;
|
||||
|
||||
uint8_t dbg_bam;
|
||||
|
||||
int8_t simd_mm;
|
||||
int8_t del_hf;
|
||||
} hifiasm_opt_t;
|
||||
|
||||
extern hifiasm_opt_t asm_opt;
|
||||
|
||||
+17613
-217
File diff suppressed because it is too large
Load Diff
@@ -144,7 +144,10 @@ typedef struct
|
||||
char misBase;
|
||||
}haplotype_evdience;
|
||||
|
||||
|
||||
#define hh_tp(z) (((z).type&1))
|
||||
#define hh_hp(z) ((((z).type>>1)&1))
|
||||
#define hh_bq(z) ((((z).type>>2))&sc_bm)
|
||||
#define hh_wq(z) ((((z).type>>(sc_bn+2)))&sc_bm)
|
||||
|
||||
typedef struct
|
||||
{
|
||||
@@ -1148,7 +1151,7 @@ void correct_ul_overlap(overlap_region_alloc* overlap_list, const ul_idx_t *uref
|
||||
void ul_lalign(overlap_region_alloc* ol, Candidates_list *cl, const ul_idx_t *uref, const ug_opt_t *uopt, char *qstr,
|
||||
uint64_t ql, UC_Read* qu, UC_Read* tu, Correct_dumy* dumy, bit_extz_t *exz,
|
||||
haplotype_evdience_alloc* hap, kvec_t_u64_warp* v_idx, overlap_region *aux_o,
|
||||
double e_rate, int64_t wl, kv_ul_ov_t *aln, int64_t sid, uint64_t hpc_k, st_mt_t *stb, void *km);
|
||||
double e_rate, int64_t wl, kv_ul_ov_t *aln, int64_t sid, uint64_t hpc_k, st_mt_t *stb, idx_emask_t *mm, mask_ul_ov_t *mk, void *km);
|
||||
|
||||
void ul_lalign_old_ed(overlap_region_alloc* ol, Candidates_list *cl, const ul_idx_t *uref, char *qstr,
|
||||
uint64_t ql, UC_Read* qu, UC_Read* tu, Correct_dumy* dumy,
|
||||
@@ -1347,7 +1350,7 @@ All_reads *rref, UC_Read* tu, asg64_v* idx, asg64_v *b0, asg64_v *b1, int64_t ql
|
||||
kv_ul_ov_t *aln, uint64_t rid, int64_t max_lgap, double sgap_rate);
|
||||
|
||||
int64_t infer_rovlp(ul_ov_t *li, ul_ov_t *lj, uc_block_t *bi, uc_block_t *bj, All_reads *ridx, ma_ug_t *ug);
|
||||
void convert_ul_ov_t(ul_ov_t *des, overlap_region *src, const ul_idx_t *uref);
|
||||
void convert_ul_ov_t(ul_ov_t *des, overlap_region *src, ma_ug_t *ug);
|
||||
uint64_t check_connect_ug(const ul_idx_t *uref, uint32_t v, uint32_t w, int64_t bw, double diff_ec_ul, int64_t dq);
|
||||
uint64_t check_connect_rg(const ul_idx_t *uref, const ug_opt_t *uopt, uint32_t uv, uint32_t uw, int64_t bw, double diff_ec_ul, int64_t dq);
|
||||
uint32_t govlp_check(const ul_idx_t *uref, const ug_opt_t *uopt, int64_t bw, double diff_ec_ul, ul_ov_t *li, ul_ov_t *lj);
|
||||
@@ -1384,8 +1387,119 @@ typedef struct {
|
||||
int64_t k, q[2], t[2], cq[2], ct[2], ci[2], werr, werr0, cerr;
|
||||
int64_t qoff, f, toff, coff, cur_qoff;
|
||||
} rtrace_iter;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
overlap_region_alloc *ol;
|
||||
Candidates_list *cl;
|
||||
All_reads *rref;
|
||||
UC_Read *qu;
|
||||
UC_Read *tu;
|
||||
bit_extz_t *exz;
|
||||
overlap_region *aux_o, *rse_o;
|
||||
double e_rate[2];
|
||||
int64_t wl[2];
|
||||
int64_t rid;
|
||||
int64_t khit;
|
||||
int64_t move_gap;
|
||||
asg16_v *buf;
|
||||
asg64_v *srt;
|
||||
asg8_v *hpz;
|
||||
ma_hit_t_alloc *in;
|
||||
|
||||
int8_t chem_drop[2];
|
||||
double align_gap_rate[2];
|
||||
int64_t align_gap_max[2];
|
||||
|
||||
uint64_t sec_aln_win;
|
||||
uint64_t sec_aln_cov;
|
||||
double sec_aln_err_rate;
|
||||
double sec_aln_max;
|
||||
asg64_v *kp;
|
||||
|
||||
asg32_v *v32;
|
||||
asg64_v *bp;
|
||||
ha_abuf_t *ab;
|
||||
uint64_t max_n_chain;
|
||||
uint64_t max_n_chain_f;
|
||||
uint64_t chain_cutoff;
|
||||
uint64_t ave_cov_min;
|
||||
uint64_t ocw;
|
||||
|
||||
uint64_t t_cut;
|
||||
|
||||
uint64_t hom_cov_a;
|
||||
} gen_hc_aln_t;
|
||||
int64_t get_rid_backward_cigar_err(rtrace_iter *it, ul_ov_t *aln, kv_rtrace_t *trace, rtrace_t *tc,
|
||||
const ul_idx_t *uref, char* qstr, UC_Read *tu, overlap_region_alloc *ol, overlap_region *o,
|
||||
bit_extz_t *exz, double e_rate, int64_t qs);
|
||||
|
||||
uint64_t gen_hc_r_alin_adp_smp_ff_ec(overlap_region_alloc* ol, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, int64_t rid, asg64_v *sp, uint64_t ocw, uint32_t *ocn, uint32_t *osc, uint64_t max_n_chain, uint64_t max_n_chain_f, uint64_t chain_cutoff, uint64_t ave_cov_min);
|
||||
uint64_t gen_hc_r_alin_adp_smp(overlap_region_alloc* ol, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v* buf, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max, uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max,
|
||||
asg64_v *sp, uint64_t ocw, uint8_t *hpf, uint32_t *a_cu, uint32_t *a_ci, uint32_t *ocn, uint32_t *osc, uint64_t *idx_cu, uint64_t n_cu, asg64_v *bp, uint64_t max_n_chain, uint64_t max_n_chain_f, uint64_t chain_cutoff, uint64_t ave_cov_min, uint8_t set_match, uint8_t is_dedup);
|
||||
uint64_t gen_hc_r_alin_adp_mmp_1(overlap_region_alloc* ol, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v* buf, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max, uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max,
|
||||
asg64_v *sp, uint64_t ocw, uint8_t *hpf, asg32_v *v32, asg64_v *bp, uint64_t max_n_chain, uint64_t max_n_chain_f, uint64_t chain_cutoff, uint64_t ave_cov_min, uint8_t set_match);
|
||||
uint64_t gen_hc_r_alin_adp_mmp_0(overlap_region_alloc* ol, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v* buf, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max, uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max,
|
||||
asg64_v *sp, uint64_t ocw, uint8_t *hpf, asg32_v *v32, asg64_v *bp, uint64_t max_n_chain, uint64_t max_n_chain_f, uint64_t chain_cutoff, uint64_t ave_cov_min, uint8_t set_match);
|
||||
uint64_t gen_gc_r_alin_adp_mmp_0(overlap_region_alloc* ol, Candidates_list *cl, ul_idx_t *uref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v* buf, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max, uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max,
|
||||
asg64_v *sp, uint64_t ocw, uint8_t *hpf, asg32_v *v32, asg64_v *bp, uint64_t max_n_chain, uint64_t max_n_chain_f, uint64_t chain_cutoff, uint64_t ave_cov_min, uint8_t set_match);
|
||||
void gen_hc_r_alin_adv_adp_smp(gen_hc_aln_t *ez, uint32_t *a_cu, uint32_t *a_ci, uint32_t *ocn, uint32_t *osc, uint64_t *idx_cu, uint64_t n_cu, uint8_t set_match);
|
||||
void gen_hc_r_alin_adv_adp_smp_0(gen_hc_aln_t *ez, uint8_t set_match);
|
||||
void gen_hc_r_alin_adv_adp_smp_1(gen_hc_aln_t *ez, uint8_t set_match);
|
||||
void pp_chn_a(overlap_region *z, Candidates_list *cl, uint8_t is_raw);
|
||||
uint64_t gen_hc_r_alin(overlap_region_alloc* ol, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v* buf, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max, uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max, asg64_v *kp, uint8_t *hpf);
|
||||
void gen_hc_r_alin_flt(overlap_region_alloc* ol, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o, overlap_region *aux_b, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v* buf, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max, uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max,
|
||||
asg64_v *kp, asg64_v *sp, uint64_t ocw, uint8_t *hpf, uint32_t *a_cu, uint32_t *a_ci, uint32_t *ocn, uint32_t *osc, uint64_t *idx_cu, uint64_t n_cu, asg64_v *bp, uint64_t max_n_chain, uint64_t max_n_chain_f, uint64_t chain_cutoff, uint64_t ave_cov_min);
|
||||
void gen_hc_r_alin_adp(overlap_region_alloc* ol, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o, overlap_region *aux_b, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v* buf, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max, uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max,
|
||||
asg64_v *kp, asg64_v *sp, uint64_t ocw, uint8_t *hpf, uint32_t *a_cu, uint32_t *a_ci, uint32_t *ocn, uint32_t *osc, uint64_t *idx_cu, uint64_t n_cu, asg64_v *bp, uint64_t max_n_chain, uint64_t max_n_chain_f, uint64_t chain_cutoff, uint64_t ave_cov_min);
|
||||
void gen_hc_r_alin_adv(gen_hc_aln_t *ez);
|
||||
uint64_t gen_hc_r_alin_nec(overlap_region_alloc* ol, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v* buf, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max, uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max, asg64_v *kp, uint8_t *hpf);
|
||||
void gen_hc_r_alin_nec_adv(gen_hc_aln_t *ez);
|
||||
uint64_t gen_hc_r_alin_re(overlap_region* z, Candidates_list *cl, char* qstr, uint64_t ql, char* tstr, uint64_t tl, bit_extz_t *exz, overlap_region *aux_o, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v* buf);
|
||||
void rphase_hc(overlap_region_alloc* ol, All_reads *rref, haplotype_evdience_alloc* hp, UC_Read* qu, UC_Read* tu, kv_ul_ov_t *c_idx, asg64_v* idx, asg64_v* buf, int64_t bd, int64_t wl, int64_t ql, uint8_t occ_thres/**, uint8_t is_dbg**/, uint64_t rid, uint64_t hpc_len, uint64_t std_bs, Chain_Data *dp, asg8_v *q8, asg8_v *t8, uint8_t lindel, uint64_t tcut, uint64_t site_sc, int64_t h0_w, asg32_v *b32,
|
||||
int64_t hap_cov_match, int64_t hap_cov_unmatch, int64_t het_cov_a, int64_t hom_cov_a, int64_t n_hap, double hf_rate);
|
||||
void set_exact_exz(bit_extz_t *exz, int64_t qs, int64_t qe, int64_t ts, int64_t te);
|
||||
void push_alnw(overlap_region *aux_o, bit_extz_t *exz);
|
||||
void cal_exz_global(char *pstr, int32_t pn, char *tstr, int32_t tn, int32_t thre, bit_extz_t *ez);
|
||||
void get_wqual(uint64_t zid, uint64_t zpos, uint64_t zrev, asg8_v *v, uint8_t *va, uint64_t scw, uint64_t *tqual, uint64_t *wqual);
|
||||
void gen_reseed_re(overlap_region_alloc *ol, Candidates_list *cl, overlap_region *aux_o, overlap_region *rse_o, All_reads *rref, UC_Read* qu, UC_Read *tu, bit_extz_t *exz, kv_ul_ov_t *c_idx, asg64_v *idx, asg64_v *res, int64_t bd, int64_t mzw, int64_t kl, int64_t rid, double err_h, double err_l, asg16_v *b16, uint64_t tqn, uint8_t *hpf);
|
||||
inline uint64_t exact_ec_check(char *qstr, uint64_t ql, char *tstr, uint64_t tl, int64_t qs, int64_t qe, int64_t ts, int64_t te)
|
||||
{
|
||||
if(qe - qs != te - ts) return 0;
|
||||
if(memcmp(qstr + qs, tstr + ts, qe - qs) == 0) return 1;
|
||||
return 0;
|
||||
}
|
||||
// void est_rep_err_rate(overlap_region_alloc* ol, asg64_v *ix, kv_ul_ov_t *c_idx, int64_t ql, int64_t wl, uint64_t *ou_a, uint64_t min_dp, int64_t ph_cov, uint8_t flg_ov, double flg_ov_sec_rate, double flg_cov_rate, uint64_t *ave_e, uint64_t *bd_e, uint64_t *tot_cov);
|
||||
void est_rep_err_rate(overlap_region_alloc* ol, asg64_v *ix, kv_ul_ov_t *c_idx, int64_t ql, int64_t wl, uint64_t *ou_a, uint64_t min_dp, int64_t ph_cov, uint8_t flg_ov, double flg_ov_sec_rate, double flg_cov_rate, uint64_t *ave_e, uint64_t *bd_e, uint64_t *tot_cov);
|
||||
#define ovlp_id(x) ((x).tn)
|
||||
#define ovlp_min_wid(x) ((x).ts)
|
||||
#define ovlp_max_wid(x) ((x).te)
|
||||
#define ovlp_cur_wid(x) ((x).qn)
|
||||
#define ovlp_cur_xoff(x) ((x).qs)
|
||||
#define ovlp_cur_yoff(x) ((x).ts)
|
||||
#define ovlp_cur_ylen(x) ((x).te)
|
||||
#define ovlp_cur_coff(x) ((x).qe)
|
||||
#define ovlp_bd(x) ((x).sec)
|
||||
#define ovlp_um(x) ((x).sec)
|
||||
#define ovlp_hf(x) ((x).el)
|
||||
|
||||
#define HPC_PL 12
|
||||
#define HPC_RR 4
|
||||
#define HPC_CC 2
|
||||
#define HPC_RR_Q 5
|
||||
#define HPC_CC_Q 3
|
||||
#define HC_MF_R 0.5
|
||||
#define HC_AV_MIN 0.7
|
||||
#define GC_MF_N 12
|
||||
|
||||
// #define FORCE_CUT 1
|
||||
|
||||
// overlap_region_alloc* ol, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o,
|
||||
|
||||
// double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v *buf, asg64_v *srt, ma_hit_t_alloc *in, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max,
|
||||
// uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max, asg64_v *kp
|
||||
|
||||
// uint64_t gen_hc_r_alin_ea_hybrid(overlap_region_alloc* ol, uint64_t bi, uint64_t bn, uint64_t tk, Candidates_list *cl, All_reads *rref, UC_Read* qu, UC_Read* tu, bit_extz_t *exz, overlap_region *aux_o, double e_rate, int64_t wl, int64_t rid, int64_t khit, int64_t move_gap, asg16_v *buf, uint8_t ec_filter, ma_hit_t_alloc *in, uint8_t chem_drop, double align_gap_rate, int64_t align_gap_max,
|
||||
// uint64_t sec_aln_win, uint64_t sec_aln_cov, double sec_aln_err_rate, double sec_aln_max, asg64_v *kp)
|
||||
|
||||
#endif
|
||||
|
||||
+358
-12
@@ -1512,6 +1512,34 @@ inline int32_t comput_sc_ch(const k_mer_hit *ai, const k_mer_hit *aj, double bw_
|
||||
return sc;
|
||||
}
|
||||
|
||||
inline int32_t comput_sc_ch_ec(const k_mer_hit *ai, const k_mer_hit *aj, double bw_rate, double chn_pen_gap, double chn_pen_skip, int64_t sl, int64_t ol)
|
||||
{
|
||||
///ai is the suffix of aj
|
||||
int32_t dq, dr, dd, dg, q_span, sc;
|
||||
dq = (int64_t)(ai->self_offset) - (int64_t)(aj->self_offset);
|
||||
if(dq <= 0) return INT32_MIN;
|
||||
dr = (int64_t)(ai->offset) - (int64_t)(aj->offset);
|
||||
if(dr <= 0) return INT32_MIN;
|
||||
dd = dr > dq? dr - dq : dq - dr;//gap
|
||||
if((dd > 16) && (dd > cal_bw(ai, aj, bw_rate, sl, ol))) return INT32_MIN;
|
||||
dg = dr < dq? dr : dq;//len
|
||||
q_span = ai->cnt&(0xffu);
|
||||
sc = q_span < dg? q_span : dg;
|
||||
sc = normal_w(sc, ((int32_t)(ai->cnt>>8)));
|
||||
if (dd || (dg > q_span && dg > 0)) {
|
||||
double lin_pen, a_pen;
|
||||
lin_pen = (chn_pen_gap*(double)dd);
|
||||
a_pen = ((double)(sc))*((((double)dd)/((double)dg))/bw_rate);
|
||||
///for long gap
|
||||
// if(lin_pen > a_pen) lin_pen = a_pen;
|
||||
if(dd < 4) lin_pen = ((lin_pen > a_pen)?(a_pen):(lin_pen));
|
||||
else lin_pen = ((lin_pen < a_pen)?(a_pen):(lin_pen));
|
||||
lin_pen += (chn_pen_skip*(double)dg);
|
||||
sc -= (int32_t)lin_pen;
|
||||
}
|
||||
return sc;
|
||||
}
|
||||
|
||||
inline int32_t comput_sc_ff(const k_mer_hit *ai, const k_mer_hit *aj, double bw_rate, double chn_pen_gap, double chn_pen_skip, int64_t sl, int64_t ol)
|
||||
{
|
||||
///ai is the suffix of aj
|
||||
@@ -1721,18 +1749,7 @@ uint64_t lchain_qdp(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp, o
|
||||
return cL;
|
||||
}
|
||||
|
||||
|
||||
#define kv_pushp_ol(type, v, p) do { \
|
||||
if ((v).length == (v).size) { \
|
||||
(v).list = (type*)realloc((v).list, sizeof(type)*((v).size?((v).size<<1):(2))); \
|
||||
memset((v).list+(v).size, 0, sizeof(overlap_region)*(((v).size?((v).size<<1):2)-(v).size));\
|
||||
(v).size = (v).size?((v).size<<1):(2); \
|
||||
} \
|
||||
*(p) = &((v).list[(v).length++]); \
|
||||
} while (0)
|
||||
|
||||
void push_ovlp_chain_qgen(overlap_region* o, uint32_t xid, int64_t xl, int64_t yl, int64_t sc,
|
||||
k_mer_hit *beg, k_mer_hit *end)
|
||||
void push_ovlp_chain_qgen(overlap_region* o, uint32_t xid, int64_t xl, int64_t yl, int64_t sc, k_mer_hit *beg, k_mer_hit *end)
|
||||
{
|
||||
int64_t xr, yr;
|
||||
o->x_id = xid; o->y_id = beg->readID;
|
||||
@@ -1986,6 +2003,284 @@ uint64_t lchain_qdp_mcopy(Candidates_list *cl, int64_t a_idx, int64_t a_n, int64
|
||||
return cL;
|
||||
}
|
||||
|
||||
void quick_ck_lchain(k_mer_hit* a, int64_t a_n, int64_t xl, int64_t yl, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
||||
int64_t *p, int64_t *t, int32_t *f, int32_t *ii, int64_t *plus, int64_t *msc, int64_t *msc_i, int64_t *movl, int64_t *si, int64_t *ei)
|
||||
{
|
||||
if(a_n <= 0) return;
|
||||
int64_t l, k, is_srt = 1, z; k_mer_hit *ai, *aj;
|
||||
int64_t dq, dr, dd, dg, q_span, sc, csc, ddt;
|
||||
int64_t plus0, msc0, msc_i0, movl0; double lin_pen, a_pen;
|
||||
|
||||
*plus = 0; *msc = *msc_i = INT32_MIN; *movl = INT32_MAX; *si = 0; *ei = a_n;
|
||||
|
||||
for (k = 1, l = 0; k <= a_n; k++) {
|
||||
if(k == a_n || a[k].strand != a[l].strand) {
|
||||
t[k-1] = 0; ii[k-1] = 0;
|
||||
// if(a_n && a[0].readID == 3125488) {
|
||||
// fprintf(stderr, "[M::%s::] ii::[%ld,%ld)(%c), is_srt::%ld, chn_pen_gap::%f, chn_pen_skip::%f, bw_rate::%f\n", __func__, l, k, "+-"[a[l].strand], is_srt, chn_pen_gap, chn_pen_skip, bw_rate);
|
||||
// }
|
||||
if(is_srt) {
|
||||
plus0 = 0; msc0 = msc_i0 = INT32_MIN; movl0 = INT32_MAX; ddt = 0;
|
||||
|
||||
|
||||
p[l] = -1; f[l] = a[l].cnt&(0xffu);
|
||||
if(f[l] >= msc0) {msc0 = f[l]; msc_i0 = l;}///difference
|
||||
if(f[l] < plus0) plus0 = f[l];
|
||||
|
||||
|
||||
for (z = l + 1; z < k; z++) {
|
||||
///roughly same to comput_sc_ch(&a[z], &a[z-1])
|
||||
ai = &a[z]; aj = &a[z-1];
|
||||
dq = (int64_t)(ai->self_offset) - (int64_t)(aj->self_offset);
|
||||
if(dq <= 0) break;
|
||||
dr = (int64_t)(ai->offset) - (int64_t)(aj->offset);
|
||||
if(dr <= 0) break;
|
||||
dd = dr > dq? dr - dq : dq - dr;//gap
|
||||
// if(a_n && a[0].readID == 3125488) {
|
||||
// fprintf(stderr, "%ld,", dd);
|
||||
// }
|
||||
if((dd > 16) && (dd > cal_bw(&(a[z]), &(a[z-1]), bw_rate, xl, yl))) break;
|
||||
dg = dr < dq? dr : dq;//len
|
||||
q_span = ai->cnt&(0xffu);
|
||||
sc = q_span < dg? q_span : dg;
|
||||
sc = normal_w(sc, ((int32_t)(ai->cnt>>8)));
|
||||
if (dd || (dg > q_span && dg > 0)) {
|
||||
lin_pen = (chn_pen_gap*(double)dd);
|
||||
a_pen = ((double)(sc))*((((double)dd)/((double)dg))/bw_rate);
|
||||
///for long gap
|
||||
// if(lin_pen > a_pen) lin_pen = a_pen;
|
||||
if(dd < 4) lin_pen = ((lin_pen > a_pen)?(a_pen):(lin_pen));
|
||||
else lin_pen = ((lin_pen < a_pen)?(a_pen):(lin_pen));
|
||||
lin_pen += (chn_pen_skip*(double)dg);
|
||||
sc -= (int32_t)lin_pen;
|
||||
}
|
||||
|
||||
sc += f[z-1]; csc = a[z].cnt&(0xffu); if(sc < csc) break;
|
||||
p[z] = z - 1; f[z] = sc; ddt += dd;
|
||||
|
||||
if(f[z] >= msc0) {msc0 = f[z]; msc_i0 = z;}///difference
|
||||
if(f[z] < plus0) plus0 = f[z];
|
||||
}
|
||||
|
||||
// if(a_n && a[0].readID == 3125488) {
|
||||
// fprintf(stderr, "\n");
|
||||
// fprintf(stderr, "[M::%s::] msc0::%ld, msc_i0::%ld, (%c)\n", __func__, msc0, msc_i0, "+-"[a[l].strand]);
|
||||
// }
|
||||
if((z >= k) && (msc_i0 == (k - 1))) {
|
||||
if((k - l >= 2) && (ddt > 16) && (ddt > cal_bw(&(a[k-1]), &(a[l]), bw_rate, xl, yl))) msc_i0 = INT32_MIN;
|
||||
if(msc_i0 == (k - 1)) {
|
||||
if(msc0 >= (*msc)) {
|
||||
movl0 = get_chainLen(a[msc_i0].self_offset, a[msc_i0].self_offset, xl, a[msc_i0].offset, a[msc_i0].offset, yl);
|
||||
if(msc0 > (*msc) || movl0 < (*movl)) {
|
||||
*msc = msc0; *msc_i = msc_i0; *movl = movl0;
|
||||
}
|
||||
}
|
||||
if(plus0 < (*plus)) *plus = plus0;
|
||||
if((*ei) > k) {
|
||||
(*si) = k;
|
||||
} else {
|
||||
(*ei) = l;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
l = k; is_srt = 1;
|
||||
} else {
|
||||
if((a[k].self_offset <= a[k-1].self_offset) || (a[k].offset <= a[k-1].offset)) is_srt = 0;
|
||||
t[k-1] = 0; ii[k-1] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
uint64_t lchain_qdp_mcopy_fast(Candidates_list *cl, int64_t a_idx, int64_t a_n, int64_t des_idx,
|
||||
Chain_Data* dp, overlap_region_alloc* res, int64_t max_skip, int64_t max_iter,
|
||||
int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
||||
uint32_t xid, int64_t xl, int64_t yl, int64_t quick_check, uint32_t apend_be,
|
||||
int64_t gen_cigar, int64_t mcopy_num, double mcopy_rate, int64_t mcopy_khit_cutoff,
|
||||
int64_t khit_n)
|
||||
{
|
||||
if(a_n <= 0) return 0;
|
||||
int64_t *p, *t, max_f, n_skip, st, max_j, end_j, sc, msc, msc_i, max_ii, ovl, movl, plus = 0, min_sc, ch_n, si, ei;
|
||||
int32_t *f, max, tmp, *ii; int64_t i, k, j, cL = 0; k_mer_hit* a; k_mer_hit* des; k_mer_hit *swap; overlap_region *z;
|
||||
resize_Chain_Data(dp, a_n, NULL); ch_n = 1; // int64_t bw; bw = ((xl < yl)?xl:yl); bw *= bw_rate;
|
||||
t = dp->tmp; f = dp->score; p = dp->pre; ii = dp->occ;
|
||||
|
||||
a = cl->list + a_idx; des = cl->list + des_idx;
|
||||
// if(a_n && (a[0].readID == 27105 || a[0].readID == 7603)) {///r833
|
||||
// fprintf(stderr, "---[M::%s::rid->%u::%c]\ta_n::%ld\n",
|
||||
// __func__, a[0].readID, "+-"[a[0].strand], a_n);
|
||||
// }
|
||||
if(quick_check) {
|
||||
quick_ck_lchain(a, a_n, xl, yl, chn_pen_gap, chn_pen_skip, bw_rate, p, t, f, ii, &plus, &msc, &msc_i, &movl, &si, &ei);
|
||||
} else {
|
||||
msc = msc_i = INT32_MIN; movl = INT32_MAX; plus = 0; si = 0; ei = a_n;
|
||||
memset(t, 0, (a_n*sizeof((*t))));
|
||||
}
|
||||
// if(a_n && a[0].readID == 4412344) {
|
||||
// fprintf(stderr, "[M::%s::] si::%ld, ei::%ld, a_n::%ld\n", __func__, si, ei, a_n);
|
||||
// }
|
||||
for (i = st = si, max_ii = -1; i < ei; ++i) {
|
||||
max_f = a[i].cnt&(0xffu);
|
||||
n_skip = 0; max_j = end_j = -1;
|
||||
if ((i-st) > max_iter) st = i-max_iter;
|
||||
while (a[i].strand != a[st].strand) ++st;
|
||||
|
||||
for (j = i - 1; j >= st; --j) {
|
||||
sc = comput_sc_ch_ec(&a[i], &a[j], bw_rate, chn_pen_gap, chn_pen_skip, xl, yl);
|
||||
if (sc == INT32_MIN) continue;
|
||||
sc += f[j];
|
||||
if (sc > max_f) {
|
||||
max_f = sc, max_j = j;
|
||||
if (n_skip > 0) --n_skip;
|
||||
} else if (t[j] == (int32_t)i) {
|
||||
if (++n_skip > max_skip)
|
||||
break;
|
||||
}
|
||||
if (p[j] >= 0) t[p[j]] = i;
|
||||
}
|
||||
end_j = j;
|
||||
|
||||
if ((max_ii<0) || (a[i].self_offset>a[max_ii].self_offset+max_dis) || (a[i].strand!=a[max_ii].strand)) {
|
||||
max = INT32_MIN; max_ii = -1;
|
||||
for (j=i-1; (j>=st) && (a[i].self_offset<=max_dis+a[j].self_offset)&&(a[i].strand==a[j].strand); --j) {
|
||||
if (max < f[j]) {
|
||||
max = f[j], max_ii = j;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if ((max_ii >= 0) && (max_ii < end_j) && (a[i].strand == a[max_ii].strand)) {///just have a try with a[i]<->a[max_ii]
|
||||
tmp = comput_sc_ch_ec(&a[i], &a[max_ii], bw_rate, chn_pen_gap, chn_pen_skip, xl, yl);
|
||||
if (tmp != INT32_MIN && max_f < tmp + f[max_ii])
|
||||
max_f = tmp + f[max_ii], max_j = max_ii;
|
||||
}
|
||||
f[i] = max_f; p[i] = max_j;
|
||||
if ((max_ii < 0) || ((a[i].self_offset<=max_dis+a[max_ii].self_offset)&&(a[i].strand==a[max_ii].strand)&&(f[max_ii]<f[i]))) {
|
||||
max_ii = i;
|
||||
}
|
||||
if(f[i] >= msc) {
|
||||
ovl = get_chainLen(a[i].self_offset, a[i].self_offset, xl, a[i].offset, a[i].offset, yl);
|
||||
if(f[i] > msc || ovl < movl) {
|
||||
msc = f[i]; msc_i = i; movl = ovl;
|
||||
}
|
||||
}
|
||||
if(f[i] < plus) plus = f[i];
|
||||
ii[i] = 0;///for mcopy, not here
|
||||
// if(a_n && (a[0].readID == 27105 || a[0].readID == 7603)) {///r833
|
||||
// fprintf(stderr, "i::%ld[M::%s::rid->%u::%c] q::%u, t::%u, st::%ld, max_ii::%ld, f[i]::%d, p[i]::%ld, msc_i::%ld, msc::%ld, movl::%ld\n",
|
||||
// i, __func__, a[i].readID, "+-"[a[i].strand],
|
||||
// a[i].self_offset, a[i].offset, st, max_ii, f[i], p[i], msc_i, msc, movl);
|
||||
// }
|
||||
}
|
||||
|
||||
for (i = msc_i, cL = 0; i >= 0; i = p[i]) { ii[i] = 1; t[cL++] = i;}///label the best chain
|
||||
|
||||
if(mcopy_num > 1) {
|
||||
// if(a[0].readID == 4412344) {
|
||||
// fprintf(stderr, "[M::%s::] msc::%ld, cL::%ld\n", __func__, msc, cL);
|
||||
// }
|
||||
if(cL >= mcopy_khit_cutoff) {///if there are too few k-mers, disable mcopy
|
||||
msc -= plus; min_sc = msc*mcopy_rate/**0.2**/; ii[msc_i] = 0;
|
||||
for (i = ch_n = 0; i < a_n; ++i) {///make all f[] positive
|
||||
f[i] -= plus; if(i >= ch_n) t[i] = 0;
|
||||
if((!(ii[i])) && (f[i] >= min_sc)) {///!(ii[i]): skip the best chain
|
||||
t[ch_n] = ((uint64_t)f[i])<<32; t[ch_n] += (i<<1); ch_n++;
|
||||
}
|
||||
}
|
||||
// if(a[0].readID == 4412344) {
|
||||
// fprintf(stderr, "[M::%s::] msc::%ld, min_sc::%ld, cL::%ld, ch_n::%ld, mcopy_num::%ld\n", __func__, msc, min_sc, cL, ch_n, mcopy_num);
|
||||
// }
|
||||
if(ch_n > 1) {
|
||||
int64_t n_v, n_v0, ni, n_u, n_u0 = res->length;
|
||||
radix_sort_hc64i(t, t + ch_n);
|
||||
for (k = ch_n-1, n_v = n_u = 0; k >= 0 && n_u < mcopy_num; --k) {
|
||||
n_v0 = n_v;
|
||||
for (i = ((uint32_t)t[k])>>1; i >= 0 && (t[i]&1) == 0; ) {
|
||||
ii[n_v++] = i; t[i] |= 1; i = p[i];
|
||||
}
|
||||
if(n_v0 == n_v) continue;
|
||||
sc = (i<0?(t[k]>>32):((t[k]>>32)-f[i]));
|
||||
// if(a[0].readID == 4412344) {
|
||||
// fprintf(stderr, "+[M::%s::] sc::%ld, n_a::%ld\n", __func__, sc, n_v-n_v0);
|
||||
// }
|
||||
if(sc >= min_sc) {
|
||||
kv_pushp_ol(overlap_region, (*res), &z);
|
||||
push_ovlp_chain_qgen(z, xid, xl, yl, sc+plus, &(a[ii[n_v-1]]), &(a[ii[n_v0]]));
|
||||
// if(a[0].readID == 4412344) {
|
||||
// fprintf(stderr, "-[M::%s::] sc::%ld, n_a::%ld, q::[%u,%u), t::[%u,%u), %c\n", __func__, sc, n_v-n_v0, z->x_pos_s, z->x_pos_e + 1, z->y_pos_s, z->y_pos_e + 1, "+-"[z->y_pos_strand]);
|
||||
// }
|
||||
///mcopy_khit_cutoff <= 1: disable the mcopy_khit_cutoff filtering, for the realignment
|
||||
// if((mcopy_khit_cutoff <= 1) || ((z->x_pos_e+1-z->x_pos_s) <= (movl<<2))) {
|
||||
if((!n_u) || (n_v - n_v0 > 1)) {
|
||||
z->align_length = n_v-n_v0; z->x_id = n_v0;
|
||||
n_u++;
|
||||
} else {///non-best is tiny
|
||||
res->length--; n_v = n_v0;
|
||||
}
|
||||
} else {
|
||||
n_v = n_v0;
|
||||
}
|
||||
}
|
||||
|
||||
// if(n_u > 1) ks_introsort_or_sss(n_u, res->list + n_u0);
|
||||
// res->length = n_u0 + filter_non_ovlp_xchains(res->list + n_u0, n_u, &n_v);
|
||||
n_u = res->length;
|
||||
if(n_u > n_u0 + 1) {
|
||||
kv_resize_cl(k_mer_hit, (*cl), (n_v+cl->length));
|
||||
a = cl->list + a_idx; des = cl->list + des_idx; swap = cl->list + cl->length;
|
||||
for (k = n_u0, i = n_v0 = n_v = 0; k < n_u; k++) {
|
||||
z = &(res->list[k]);
|
||||
z->non_homopolymer_errors = des_idx + i;
|
||||
n_v0 = z->x_id; ni = z->align_length;
|
||||
for (j = 0; j < ni; j++, i++) {
|
||||
///k0 + (ni - j - 1)
|
||||
swap[i] = a[ii[n_v0 + (ni- j - 1)]];
|
||||
swap[i].readID = k;
|
||||
}
|
||||
z->x_id = xid;
|
||||
if(gen_cigar) gen_fake_cigar(&(z->f_cigar), z, apend_be, swap+i-ni, ni);
|
||||
if(!khit_n) z->align_length = 0;
|
||||
}
|
||||
memcpy(des, swap, i*sizeof((*swap))); //assert(i == ch_n);
|
||||
|
||||
// fprintf(stderr, "[M::%s::msc->%ld] msc_k_hits::%u, cL::%ld, min_sc::%ld, best_sc::%ld, n_u0_sc::%d, mcopy_rate::%f, # chains::%ld\n",
|
||||
// __func__, msc, res->list[n_u0].align_length, cL, min_sc, msc+plus, res->list[n_u0].shared_seed,
|
||||
// mcopy_rate, n_u-n_u0);
|
||||
} else if(n_u == n_u0 + 1) {
|
||||
z = &(res->list[n_u0]); k = n_u0; i = 0;
|
||||
z->non_homopolymer_errors = des_idx + i;
|
||||
n_v0 = z->x_id; ni = z->align_length;
|
||||
for (j = 0; j < ni; j++, i++) {
|
||||
///k0 + (ni - j - 1)
|
||||
des[i] = a[ii[n_v0 + (ni- j - 1)]];
|
||||
des[i].readID = k;
|
||||
}
|
||||
z->x_id = xid;
|
||||
if(gen_cigar) gen_fake_cigar(&(z->f_cigar), z, apend_be, des+i-ni, ni);
|
||||
if(!khit_n) z->align_length = 0;
|
||||
}
|
||||
return i;
|
||||
} else {
|
||||
msc += plus; i = msc_i; cL = 0;
|
||||
while (i >= 0) {t[cL++] = i; i = p[i];}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
///a[] has been sorted by self_offset
|
||||
// i = msc_i; cL = 0;
|
||||
// while (i >= 0) {t[cL++] = i; i = p[i];}
|
||||
kv_pushp_ol(overlap_region, (*res), &z);
|
||||
push_ovlp_chain_qgen(z, xid, xl, yl, msc, &(a[t[cL-1]]), &(a[t[0]]));
|
||||
for (i = 0; i < cL; i++) {des[i] = a[t[cL-i-1]]; des[i].readID = res->length-1;}
|
||||
z->non_homopolymer_errors = des_idx;
|
||||
if(gen_cigar) gen_fake_cigar(&(z->f_cigar), z, apend_be, des, cL);
|
||||
if(khit_n) z->align_length = cL;
|
||||
return cL;
|
||||
}
|
||||
|
||||
|
||||
#define rev_khit(an, xl, yl) do { \
|
||||
@@ -2308,6 +2603,57 @@ uint64_t lchain_simple(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp
|
||||
return cL;
|
||||
}
|
||||
|
||||
uint64_t lchain_simple0(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp, int64_t max_skip, int64_t max_iter)
|
||||
{
|
||||
if(a_n <= 0) return 0;
|
||||
int64_t *p, *t, max_f, n_skip, st, max_j, sc, msc, msc_i;
|
||||
int32_t *f; int64_t i, j, cL = 0;
|
||||
resize_Chain_Data(dp, a_n, NULL);
|
||||
t = dp->tmp; f = dp->score; p = dp->pre; msc = msc_i = -1;
|
||||
|
||||
memset(t, 0, (a_n*sizeof((*t))));
|
||||
f[0]=a[0].cnt; p[0]=-1; msc = f[0]; msc_i = 0;
|
||||
|
||||
for (i = 1, st = 0; i < a_n; ++i) {
|
||||
max_f = INT32_MIN; n_skip = 0; max_j = -1;
|
||||
if ((i-st) > max_iter) st = i-max_iter;
|
||||
///[st, i-2]
|
||||
for (j=i-1; j >= st; --j) {
|
||||
if((a[i].self_offset > a[j].self_offset)&&(a[i].offset > a[j].offset)) {
|
||||
sc = f[j]+a[i].cnt;
|
||||
if (sc > max_f) {
|
||||
max_f = sc, max_j = j;
|
||||
if (n_skip > 0) --n_skip;
|
||||
} else if (t[j] == (int32_t)i) {
|
||||
if (++n_skip > max_skip)
|
||||
break;
|
||||
}
|
||||
if (p[j] >= 0) t[p[j]] = i;
|
||||
}
|
||||
}
|
||||
f[i] = max_f; p[i] = max_j;
|
||||
if(f[i] > msc) {
|
||||
msc = f[i]; msc_i = i;
|
||||
}
|
||||
}
|
||||
|
||||
///a[] has been sorted by self_offset
|
||||
i = msc_i;
|
||||
cL = 0;
|
||||
while (i >= 0) {
|
||||
t[cL++] = i; i = p[i];
|
||||
}
|
||||
|
||||
n_skip = cL>>1;
|
||||
for (i = 0; i < n_skip; i++) {
|
||||
msc_i = t[i]; t[i] = t[cL-i-1]; t[cL-i-1] = msc_i;
|
||||
}
|
||||
if(des) {
|
||||
for (i = 0; i < cL; i++) des[i] = a[t[i]];
|
||||
}
|
||||
return cL;
|
||||
}
|
||||
|
||||
inline int64_t hit_long_gap(k_mer_hit *a, k_mer_hit *b, int64_t max_lgap, double small_bw_rate, int64_t min_small_bw)
|
||||
{
|
||||
int64_t dq, dr, dd, dm;
|
||||
|
||||
@@ -8,10 +8,17 @@
|
||||
|
||||
#define WINDOW 375
|
||||
#define WINDOW_BOUNDARY 375
|
||||
#define WINDOW_HC 775
|
||||
///ONT high error
|
||||
// #define WINDOW_OHC 475
|
||||
#define WINDOW_OHC 375
|
||||
#define WINDOW_HC_FAST 512
|
||||
///for one side, the first or last WINDOW_UNCORRECT_SINGLE_SIDE_BOUNDARY bases should not be corrected
|
||||
#define WINDOW_UNCORRECT_SINGLE_SIDE_BOUNDARY 25
|
||||
#define THRESHOLD 15
|
||||
#define OVERLAP_THRESHOLD_HIFI_FILTER 0.9
|
||||
#define OVERLAP_THRESHOLD_HIFI_FF_FILTER 0.6
|
||||
#define OVERLAP_THRESHOLD_HIFI_FF_DE_FILTER 0.5
|
||||
#define OVERLAP_THRESHOLD_NOSI_FILTER 0.7
|
||||
#define OVERLAP_THRESHOLD_FILTER_HPC 0.75
|
||||
#define HIGH_HET_OVERLAP_THRESHOLD_FILTER 0.3
|
||||
@@ -229,6 +236,9 @@ uint64_t lchain_qdp_fix(k_mer_hit* a, int64_t a_n, Chain_Data* dp, int64_t max_s
|
||||
int64_t left_fix, int64_t right_fix);
|
||||
uint64_t lchain_simple(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp,
|
||||
int64_t max_skip, int64_t max_iter);
|
||||
uint64_t lchain_simple0(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp, int64_t max_skip, int64_t max_iter);
|
||||
void push_ovlp_chain_qgen(overlap_region* o, uint32_t xid, int64_t xl, int64_t yl, int64_t sc, k_mer_hit *beg, k_mer_hit *end);
|
||||
|
||||
uint64_t lchain_qdp_mcopy(Candidates_list *cl, int64_t a_idx, int64_t a_n, int64_t des_idx,
|
||||
Chain_Data* dp, overlap_region_alloc* res, int64_t max_skip, int64_t max_iter,
|
||||
int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
||||
@@ -236,4 +246,20 @@ uint64_t lchain_qdp_mcopy(Candidates_list *cl, int64_t a_idx, int64_t a_n, int64
|
||||
int64_t gen_cigar, int64_t enable_mcopy, double mcopy_rate, int64_t mcopy_khit_cutoff,
|
||||
int64_t khit_n);
|
||||
|
||||
uint64_t lchain_qdp_mcopy_fast(Candidates_list *cl, int64_t a_idx, int64_t a_n, int64_t des_idx,
|
||||
Chain_Data* dp, overlap_region_alloc* res, int64_t max_skip, int64_t max_iter,
|
||||
int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
||||
uint32_t xid, int64_t xl, int64_t yl, int64_t quick_check, uint32_t apend_be,
|
||||
int64_t gen_cigar, int64_t enable_mcopy, double mcopy_rate, int64_t mcopy_khit_cutoff,
|
||||
int64_t khit_n);
|
||||
|
||||
#define kv_pushp_ol(type, v, p) do { \
|
||||
if ((v).length == (v).size) { \
|
||||
(v).list = (type*)realloc((v).list, sizeof(type)*((v).size?((v).size<<1):(2))); \
|
||||
memset((v).list+(v).size, 0, sizeof(overlap_region)*(((v).size?((v).size<<1):2)-(v).size));\
|
||||
(v).size = (v).size?((v).size<<1):(2); \
|
||||
} \
|
||||
*(p) = &((v).list[(v).length++]); \
|
||||
} while (0)
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,162 @@
|
||||
#include "Levenshtein_distance.h"
|
||||
#include <immintrin.h>
|
||||
|
||||
|
||||
#define init_simd_ed4(PSA, PNA, THRE, ABS_DIAG, R_ERR, R_PE, SI, TN, CUT, BD, I, MM, PEQ_MM, LZ, IBD) {\
|
||||
(R_ERR)[(SI)] = INT32_MAX; (R_PE)[(SI)] = -1; (IBD)[(SI)] = ((THRE)<<1) - (ABS_DIAG)[(SI)];\
|
||||
if(((PNA)[(SI)] <= (TN) + (CUT)) && ((TN) <= (PNA)[(SI)] + (CUT))) {\
|
||||
(BD) = (((THRE)<<1)+1)-(ABS_DIAG)[(SI)]; (BD) = (((BD)<=(PNA)[(SI)])?(BD):(PNA)[(SI)]); (LZ) |= (((int32_t)1u) << (SI));\
|
||||
for ((I) = 0, (MM) = (((Word)1)<<((ABS_DIAG)[(SI)])); (I) < (BD); (I)++) {\
|
||||
(PEQ_MM)[seq_nt4_table[(uint8_t)(PSA)[(SI)][(I)]]][(SI)] |= (MM); (MM) <<= 1;\
|
||||
}\
|
||||
}\
|
||||
}
|
||||
|
||||
#define ed_core_64x4(PEQz, VPz, VNz, Xz, D0z, HNz, HPz) { \
|
||||
/**(X) = (Peq)|(VN);**/\
|
||||
(Xz) = _mm256_or_si256((PEQz), (VNz)); \
|
||||
/**(D0) = (((VP) + ((X)&(VP))) ^ (VP)) | (X);**/\
|
||||
(D0z) = _mm256_or_si256(_mm256_xor_si256(_mm256_add_epi64((VPz), _mm256_and_si256((Xz), (VPz))), (VPz)), (Xz)); \
|
||||
/**(HN) = (VP)&(D0);**/\
|
||||
(HNz) = _mm256_and_si256((VPz), (D0z)); \
|
||||
/**(HP) = (VN) | ~((VP) | (D0));**/\
|
||||
(HPz) = _mm256_or_si256((VNz), _mm256_andnot_si256(_mm256_or_si256((VPz), (D0z)), _mm256_set1_epi64x(-1))); \
|
||||
/**(X) = (D0) >> 1;**/\
|
||||
(Xz) = _mm256_srli_epi64((D0z), 1); \
|
||||
/**(VN) = (X)&(HP);**/\
|
||||
(VNz) = _mm256_and_si256((Xz), (HPz)); \
|
||||
/**(VP) = (HN) | ~((X) | (HP));**/\
|
||||
(VPz) = _mm256_or_si256((HNz), _mm256_andnot_si256(_mm256_or_si256((Xz), (HPz)), _mm256_set1_epi64x(-1))); \
|
||||
}
|
||||
|
||||
#define ed_core_upx4(PEQz, PSA, PNA, IBD, HT, CC, MMK, SI) { \
|
||||
if((HT) & (((int32_t)1u) << (SI))) {\
|
||||
(IBD)[(SI)]++;\
|
||||
if((IBD)[(SI)] < (PNA)[(SI)]) {\
|
||||
(CC) = seq_nt4_table[(uint8_t)(PSA)[(SI)][(IBD)[(SI)]]];\
|
||||
if((CC) < 4) (PEQz)[(CC)] = _mm256_or_si256((PEQz)[(CC)], (MMK)[(SI)]);\
|
||||
}\
|
||||
}\
|
||||
}
|
||||
|
||||
#define ed_tail_upx4(HT, SI, ST, AI, PNA, ABS_DIAG, K, ERR_MM, VP_MM, VN_MM, THRE, R_ERR, R_PE, BD, I) {\
|
||||
if((HT) & (((int32_t)1u) << (SI))) {\
|
||||
(ST)[(SI)] -= (ABS_DIAG)[(SI)]; (AI)[(SI)] += (PNA)[(SI)] + (ABS_DIAG)[(SI)];\
|
||||
for ((K)[(SI)] = 0; (ST)[(SI)] < 0 && (K)[(SI)] < (AI)[(SI)]; (K)[(SI)]++, (ST)[(SI)]++) {\
|
||||
(ERR_MM)[(SI)] += ((VP_MM)[(SI)]&(1ULL)); (VP_MM)[(SI)]>>=1;\
|
||||
(ERR_MM)[(SI)] -= ((VN_MM)[(SI)]&(1ULL)); (VN_MM)[(SI)]>>=1;\
|
||||
}\
|
||||
if (((ERR_MM)[(SI)] <= (THRE)) && ((ERR_MM)[(SI)] <= (R_ERR)[(SI)])) {\
|
||||
(R_ERR)[(SI)] = (ERR_MM)[(SI)]; (R_PE)[(SI)] = (ST)[(SI)];\
|
||||
}\
|
||||
(ST)[(SI)] -= (K)[(SI)]; (BD)++; (I) = (SI);\
|
||||
}\
|
||||
}
|
||||
|
||||
#define ED_TAIL_LANE(K) do { \
|
||||
if ((ht) & (((int32_t)1u) << (K))) { \
|
||||
st = tn - 1 - abs_diag_a[(K)]; \
|
||||
ai = pna[(K)] - tn + abs_diag_a[(K)]; \
|
||||
for (i = 0, uge = INT64_MAX; st < 0 && i < ai; i++, st++) { \
|
||||
err_mm[(K)] += ((VP_mm[(K)] >> i) & 1ULL); \
|
||||
err_mm[(K)] -= ((VN_mm[(K)] >> i) & 1ULL); \
|
||||
} \
|
||||
if ((err_mm[(K)] <= thre) && (err_mm[(K)] <= r_err[(K)])) { \
|
||||
r_err[(K)] = err_mm[(K)]; \
|
||||
r_pe[(K)] = st; \
|
||||
} \
|
||||
st -= i; \
|
||||
while (i < ai) { \
|
||||
err_mm[(K)] += ((VP_mm[(K)] >> i) & 1ULL); \
|
||||
err_mm[(K)] -= ((VN_mm[(K)] >> i) & 1ULL); \
|
||||
++i; \
|
||||
if ((err_mm[(K)] <= thre) && (err_mm[(K)] <= r_err[(K)])) { \
|
||||
r_err[(K)] = err_mm[(K)]; \
|
||||
r_pe[(K)] = st + i; \
|
||||
} \
|
||||
if (i == thre) uge = err_mm[(K)]; \
|
||||
} \
|
||||
if ((uge <= thre) && (uge == r_err[(K)])) r_pe[(K)] = st + thre; \
|
||||
} \
|
||||
} while (0)
|
||||
|
||||
void ed_band_cal_semi_64_w_absent_diag_avx4(char **psa, int32_t *pna, char *tstr, int32_t tn, int32_t thre, int32_t *abs_diag_a, int64_t *r_err, int64_t *r_pe)
|
||||
{
|
||||
// r_err[0] = r_err[1] = r_err[2] = r_err[3] = r_err[4] = r_err[5] = r_err[6] = r_err[7] = thre+1;
|
||||
// r_pe[0] = r_pe[1] = r_pe[2] = r_pe[3] = r_pe[4] = r_pe[5] = r_pe[6] = r_pe[7] = -1;
|
||||
|
||||
Word mm, Peq_mm[5][AVX_GS2] = {{0}}, *VN_mm = NULL, *VP_mm = NULL, c = 0; __m256i Peq[5], VP, VN, X, D0, HN, HP, lone, E, C, mmk[AVX_GS2];
|
||||
int32_t lz = 0, ht = (((int32_t)1u)<<AVX_GS2)-1; int32_t bd, ibd[AVX_GS2], i, last_high = (thre<<1), tn0 = tn - 1, cut = thre+last_high;
|
||||
|
||||
lone = _mm256_set1_epi64x(1);
|
||||
VP = _mm256_setzero_si256();
|
||||
|
||||
VN_mm = Peq_mm[0];
|
||||
VN_mm[0] = (((Word)1)<<(abs_diag_a[0]))-1; VN_mm[1] = (((Word)1)<<(abs_diag_a[1]))-1; VN_mm[2] = (((Word)1)<<(abs_diag_a[2]))-1; VN_mm[3] = (((Word)1)<<(abs_diag_a[3]))-1;
|
||||
VN = _mm256_loadu_si256((const __m256i *)VN_mm);
|
||||
|
||||
VN_mm[0] = abs_diag_a[0]; VN_mm[1] = abs_diag_a[1]; VN_mm[2] = abs_diag_a[2]; VN_mm[3] = abs_diag_a[3];
|
||||
E = _mm256_loadu_si256((const __m256i *)VN_mm);
|
||||
|
||||
memset(VN_mm, 0, (sizeof((*VN_mm))*AVX_GS2)); VN_mm = NULL;///reset
|
||||
|
||||
init_simd_ed4(psa, pna, thre, abs_diag_a, r_err, r_pe, 0, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed4(psa, pna, thre, abs_diag_a, r_err, r_pe, 1, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed4(psa, pna, thre, abs_diag_a, r_err, r_pe, 2, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed4(psa, pna, thre, abs_diag_a, r_err, r_pe, 3, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
|
||||
ht &= lz;
|
||||
if(ht == 0) return;
|
||||
|
||||
Peq[0] = _mm256_loadu_si256((const __m256i *)Peq_mm[0]);
|
||||
Peq[1] = _mm256_loadu_si256((const __m256i *)Peq_mm[1]);
|
||||
Peq[2] = _mm256_loadu_si256((const __m256i *)Peq_mm[2]);
|
||||
Peq[3] = _mm256_loadu_si256((const __m256i *)Peq_mm[3]);
|
||||
Peq[4] = _mm256_setzero_si256();
|
||||
|
||||
C = _mm256_set1_epi64x(cut + 1);///_mm512_set1_epi64x(cut);
|
||||
|
||||
mm = ((Word)1 << (thre<<1));///for the incoming char/last char**
|
||||
mmk[0] = _mm256_set_epi64x(0, 0, 0, mm);
|
||||
mmk[1] = _mm256_set_epi64x(0, 0, mm, 0);
|
||||
mmk[2] = _mm256_set_epi64x(0, mm, 0, 0);
|
||||
mmk[3] = _mm256_set_epi64x(mm, 0, 0, 0);
|
||||
|
||||
i = 0;
|
||||
|
||||
while (i < tn0) {
|
||||
ed_core_64x4(Peq[seq_nt4_table[(uint8_t)tstr[i]]], VP, VN, X, D0, HN, HP);
|
||||
E = _mm256_add_epi64(_mm256_xor_si256(lone, _mm256_and_si256(D0, lone)), E);
|
||||
// ht = _mm512_cmple_epi64_mask(E, C);
|
||||
ht =_mm256_movemask_pd(_mm256_castsi256_pd(_mm256_cmpgt_epi64(C, E)));
|
||||
ht &= lz;
|
||||
if(ht == 0) return;
|
||||
|
||||
Peq[0] = _mm256_srli_epi64(Peq[0], 1);
|
||||
Peq[1] = _mm256_srli_epi64(Peq[1], 1);
|
||||
Peq[2] = _mm256_srli_epi64(Peq[2], 1);
|
||||
Peq[3] = _mm256_srli_epi64(Peq[3], 1);
|
||||
i++; ///c = 4;
|
||||
|
||||
ed_core_upx4(Peq, psa, pna, ibd, ht, c, mmk, 0);
|
||||
ed_core_upx4(Peq, psa, pna, ibd, ht, c, mmk, 1);
|
||||
ed_core_upx4(Peq, psa, pna, ibd, ht, c, mmk, 2);
|
||||
ed_core_upx4(Peq, psa, pna, ibd, ht, c, mmk, 3);
|
||||
}
|
||||
|
||||
ed_core_64x4(Peq[seq_nt4_table[(uint8_t)tstr[i]]], VP, VN, X, D0, HN, HP);
|
||||
E = _mm256_add_epi64(_mm256_xor_si256(lone, _mm256_and_si256(D0, lone)), E);
|
||||
// ht = _mm512_cmple_epi64_mask(E, C);
|
||||
ht =_mm256_movemask_pd(_mm256_castsi256_pd(_mm256_cmpgt_epi64(C, E)));
|
||||
ht &= lz;
|
||||
if(ht == 0) return;
|
||||
|
||||
int32_t st, ai; int64_t err_mm[AVX_GS2], uge;
|
||||
VN_mm = Peq_mm[0]; VP_mm = Peq_mm[1];
|
||||
_mm256_storeu_si256((__m256i *)VN_mm, VN); _mm256_storeu_si256((__m256i *)VP_mm, VP); _mm256_storeu_si256((__m256i *)err_mm, E);
|
||||
|
||||
ED_TAIL_LANE(0);
|
||||
ED_TAIL_LANE(1);
|
||||
ED_TAIL_LANE(2);
|
||||
ED_TAIL_LANE(3);
|
||||
}
|
||||
@@ -0,0 +1,276 @@
|
||||
#include "Levenshtein_distance.h"
|
||||
#include <immintrin.h>
|
||||
|
||||
#define init_simd_ed(PSA, PNA, THRE, ABS_DIAG, R_ERR, R_PE, SI, TN, CUT, BD, I, MM, PEQ_MM, LZ, IBD) {\
|
||||
(R_ERR)[(SI)] = INT32_MAX; (R_PE)[(SI)] = -1; (IBD)[(SI)] = ((THRE)<<1) - (ABS_DIAG)[(SI)];\
|
||||
if(((PNA)[(SI)] <= (TN) + (CUT)) && ((TN) <= (PNA)[(SI)] + (CUT))) {\
|
||||
(BD) = (((THRE)<<1)+1)-(ABS_DIAG)[(SI)]; (BD) = (((BD)<=(PNA)[(SI)])?(BD):(PNA)[(SI)]); (LZ) |= (((__mmask8)1u) << (SI));\
|
||||
for ((I) = 0, (MM) = (((Word)1)<<((ABS_DIAG)[(SI)])); (I) < (BD); (I)++) {\
|
||||
(PEQ_MM)[seq_nt4_table[(uint8_t)(PSA)[(SI)][(I)]]][(SI)] |= (MM); (MM) <<= 1;\
|
||||
}\
|
||||
}\
|
||||
}
|
||||
|
||||
#define ed_core_64x8(PEQz, VPz, VNz, Xz, D0z, HNz, HPz) { \
|
||||
/**(X) = (Peq)|(VN);**/\
|
||||
(Xz) = _mm512_or_si512((PEQz), (VNz));\
|
||||
/**(D0) = (((VP) + ((X)&(VP))) ^ (VP)) | (X);**/\
|
||||
(D0z) = _mm512_or_si512(_mm512_xor_si512(_mm512_add_epi64((VPz), _mm512_and_si512((Xz), (VPz))), (VPz)), (Xz));\
|
||||
/**(HN) = (VP)&(D0);**/\
|
||||
(HNz) = _mm512_and_si512((VPz), (D0z));\
|
||||
/**(HP) = (VN) | ~((VP) | (D0));**/\
|
||||
(HPz) = _mm512_or_si512((VNz), _mm512_andnot_si512(_mm512_or_si512((VPz), (D0z)), _mm512_set1_epi64(-1)));\
|
||||
/**(X) = (D0) >> 1;**/\
|
||||
(Xz) = _mm512_srli_epi64((D0z), 1);\
|
||||
/**(VN) = (X)&(HP);**/\
|
||||
(VNz) = _mm512_and_si512((Xz), (HPz));\
|
||||
/**(VP) = (HN) | ~((X) | (HP));**/\
|
||||
(VPz) = _mm512_or_si512((HNz), _mm512_andnot_si512(_mm512_or_si512((Xz), (HPz)), _mm512_set1_epi64(-1)));\
|
||||
}
|
||||
|
||||
#define ed_core_upx8(PEQz, PSA, PNA, IBD, HT, CC, MMK, SI) { \
|
||||
if((HT) & (((__mmask8)1u) << (SI))) {\
|
||||
(IBD)[(SI)]++;\
|
||||
if((IBD)[(SI)] < (PNA)[(SI)]) {\
|
||||
(CC) = seq_nt4_table[(uint8_t)(PSA)[(SI)][(IBD)[(SI)]]];\
|
||||
if((CC) < 4) (PEQz)[(CC)] = _mm512_or_si512((PEQz)[(CC)], (MMK)[(SI)]);\
|
||||
}\
|
||||
}\
|
||||
}
|
||||
|
||||
#define ed_tail_upx8(HT, SI, ST, AI, PNA, ABS_DIAG, K, ERR_MM, VP_MM, VN_MM, THRE, R_ERR, R_PE, BD, I) {\
|
||||
if((HT) & (((__mmask8)1u) << (SI))) {\
|
||||
(ST)[(SI)] -= (ABS_DIAG)[(SI)]; (AI)[(SI)] += (PNA)[(SI)] + (ABS_DIAG)[(SI)];\
|
||||
for ((K)[(SI)] = 0; (ST)[(SI)] < 0 && (K)[(SI)] < (AI)[(SI)]; (K)[(SI)]++, (ST)[(SI)]++) {\
|
||||
(ERR_MM)[(SI)] += ((VP_MM)[(SI)]&(1ULL)); (VP_MM)[(SI)]>>=1;\
|
||||
(ERR_MM)[(SI)] -= ((VN_MM)[(SI)]&(1ULL)); (VN_MM)[(SI)]>>=1;\
|
||||
}\
|
||||
if (((ERR_MM)[(SI)] <= (THRE)) && ((ERR_MM)[(SI)] <= (R_ERR)[(SI)])) {\
|
||||
(R_ERR)[(SI)] = (ERR_MM)[(SI)]; (R_PE)[(SI)] = (ST)[(SI)];\
|
||||
}\
|
||||
(ST)[(SI)] -= (K)[(SI)]; (BD)++; (I) = (SI);\
|
||||
}\
|
||||
}
|
||||
|
||||
#define ed_tail_ck8(MBEST, SI, R_PE, ST, K, THRE, UGE_MM, ERR_MM, AI, HT) {\
|
||||
(K)[(SI)]++;\
|
||||
if((MBEST) & (((__mmask8)1u) << (SI))) {\
|
||||
(R_PE)[(SI)] = (ST)[(SI)] + (K)[(SI)];\
|
||||
}\
|
||||
if((K)[(SI)] >= (AI)[(SI)]) (HT) &= ~(((__mmask8)1u) << (SI));\
|
||||
if((K)[(SI)] == (THRE)) (UGE_MM)[(SI)] = (ERR_MM)[(SI)];\
|
||||
}
|
||||
|
||||
|
||||
void ed_band_cal_semi_64_w_absent_diag_avx8(char **psa, int32_t *pna, char *tstr, int32_t tn, int32_t thre, int32_t *abs_diag_a, int64_t *r_err, int64_t *r_pe)
|
||||
{
|
||||
// r_err[0] = r_err[1] = r_err[2] = r_err[3] = r_err[4] = r_err[5] = r_err[6] = r_err[7] = thre+1;
|
||||
// r_pe[0] = r_pe[1] = r_pe[2] = r_pe[3] = r_pe[4] = r_pe[5] = r_pe[6] = r_pe[7] = -1;
|
||||
/**
|
||||
ed_band_cal_semi_64_w_absent_diag_avx4(psa, pna, tstr, tn, thre, abs_diag_a, r_err, r_pe);
|
||||
ed_band_cal_semi_64_w_absent_diag_avx4(psa + 4, pna + 4, tstr, tn, thre, abs_diag_a + 4, r_err + 4, r_pe + 4);
|
||||
return;
|
||||
**/
|
||||
|
||||
|
||||
Word mm, Peq_mm[5][AVX_GS] = {{0}}, *VN_mm = NULL, *VP_mm = NULL, c = 0; __m512i Peq[5], VP, VN, X, D0, HN, HP, lone, E, C, bestE, bestPE, curPE, cutPE, threPE, ugE, mmk[AVX_GS];
|
||||
__mmask8 lz = ((__mmask8)0u), ht = (((__mmask8)1u)<<AVX_GS)-1, mtf, mt, mbest; int32_t bd, ibd[AVX_GS], i, last_high = (thre<<1), tn0 = tn - 1, cut = thre+last_high;
|
||||
|
||||
lone = _mm512_set1_epi64(1);
|
||||
VP = _mm512_setzero_si512();
|
||||
|
||||
VN_mm = Peq_mm[0];
|
||||
VN_mm[0] = (((Word)1)<<(abs_diag_a[0]))-1; VN_mm[1] = (((Word)1)<<(abs_diag_a[1]))-1; VN_mm[2] = (((Word)1)<<(abs_diag_a[2]))-1; VN_mm[3] = (((Word)1)<<(abs_diag_a[3]))-1;
|
||||
VN_mm[4] = (((Word)1)<<(abs_diag_a[4]))-1; VN_mm[5] = (((Word)1)<<(abs_diag_a[5]))-1; VN_mm[6] = (((Word)1)<<(abs_diag_a[6]))-1; VN_mm[7] = (((Word)1)<<(abs_diag_a[7]))-1;
|
||||
VN = _mm512_loadu_si512(VN_mm);
|
||||
|
||||
VN_mm[0] = abs_diag_a[0]; VN_mm[1] = abs_diag_a[1]; VN_mm[2] = abs_diag_a[2]; VN_mm[3] = abs_diag_a[3];
|
||||
VN_mm[4] = abs_diag_a[4]; VN_mm[5] = abs_diag_a[5]; VN_mm[6] = abs_diag_a[6]; VN_mm[7] = abs_diag_a[7];
|
||||
E = _mm512_loadu_si512(VN_mm);
|
||||
|
||||
memset(VN_mm, 0, (sizeof((*VN_mm))*AVX_GS)); VN_mm = NULL;///reset
|
||||
|
||||
init_simd_ed(psa, pna, thre, abs_diag_a, r_err, r_pe, 0, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed(psa, pna, thre, abs_diag_a, r_err, r_pe, 1, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed(psa, pna, thre, abs_diag_a, r_err, r_pe, 2, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed(psa, pna, thre, abs_diag_a, r_err, r_pe, 3, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed(psa, pna, thre, abs_diag_a, r_err, r_pe, 4, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed(psa, pna, thre, abs_diag_a, r_err, r_pe, 5, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed(psa, pna, thre, abs_diag_a, r_err, r_pe, 6, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
init_simd_ed(psa, pna, thre, abs_diag_a, r_err, r_pe, 7, tn, cut, bd, i, mm, Peq_mm, lz, ibd);
|
||||
|
||||
ht &= lz;
|
||||
if(ht == 0) return;
|
||||
|
||||
Peq[0] = _mm512_loadu_si512(Peq_mm[0]);
|
||||
Peq[1] = _mm512_loadu_si512(Peq_mm[1]);
|
||||
Peq[2] = _mm512_loadu_si512(Peq_mm[2]);
|
||||
Peq[3] = _mm512_loadu_si512(Peq_mm[3]);
|
||||
Peq[4] = _mm512_setzero_si512();
|
||||
|
||||
C = _mm512_set1_epi64(cut);
|
||||
|
||||
mm = ((Word)1 << (thre<<1));///for the incoming char/last char**
|
||||
mmk[0] = _mm512_mask_set1_epi64(VP, 1, mm);
|
||||
mmk[1] = _mm512_mask_set1_epi64(VP, 2, mm);
|
||||
mmk[2] = _mm512_mask_set1_epi64(VP, 4, mm);
|
||||
mmk[3] = _mm512_mask_set1_epi64(VP, 8, mm);
|
||||
mmk[4] = _mm512_mask_set1_epi64(VP, 16, mm);
|
||||
mmk[5] = _mm512_mask_set1_epi64(VP, 32, mm);
|
||||
mmk[6] = _mm512_mask_set1_epi64(VP, 64, mm);
|
||||
mmk[7] = _mm512_mask_set1_epi64(VP, 128, mm);
|
||||
|
||||
i = 0;
|
||||
|
||||
while (i < tn0) {
|
||||
ed_core_64x8(Peq[seq_nt4_table[(uint8_t)tstr[i]]], VP, VN, X, D0, HN, HP);
|
||||
E = _mm512_add_epi64(_mm512_xor_si512(lone, _mm512_and_si512(D0, lone)), E);
|
||||
ht = _mm512_cmple_epi64_mask(E, C);
|
||||
ht &= lz;
|
||||
if(ht == 0) return;
|
||||
|
||||
Peq[0] = _mm512_srli_epi64(Peq[0], 1);
|
||||
Peq[1] = _mm512_srli_epi64(Peq[1], 1);
|
||||
Peq[2] = _mm512_srli_epi64(Peq[2], 1);
|
||||
Peq[3] = _mm512_srli_epi64(Peq[3], 1);
|
||||
i++; ///c = 4;
|
||||
|
||||
ed_core_upx8(Peq, psa, pna, ibd, ht, c, mmk, 0);
|
||||
ed_core_upx8(Peq, psa, pna, ibd, ht, c, mmk, 1);
|
||||
ed_core_upx8(Peq, psa, pna, ibd, ht, c, mmk, 2);
|
||||
ed_core_upx8(Peq, psa, pna, ibd, ht, c, mmk, 3);
|
||||
ed_core_upx8(Peq, psa, pna, ibd, ht, c, mmk, 4);
|
||||
ed_core_upx8(Peq, psa, pna, ibd, ht, c, mmk, 5);
|
||||
ed_core_upx8(Peq, psa, pna, ibd, ht, c, mmk, 6);
|
||||
ed_core_upx8(Peq, psa, pna, ibd, ht, c, mmk, 7);
|
||||
}
|
||||
|
||||
ed_core_64x8(Peq[seq_nt4_table[(uint8_t)tstr[i]]], VP, VN, X, D0, HN, HP);
|
||||
E = _mm512_add_epi64(_mm512_xor_si512(lone, _mm512_and_si512(D0, lone)), E);
|
||||
ht = _mm512_cmple_epi64_mask(E, C);
|
||||
ht &= lz;
|
||||
if(ht == 0) return;
|
||||
|
||||
|
||||
// site = tn - 1 - abs_diag;/**up bound**/
|
||||
// ai = pn - tn + abs_diag; /**in most cases, ai = (thre<<1)**/
|
||||
int32_t st[AVX_GS] = {tn-1, tn-1, tn-1, tn-1, tn-1, tn-1, tn-1, tn-1};
|
||||
int32_t ai[AVX_GS] = {-tn, -tn, -tn, -tn, -tn, -tn, -tn, -tn};
|
||||
int32_t k[AVX_GS] = {0}; i = -1; bd = 0;
|
||||
int64_t err_mm[AVX_GS], uge_mm[AVX_GS] = {INT32_MAX, INT32_MAX, INT32_MAX, INT32_MAX, INT32_MAX, INT32_MAX, INT32_MAX, INT32_MAX};
|
||||
VN_mm = Peq_mm[0]; VP_mm = Peq_mm[1];
|
||||
_mm512_storeu_si512(VN_mm, VN); _mm512_storeu_si512(VP_mm, VP); _mm512_storeu_si512(err_mm, E);
|
||||
|
||||
ed_tail_upx8(ht, 0, st, ai, pna, abs_diag_a, k, err_mm, VP_mm, VN_mm, thre, r_err, r_pe, bd, i);
|
||||
ed_tail_upx8(ht, 1, st, ai, pna, abs_diag_a, k, err_mm, VP_mm, VN_mm, thre, r_err, r_pe, bd, i);
|
||||
ed_tail_upx8(ht, 2, st, ai, pna, abs_diag_a, k, err_mm, VP_mm, VN_mm, thre, r_err, r_pe, bd, i);
|
||||
ed_tail_upx8(ht, 3, st, ai, pna, abs_diag_a, k, err_mm, VP_mm, VN_mm, thre, r_err, r_pe, bd, i);
|
||||
ed_tail_upx8(ht, 4, st, ai, pna, abs_diag_a, k, err_mm, VP_mm, VN_mm, thre, r_err, r_pe, bd, i);
|
||||
ed_tail_upx8(ht, 5, st, ai, pna, abs_diag_a, k, err_mm, VP_mm, VN_mm, thre, r_err, r_pe, bd, i);
|
||||
ed_tail_upx8(ht, 6, st, ai, pna, abs_diag_a, k, err_mm, VP_mm, VN_mm, thre, r_err, r_pe, bd, i);
|
||||
ed_tail_upx8(ht, 7, st, ai, pna, abs_diag_a, k, err_mm, VP_mm, VN_mm, thre, r_err, r_pe, bd, i);
|
||||
|
||||
if(bd <= 0) return;
|
||||
|
||||
if(bd > 1) {
|
||||
VN = _mm512_loadu_si512(VN_mm); VP = _mm512_loadu_si512(VP_mm); E = _mm512_loadu_si512(err_mm); i = 0;
|
||||
|
||||
bestE = _mm512_loadu_si512(r_err); ///threE = _mm512_set1_epi64(thre);
|
||||
if(k[0] >= ai[0]) ht &= ((__mmask8)(255-1));
|
||||
if(k[1] >= ai[1]) ht &= ((__mmask8)(255-2));
|
||||
if(k[2] >= ai[2]) ht &= ((__mmask8)(255-4));
|
||||
if(k[3] >= ai[3]) ht &= ((__mmask8)(255-8));
|
||||
if(k[4] >= ai[4]) ht &= ((__mmask8)(255-16));
|
||||
if(k[5] >= ai[5]) ht &= ((__mmask8)(255-32));
|
||||
if(k[6] >= ai[6]) ht &= ((__mmask8)(255-64));
|
||||
if(k[7] >= ai[7]) ht &= ((__mmask8)(255-128));
|
||||
|
||||
err_mm[0] = r_pe[0]; err_mm[1] = r_pe[1]; err_mm[2] = r_pe[2]; err_mm[3] = r_pe[3];
|
||||
err_mm[4] = r_pe[4]; err_mm[5] = r_pe[5]; err_mm[6] = r_pe[6]; err_mm[7] = r_pe[7];
|
||||
bestPE = _mm512_loadu_si512(err_mm);
|
||||
err_mm[0] = st[0] + k[0]; err_mm[1] = st[1] + k[1]; err_mm[2] = st[2] + k[2]; err_mm[3] = st[3] + k[3];
|
||||
err_mm[4] = st[4] + k[4]; err_mm[5] = st[5] + k[5]; err_mm[6] = st[6] + k[6]; err_mm[7] = st[7] + k[7];
|
||||
curPE = _mm512_loadu_si512(err_mm);
|
||||
err_mm[0] = st[0] + thre; err_mm[1] = st[1] + thre; err_mm[2] = st[2] + thre; err_mm[3] = st[3] + thre;
|
||||
err_mm[4] = st[4] + thre; err_mm[5] = st[5] + thre; err_mm[6] = st[6] + thre; err_mm[7] = st[7] + thre;
|
||||
threPE = _mm512_loadu_si512(err_mm);
|
||||
err_mm[0] = st[0] + ai[0]; err_mm[1] = st[1] + ai[1]; err_mm[2] = st[2] + ai[2]; err_mm[3] = st[3] + ai[3];
|
||||
err_mm[4] = st[4] + ai[4]; err_mm[5] = st[5] + ai[5]; err_mm[6] = st[6] + ai[6]; err_mm[7] = st[7] + ai[7];
|
||||
cutPE = _mm512_loadu_si512(err_mm);
|
||||
|
||||
ugE = _mm512_loadu_si512(uge_mm);
|
||||
|
||||
// mtf = _mm512_cmpge_epi64_mask(curPE, threPE) | ((__mmask8)(~ht));
|
||||
mtf = _mm512_cmpge_epi64_mask(curPE, threPE);
|
||||
|
||||
while ((ht != 0) && ((mtf|((__mmask8)(~ht))) != (__mmask8)255)) {
|
||||
E = _mm512_add_epi64(E, _mm512_and_si512(VP, lone)); VP = _mm512_srli_epi64(VP, 1);
|
||||
E = _mm512_sub_epi64(E, _mm512_and_si512(VN, lone)); VN = _mm512_srli_epi64(VN, 1);
|
||||
// i++;
|
||||
|
||||
curPE = _mm512_add_epi64(curPE, lone);
|
||||
ht &= _mm512_cmple_epi64_mask(curPE, cutPE);
|
||||
if (ht == 0) break;
|
||||
|
||||
mbest = _mm512_cmple_epi64_mask(E, bestE) & ht;
|
||||
|
||||
bestE = _mm512_mask_mov_epi64(bestE, mbest, E);
|
||||
bestPE = _mm512_mask_mov_epi64(bestPE, mbest, curPE);
|
||||
|
||||
mt = _mm512_cmpeq_epi64_mask(curPE, threPE);
|
||||
ugE = _mm512_mask_mov_epi64(ugE, mt&ht, E);
|
||||
|
||||
mtf |= mt;
|
||||
|
||||
// if(mbest && i < thre) _mm512_storeu_si512(err_mm, E);
|
||||
|
||||
// ed_tail_ck8(mbest, 0, r_pe, st, k, thre, uge_mm, err_mm, ai, ht);
|
||||
// ed_tail_ck8(mbest, 1, r_pe, st, k, thre, uge_mm, err_mm, ai, ht);
|
||||
// ed_tail_ck8(mbest, 2, r_pe, st, k, thre, uge_mm, err_mm, ai, ht);
|
||||
// ed_tail_ck8(mbest, 3, r_pe, st, k, thre, uge_mm, err_mm, ai, ht);
|
||||
// ed_tail_ck8(mbest, 4, r_pe, st, k, thre, uge_mm, err_mm, ai, ht);
|
||||
// ed_tail_ck8(mbest, 5, r_pe, st, k, thre, uge_mm, err_mm, ai, ht);
|
||||
// ed_tail_ck8(mbest, 6, r_pe, st, k, thre, uge_mm, err_mm, ai, ht);
|
||||
// ed_tail_ck8(mbest, 7, r_pe, st, k, thre, uge_mm, err_mm, ai, ht);
|
||||
}
|
||||
|
||||
|
||||
while (ht != 0) {
|
||||
E = _mm512_add_epi64(E, _mm512_and_si512(VP, lone)); VP = _mm512_srli_epi64(VP, 1);
|
||||
E = _mm512_sub_epi64(E, _mm512_and_si512(VN, lone)); VN = _mm512_srli_epi64(VN, 1);
|
||||
// i++;
|
||||
|
||||
curPE = _mm512_add_epi64(curPE, lone);
|
||||
ht &= _mm512_cmple_epi64_mask(curPE, cutPE);
|
||||
if (ht == 0) break;
|
||||
|
||||
mbest = _mm512_cmple_epi64_mask(E, bestE) & ht;
|
||||
|
||||
bestE = _mm512_mask_mov_epi64(bestE, mbest, E);
|
||||
bestPE = _mm512_mask_mov_epi64(bestPE, mbest, curPE);
|
||||
}
|
||||
|
||||
cutPE = _mm512_set1_epi64(thre);
|
||||
ht = _mm512_cmpgt_epi64_mask(bestE, cutPE);
|
||||
bestE = _mm512_mask_set1_epi64(bestE, ht, INT32_MAX);
|
||||
bestPE = _mm512_mask_set1_epi64(bestPE, ht, -1);
|
||||
|
||||
ht = _mm512_cmple_epi64_mask(ugE, cutPE) & _mm512_cmpeq_epi64_mask(ugE, bestE);
|
||||
bestPE = _mm512_mask_mov_epi64(bestPE, ht, threPE);
|
||||
|
||||
_mm512_storeu_si512(r_err, bestE);
|
||||
_mm512_storeu_si512(r_pe, bestPE);
|
||||
_mm512_storeu_si512(uge_mm, ugE);
|
||||
} else {///bd == 1
|
||||
while (k[i] < ai[i]) {
|
||||
err_mm[i] += (VP_mm[i]&(1ULL)); VP_mm[i]>>=1;
|
||||
err_mm[i] -= (VN_mm[i]&(1ULL)); VN_mm[i]>>=1;
|
||||
++k[i];
|
||||
if ((err_mm[i] <= thre) && (err_mm[i] <= r_err[i])) {
|
||||
r_err[i] = err_mm[i]; r_pe[i] = st[i] + k[i];
|
||||
}
|
||||
if(k[i] == thre) uge_mm[i] = err_mm[i];
|
||||
}
|
||||
if((uge_mm[i] <= thre) && (uge_mm[i] == r_err[i])) r_pe[i] = st[i] + thre;
|
||||
}
|
||||
}
|
||||
+214
-2
@@ -12,6 +12,9 @@
|
||||
#include <stdio.h>
|
||||
#include "kvec.h"
|
||||
|
||||
#define AVX_GS 8
|
||||
#define AVX_GS2 4
|
||||
|
||||
extern const unsigned char seq_nt4_table[256];
|
||||
typedef uint64_t Word;
|
||||
typedef uint32_t Word_32;
|
||||
@@ -548,6 +551,201 @@ inline int32_t pop_trace_back(asg16_v *res, int32_t i, uint16_t *c, uint32_t *le
|
||||
return i;
|
||||
}
|
||||
|
||||
///compact functions
|
||||
#define pop_trac_bpc(in, rc, rb, rl) do { \
|
||||
(rc) = ((in)>>14);\
|
||||
if((rc) == 1 || (rc) == 2) {(rb) = (((in)>>12)&3); (rl) = ((in)&(0xfff));}\
|
||||
else {(rl) = ((in)&(0x3fff));}\
|
||||
} while (0)
|
||||
|
||||
inline void push_trace_bp(asg16_v *res, uint16_t c, uint16_t b, uint32_t len, uint32_t is_append)
|
||||
{
|
||||
uint16_t p, c0, b0, len0, mm;
|
||||
if((is_append) && (res->n)) {
|
||||
b0 = b;
|
||||
pop_trac_bpc(res->a[res->n-1], c0, b0, len0);
|
||||
if((c == c0) && (b == b0)) {
|
||||
res->n--; len += len0;
|
||||
}
|
||||
}
|
||||
|
||||
mm = (0x3fff); c0 = c; c <<= 14;
|
||||
if(c0 == 1 || c0 == 2) {
|
||||
mm = (0xfff); c += ((b&3) << 12);
|
||||
}
|
||||
|
||||
|
||||
while (len >= mm) {
|
||||
p = (c + mm); kv_push(uint16_t, *res, p); len -= mm;
|
||||
}
|
||||
// fprintf(stderr, "[M::%s] c::%u, len::%u\n", __func__, c, len);
|
||||
if(len) {
|
||||
p = (c + len); kv_push(uint16_t, *res, p);
|
||||
}
|
||||
}
|
||||
|
||||
inline uint32_t pop_trace_bp(asg16_v *res, uint32_t i, uint16_t *c, uint16_t *b, uint32_t *len)
|
||||
{
|
||||
(*c) = (res->a[i]>>14);
|
||||
if((*c) == 1 || (*c) == 2) {
|
||||
(*b) = ((res->a[i]>>12)&3);
|
||||
(*len) = (res->a[i]&(0xfff));
|
||||
} else {
|
||||
(*b) = (uint16_t)-1;
|
||||
(*len) = (res->a[i]&(0x3fff));
|
||||
}
|
||||
|
||||
uint32_t sl; uint16_t sb;
|
||||
for (i++; (i < res->n) && ((*c) == (res->a[i]>>14)); i++) {
|
||||
if((*c) == 1 || (*c) == 2) {
|
||||
sb = ((res->a[i]>>12)&3); sl = (res->a[i]&(0xfff));
|
||||
} else {
|
||||
sb = (uint16_t)-1; sl = (res->a[i]&(0x3fff));
|
||||
}
|
||||
if((*b) != sb) break;
|
||||
(*len) += sl;
|
||||
}
|
||||
return i;
|
||||
}
|
||||
|
||||
inline int64_t pop_trace_bp_rev(asg16_v *res, int64_t i, uint16_t *c, uint16_t *b, uint32_t *len)
|
||||
{
|
||||
(*c) = (res->a[i]>>14);
|
||||
if((*c) == 1 || (*c) == 2) {
|
||||
(*b) = ((res->a[i]>>12)&3);
|
||||
(*len) = (res->a[i]&(0xfff));
|
||||
} else {
|
||||
(*b) = (uint16_t)-1;
|
||||
(*len) = (res->a[i]&(0x3fff));
|
||||
}
|
||||
|
||||
uint32_t sl; uint16_t sb;
|
||||
for (i--; (i >= 0) && ((*c) == (res->a[i]>>14)); i--) {
|
||||
if((*c) == 1 || (*c) == 2) {
|
||||
sb = ((res->a[i]>>12)&3); sl = (res->a[i]&(0xfff));
|
||||
} else {
|
||||
sb = (uint16_t)-1; sl = (res->a[i]&(0x3fff));
|
||||
}
|
||||
if((*b) != sb) break;
|
||||
(*len) += sl;
|
||||
}
|
||||
return i;
|
||||
}
|
||||
|
||||
///full functions
|
||||
#define pop_trac_bpc_f(in, rc, rbq, rbt, rl) do { \
|
||||
(rc) = ((in)>>14);\
|
||||
if((rc) == 2 || (rc) == 3) {(rbt) = (((in)>>12)&3); (rl) = ((in)&(0xfff));}\
|
||||
else if((rc) == 1) {(rbt) = (((in)>>12)&3); (rbq) = (((in)>>10)&3); (rl) = ((in)&(0x3ff));}\
|
||||
else {(rl) = ((in)&(0x3fff));}\
|
||||
} while (0)
|
||||
|
||||
inline void push_trace_bp_f(asg16_v *res, uint16_t c, uint16_t bq, uint16_t bt, uint32_t len, uint32_t is_append)
|
||||
{
|
||||
uint16_t p, c0 = c, bq0, bt0, len0, mm;
|
||||
if(c == 3) {
|
||||
bt = bq; bq = (uint16_t)-1;
|
||||
}
|
||||
if((is_append) && (res->n)) {
|
||||
bq0 = bq; bt0 = bt;
|
||||
pop_trac_bpc_f(res->a[res->n-1], c0, bq0, bt0, len0);
|
||||
if((c == c0) && (bq == bq0) && (bt == bt0)) {
|
||||
res->n--; len += len0;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
c0 = c; c <<= 14;
|
||||
if(c0 == 2 || c0 == 3) {
|
||||
mm = (0xfff); c += ((bt&3) << 12);
|
||||
} else if(c0 == 1) {
|
||||
mm = (0x3ff); c += ((bt&3) << 12); c += ((bq&3) << 10);
|
||||
} else {
|
||||
mm = (0x3fff);
|
||||
}
|
||||
|
||||
|
||||
while (len >= mm) {
|
||||
p = (c + mm); kv_push(uint16_t, *res, p); len -= mm;
|
||||
}
|
||||
// fprintf(stderr, "[M::%s] c::%u, len::%u\n", __func__, c, len);
|
||||
if(len) {
|
||||
p = (c + len); kv_push(uint16_t, *res, p);
|
||||
}
|
||||
}
|
||||
|
||||
inline uint32_t pop_trace_bp_f(asg16_v *res, uint32_t i, uint16_t *c, uint16_t *bq, uint16_t *bt, uint32_t *len)
|
||||
{
|
||||
(*c) = (res->a[i]>>14); (*bq) = (*bt) = (uint16_t)-1;
|
||||
if((*c) == 2 || (*c) == 3) {
|
||||
(*bt) = ((res->a[i]>>12)&3);
|
||||
(*len) = (res->a[i]&(0xfff));
|
||||
} else if((*c) == 1) {
|
||||
(*bt) = ((res->a[i]>>12)&3);
|
||||
(*bq) = ((res->a[i]>>10)&3);
|
||||
(*len) = (res->a[i]&(0x3ff));
|
||||
} else {
|
||||
(*len) = (res->a[i]&(0x3fff));
|
||||
}
|
||||
|
||||
uint32_t sl; uint16_t sbq, sbt;
|
||||
for (i++; (i < res->n) && ((*c) == (res->a[i]>>14)); i++) {
|
||||
sbq = sbt = (uint16_t)-1;
|
||||
if((*c) == 2 || (*c) == 3) {
|
||||
sbt = ((res->a[i]>>12)&3);
|
||||
sl = (res->a[i]&(0xfff));
|
||||
} else if((*c) == 1) {
|
||||
sbt = ((res->a[i]>>12)&3);
|
||||
sbq = ((res->a[i]>>10)&3);
|
||||
sl = (res->a[i]&(0x3ff));
|
||||
} else {
|
||||
sl = (res->a[i]&(0x3fff));
|
||||
}
|
||||
if((*bq) != sbq || (*bt) != sbt) break;
|
||||
(*len) += sl;
|
||||
}
|
||||
if((*c) == 3) {
|
||||
(*bq) = (*bt); (*bt) = (uint16_t)-1;
|
||||
}
|
||||
return i;
|
||||
}
|
||||
|
||||
inline int64_t pop_trace_bp_rev_f(asg16_v *res, int64_t i, uint16_t *c, uint16_t *bq, uint16_t *bt, uint32_t *len)
|
||||
{
|
||||
(*c) = (res->a[i]>>14); (*bq) = (*bt) = (uint16_t)-1;
|
||||
if((*c) == 2 || (*c) == 3) {
|
||||
(*bt) = ((res->a[i]>>12)&3);
|
||||
(*len) = (res->a[i]&(0xfff));
|
||||
} else if((*c) == 1) {
|
||||
(*bt) = ((res->a[i]>>12)&3);
|
||||
(*bq) = ((res->a[i]>>10)&3);
|
||||
(*len) = (res->a[i]&(0x3ff));
|
||||
} else {
|
||||
(*len) = (res->a[i]&(0x3fff));
|
||||
}
|
||||
|
||||
uint32_t sl; uint16_t sbq, sbt;
|
||||
for (i--; (i >= 0) && ((*c) == (res->a[i]>>14)); i--) {
|
||||
sbq = sbt = (uint16_t)-1;
|
||||
if((*c) == 2 || (*c) == 3) {
|
||||
sbt = ((res->a[i]>>12)&3);
|
||||
sl = (res->a[i]&(0xfff));
|
||||
} else if((*c) == 1) {
|
||||
sbt = ((res->a[i]>>12)&3);
|
||||
sbq = ((res->a[i]>>10)&3);
|
||||
sl = (res->a[i]&(0x3ff));
|
||||
} else {
|
||||
sl = (res->a[i]&(0x3fff));
|
||||
}
|
||||
if((*bq) != sbq || (*bt) != sbt) break;
|
||||
(*len) += sl;
|
||||
}
|
||||
if((*c) == 3) {
|
||||
(*bq) = (*bt); (*bt) = (uint16_t)-1;
|
||||
}
|
||||
return i;
|
||||
}
|
||||
|
||||
///511 -> 16 64-bits
|
||||
// #define MAX_E 511
|
||||
// #define MAX_L 2500
|
||||
@@ -601,7 +799,7 @@ inline uint32_t cigar_check(char *pstr, char *tstr, bit_extz_t *ez)
|
||||
if(c == 0) {
|
||||
for (k=0;(k<cl)&&(pstr[pi]==tstr[ti]);k++,pi++,ti++);
|
||||
if(k!=cl) {
|
||||
fprintf(stderr, "ERROR-d-0\n");
|
||||
fprintf(stderr, "ERROR-d-0, pi::%d, ti::%d, ci::%u\n", pi, ti, ci);
|
||||
return 0;
|
||||
}
|
||||
} else {
|
||||
@@ -609,7 +807,7 @@ inline uint32_t cigar_check(char *pstr, char *tstr, bit_extz_t *ez)
|
||||
if(c == 1) {
|
||||
for (k=0;(k<cl)&&(pstr[pi]!=tstr[ti]);k++,pi++,ti++);
|
||||
if(k!=cl) {
|
||||
fprintf(stderr, "ERROR-d-1\n");
|
||||
fprintf(stderr, "ERROR-d-1, pi::%d, ti::%d, ci::%u\n", pi, ti, ci);
|
||||
return 0;
|
||||
}
|
||||
} else if(c == 2) {///more p
|
||||
@@ -619,10 +817,20 @@ inline uint32_t cigar_check(char *pstr, char *tstr, bit_extz_t *ez)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if(err != ez->err) {
|
||||
fprintf(stderr, "ERROR-err\n");
|
||||
return 0;
|
||||
}
|
||||
if(pi != ez->pe + 1) {
|
||||
fprintf(stderr, "ERROR-pi\n");
|
||||
return 0;
|
||||
}
|
||||
if(ti != ez->te + 1) {
|
||||
fprintf(stderr, "ERROR-ti\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
@@ -3529,6 +3737,10 @@ inline void ed_band_cal_extension_64_1_w_trace(char *pstr, int32_t pn, char *tst
|
||||
return;
|
||||
}
|
||||
|
||||
void ed_band_cal_semi_64_w_absent_diag_avx4(char **psa, int32_t *pna, char *tstr, int32_t tn, int32_t thre, int32_t *abs_diag_a, int64_t *r_err, int64_t *r_pe);
|
||||
|
||||
void ed_band_cal_semi_64_w_absent_diag_avx8(char **psa, int32_t *pna, char *tstr, int32_t tn, int32_t thre, int32_t *abs_diag_a, int64_t *r_err, int64_t *r_pe);
|
||||
|
||||
inline void ed_band_cal_semi_64_w_absent_diag(char *pstr, int32_t pn, char *tstr, int32_t tn, int32_t thre, int32_t abs_diag, bit_extz_t *ez)
|
||||
{
|
||||
init_base_ed(*ez, thre, pn, tn); ez->ps = ez->pe = -1; ez->ts = 0; ez->te = tn-1;
|
||||
|
||||
@@ -5,8 +5,8 @@ CFLAGS= $(CXXFLAGS)
|
||||
CPPFLAGS=
|
||||
INCLUDES=
|
||||
OBJS= CommandLines.o Process_Read.o Assembly.o Hash_Table.o \
|
||||
POA.o Correct.o Levenshtein_distance.o Overlaps.o Trio.o kthread.o Purge_Dups.o \
|
||||
htab.o hist.o sketch.o anchor.o extract.o sys.o ksw2_extz2_sse.o hic.o rcut.o horder.o \
|
||||
POA.o Correct.o Levenshtein_distance.o Levenshtein_avx2.o Levenshtein_avx512.o Overlaps.o Trio.o kthread.o Purge_Dups.o \
|
||||
htab.o hist.o sketch.o anchor.o extract.o sys.o hic.o rcut.o horder.o ecovlp.o\
|
||||
tovlp.o inter.o kalloc.o gfa_ut.o gchain_map.o
|
||||
EXE= hifiasm
|
||||
LIBS= -lz -lpthread -lm
|
||||
@@ -19,13 +19,22 @@ endif
|
||||
.SUFFIXES:.cpp .c .o
|
||||
.PHONY:all clean depend
|
||||
|
||||
all:$(EXE)
|
||||
|
||||
.cpp.o:
|
||||
$(CXX) -c $(CXXFLAGS) $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||
|
||||
.c.o:
|
||||
$(CC) -c $(CFLAGS) $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||
|
||||
all:$(EXE)
|
||||
# compiled only with AVX2
|
||||
Levenshtein_avx2.o: Levenshtein_avx2.cpp Levenshtein_distance.h
|
||||
$(CXX) -c $(CXXFLAGS) -mavx2 $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||
|
||||
# compiled only with AVX512
|
||||
Levenshtein_avx512.o: Levenshtein_avx512.cpp Levenshtein_distance.h
|
||||
$(CXX) -c $(CXXFLAGS) -mavx512f $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||
|
||||
|
||||
$(EXE):$(OBJS) main.o
|
||||
$(CXX) $(CXXFLAGS) $^ -o $@ $(LIBS)
|
||||
@@ -40,11 +49,11 @@ depend:
|
||||
|
||||
Assembly.o: Assembly.h CommandLines.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||
Assembly.o: Hash_Table.h htab.h POA.h Correct.h Levenshtein_distance.h
|
||||
Assembly.o: kthread.h
|
||||
Assembly.o: kthread.h ecovlp.h
|
||||
CommandLines.o: CommandLines.h ketopt.h
|
||||
Correct.o: Correct.h Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h
|
||||
Correct.o: kdq.h CommandLines.h Levenshtein_distance.h POA.h Assembly.h
|
||||
Correct.o: ksw2.h ksort.h
|
||||
Correct.o: ksort.h
|
||||
Hash_Table.o: Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||
Hash_Table.o: CommandLines.h ksort.h
|
||||
Levenshtein_distance.o: Levenshtein_distance.h
|
||||
@@ -58,6 +67,7 @@ Process_Read.o: Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||
Purge_Dups.o: ksort.h Purge_Dups.h kvec.h kdq.h Overlaps.h Hash_Table.h
|
||||
Purge_Dups.o: htab.h Process_Read.h CommandLines.h Correct.h
|
||||
Purge_Dups.o: Levenshtein_distance.h POA.h kthread.h
|
||||
ecovlp.o: Hash_Table.h Process_Read.h Overlaps.h kthread.h
|
||||
Trio.o: khashl.h kthread.h kseq.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||
Trio.o: CommandLines.h htab.h
|
||||
anchor.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||
|
||||
+6558
-375
File diff suppressed because it is too large
Load Diff
+50
-8
@@ -35,6 +35,7 @@
|
||||
// #define ALTER_LABLE 2
|
||||
// #define HAP_LABLE 4
|
||||
#define ug_ext_len 75000
|
||||
#define UL_COV_THRES 2
|
||||
|
||||
#define Get_qn(RECORD) ((uint32_t)((RECORD).qns>>32))
|
||||
#define Get_qs(RECORD) ((uint32_t)((RECORD).qns))
|
||||
@@ -85,6 +86,12 @@ typedef struct {
|
||||
uint32_t n;
|
||||
} idx_emask_t;
|
||||
|
||||
typedef struct {
|
||||
uint64_t n, mask;
|
||||
uint8_t *hh;
|
||||
uint64_t tlen, tm;
|
||||
} telo_end_t;
|
||||
|
||||
typedef struct {
|
||||
///off: start idx in mg128_t * a[];
|
||||
///cnt: how many eles in this chain
|
||||
@@ -133,8 +140,8 @@ void add_ma_hit_t_alloc(ma_hit_t_alloc* x, ma_hit_t* element);
|
||||
void ma_hit_sort_tn(ma_hit_t *a, long long n);
|
||||
void ma_hit_sort_qns(ma_hit_t *a, long long n);
|
||||
|
||||
int load_all_data_from_disk(ma_hit_t_alloc **sources, ma_hit_t_alloc **reverse_sources,
|
||||
char* output_file_name);
|
||||
int load_all_data_from_disk(ma_hit_t_alloc **sources, ma_hit_t_alloc **reverse_sources, char* output_file_name);
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint32_t s:31, del:1, e;
|
||||
@@ -256,6 +263,7 @@ typedef struct { size_t n, m; uint64_t *a; } asg64_v;
|
||||
typedef struct { size_t n, m; uint32_t *a; } asg32_v;
|
||||
typedef struct { size_t n, m; ma_utg_t *a;} ma_utg_v;
|
||||
typedef struct { asg64_v idx; kv_ul_ov_t srt;} mask_ul_ov_t;
|
||||
typedef struct { uint64_t n, m; char *a; } asgchr_v;
|
||||
|
||||
typedef struct {
|
||||
ma_utg_v u;
|
||||
@@ -625,7 +633,7 @@ static inline int count_out_without_del(const asg_t *g, uint32_t v)
|
||||
|
||||
void build_string_graph_without_clean(
|
||||
int min_dp, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
|
||||
long long n_read, uint64_t* readLen, long long mini_overlap_length,
|
||||
uint64_t n_read, uint64_t* readLen, long long mini_overlap_length,
|
||||
long long max_hang_length, long long clean_round, long long gap_fuzz,
|
||||
float min_ovlp_drop_ratio, float max_ovlp_drop_ratio, char* output_file_name,
|
||||
long long bubble_dist, int read_graph, int write);
|
||||
@@ -920,7 +928,7 @@ ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, int m
|
||||
void rescue_bubble_by_chain(asg_t *sg, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
|
||||
long long tipsLen, float tip_drop_ratio, long long stops_threshold, R_to_U* ruIndex,
|
||||
float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp, uint32_t chainLenThres, long long gap_fuzz,
|
||||
bub_label_t* b_mask_t, long long no_trio_recover);
|
||||
bub_label_t* b_mask_t, long long no_trio_recover, uint8_t *cmk);
|
||||
|
||||
typedef struct{
|
||||
double weight;
|
||||
@@ -1066,7 +1074,7 @@ ma_ug_t* copy_untig_graph(ma_ug_t *src);
|
||||
ma_ug_t* output_trio_unitig_graph(asg_t *sg, ma_sub_t* coverage_cut, char* output_file_name,
|
||||
uint8_t flag, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
|
||||
long long tipsLen, float tip_drop_ratio, long long stops_threshold, R_to_U* ruIndex,
|
||||
float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp, int is_bench, bub_label_t* b_mask_t, char *f_prefix, uint8_t *kpt_buf, kvec_asg_arc_t_warp *r_edges);
|
||||
float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp, int gap_fuzz, int is_bench, bub_label_t* b_mask_t, char *f_prefix, uint8_t *kpt_buf, kvec_asg_arc_t_warp *r_edges);
|
||||
asg_t* copy_read_graph(asg_t *src);
|
||||
ma_ug_t *ma_ug_gen(asg_t *g);
|
||||
void ma_ug_destroy(ma_ug_t *ug);
|
||||
@@ -1115,6 +1123,7 @@ typedef struct{
|
||||
int64_t min_dp;
|
||||
bub_label_t* b_mask_t;
|
||||
uint64_t* readLen;
|
||||
telo_end_t *te;
|
||||
}ug_opt_t;
|
||||
|
||||
typedef struct{
|
||||
@@ -1130,12 +1139,11 @@ typedef struct{
|
||||
int64_t mini_ovlp;
|
||||
}ul_renew_t;
|
||||
|
||||
|
||||
void adjust_utg_by_trio(ma_ug_t **ug, asg_t* read_g, uint8_t flag, float drop_rate,
|
||||
ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, ma_sub_t* coverage_cut,
|
||||
long long tipsLen, float tip_drop_ratio, long long stops_threshold,
|
||||
R_to_U* ruIndex, float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp,
|
||||
kvec_asg_arc_t_warp* new_rtg_edges, bub_label_t* b_mask_t);
|
||||
int gap_fuzz, kvec_asg_arc_t_warp* new_rtg_edges, bub_label_t* b_mask_t);
|
||||
uint32_t cmp_untig_graph(ma_ug_t *src, ma_ug_t *dest);
|
||||
void reduce_hamming_error(asg_t *sg, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
int max_hang, int min_ovlp, long long gap_fuzz);
|
||||
@@ -1176,12 +1184,13 @@ void extract_sub_overlaps(uint32_t i_tScur, uint32_t i_tEcur, uint32_t i_tSpre,
|
||||
uint32_t tn, kv_u_trans_hit_t* ktb, uint32_t bn);
|
||||
void clean_u_trans_t_idx(kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g);
|
||||
void clean_u_trans_t_idx_adv(kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g);
|
||||
void clean_u_trans_t_idx_filter_adv(kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g);
|
||||
void clean_u_trans_t_idx_filter_adv(kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g, double sc_sec_rate, uint64_t uniform_only);
|
||||
uint32_t test_dbug(ma_ug_t* ug, FILE* fp);
|
||||
void write_dbug(ma_ug_t* ug, FILE* fp);
|
||||
int asg_arc_identify_simple_bubbles_multi(asg_t *g, bub_label_t* x, int check_cross);
|
||||
uint8_t get_tip_trio_infor(asg_t *sg, uint32_t begNode);
|
||||
int asg_topocut_aux(asg_t *g, uint32_t v, int max_ext);
|
||||
int asg_topocut_aux_pg(asg_t *g, uint32_t v, int max_ext, uint32_t *rv);
|
||||
int asg_arc_del_triangular_directly(asg_t *g, long long min_edge_length,
|
||||
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex);
|
||||
int asg_arc_del_short_diploid_by_exact(asg_t *g, int max_ext, ma_hit_t_alloc* sources);
|
||||
@@ -1205,8 +1214,41 @@ void ma_hit_contained_advance(ma_hit_t_alloc* sources, long long n_read, ma_sub_
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp);
|
||||
void hic_clean_adv(asg_t *sg, ug_opt_t *uopt);
|
||||
void update_ug_ou(ma_ug_t *ug, asg_t *sg);
|
||||
int asg_arc_del_trans_ul(asg_t *g, int fuzz);
|
||||
|
||||
#define JUNK_COV 5
|
||||
#define DISCARD_RATE 0.8
|
||||
|
||||
typedef struct {
|
||||
uint32_t n, m, a;
|
||||
} mmhap_status_t;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(mmhap_status_t) h;
|
||||
kvec_t(uint32_t) a;
|
||||
} mmhap_t;
|
||||
|
||||
typedef struct { // global data structure for kt_pipeline()
|
||||
ma_ug_t *ug;
|
||||
asg_t *rg;
|
||||
uint64_t *idx;
|
||||
asg64_v cov;
|
||||
uint64_t hom_min, hom_max, hom_cov, het_cov;
|
||||
} ug_rid_cov_t;
|
||||
|
||||
ug_rid_cov_t* gen_ug_rid_cov_t(ma_ug_t *ug, asg_t *rg, ma_hit_t_alloc *src);
|
||||
void destory_ug_rid_cov_t(ug_rid_cov_t *p);
|
||||
uint32_t append_cov_line_ug_rid_cov_t(uint64_t uid, uint64_t *qcc, u_trans_t *p, ug_rid_cov_t *idx, uint64_t hom_cut, double cut_rate);
|
||||
uint64_t infer_mmhap_copy(ma_ug_t *ug, asg_t *sg, ma_hit_t_alloc *src, uint8_t *ff, uint64_t uid, uint64_t het_cov, uint64_t n_hap);
|
||||
uint64_t trans_sec_cut0(kv_u_trans_t *ta, asg64_v *srt, uint32_t id, double sec_rate, uint64_t bd, ma_ug_t *ug);
|
||||
void clean_u_trans_t_idx_filter_mmhap_adv(kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g, ma_hit_t_alloc* src, ug_rid_cov_t *in);
|
||||
void gen_ug_rid_cov_t_by_ovlp(kv_u_trans_t *ta, ug_rid_cov_t *cc);
|
||||
void rescue_chimeric_reads_aggressive(ma_ug_t *i_ug, asg_t *rg, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, uint32_t chainLenThres, uint32_t is_bubble_check, uint32_t is_primary_check, kvec_asg_arc_t_warp* new_rtg_edges,
|
||||
kvec_t_u32_warp* new_rtg_nodes, bub_label_t* b_mask_t, uint8_t *cmk);
|
||||
|
||||
#define UC_Read_resize(v, s) do {\
|
||||
if ((v).size<(s)) {REALLOC((v).seq,(s));(v).size=(s);}\
|
||||
} while (0)
|
||||
|
||||
#endif
|
||||
|
||||
+504
-32
@@ -46,13 +46,20 @@ void init_All_reads(All_reads* r)
|
||||
void destory_All_reads(All_reads* r)
|
||||
{
|
||||
uint64_t i = 0;
|
||||
for (i = 0; i < r->total_reads; i++) {
|
||||
for (i = 0; i < r->tqn; i++) {
|
||||
if (r->N_site[i]) free(r->N_site[i]);
|
||||
if (r->read_sperate[i]) free(r->read_sperate[i]);
|
||||
if (r->paf && r->paf[i].buffer) free(r->paf[i].buffer);
|
||||
if (r->reverse_paf && r->reverse_paf[i].buffer) free(r->reverse_paf[i].buffer);
|
||||
///if (r->pb_regions) kv_destroy(r->pb_regions[i].a);
|
||||
if(r->rsc && r->rsc[i]) free(r->rsc[i]);
|
||||
}
|
||||
for (; i < r->total_reads; i++) {
|
||||
if (r->N_site[i]) free(r->N_site[i]);
|
||||
if (r->read_sperate[i]) free(r->read_sperate[i]);
|
||||
if (r->paf && r->paf[i].buffer) free(r->paf[i].buffer);
|
||||
if (r->reverse_paf && r->reverse_paf[i].buffer) free(r->reverse_paf[i].buffer);
|
||||
}
|
||||
|
||||
free(r->paf);
|
||||
free(r->reverse_paf);
|
||||
free(r->N_site);
|
||||
@@ -61,6 +68,7 @@ void destory_All_reads(All_reads* r)
|
||||
free(r->name_index);
|
||||
free(r->read_length);
|
||||
free(r->trio_flag);
|
||||
free(r->rsc);
|
||||
///if (r->pb_regions) free(r->pb_regions);
|
||||
}
|
||||
|
||||
@@ -108,6 +116,15 @@ void write_All_reads(All_reads* r, char* read_file_name)
|
||||
fwrite(&(asm_opt.hom_cov), sizeof(asm_opt.hom_cov), 1, fp);
|
||||
fwrite(&(asm_opt.het_cov), sizeof(asm_opt.het_cov), 1, fp);
|
||||
|
||||
uint64_t mm = 2;///1;
|
||||
if(asm_opt.is_sc) {
|
||||
fwrite(&mm, sizeof(mm), 1, fp);
|
||||
fwrite(&(r->tqn), sizeof(r->tqn), 1, fp);
|
||||
for (i = 0; i < r->tqn; i++) {
|
||||
fwrite(r->rsc[i], sizeof(uint8_t), ((r->read_length[i]/sc_bn) + ((r->read_length[i]%sc_bn)?1:0)), fp);
|
||||
}
|
||||
}
|
||||
|
||||
free(index_name);
|
||||
fflush(fp);
|
||||
fclose(fp);
|
||||
@@ -123,6 +140,7 @@ int load_All_reads(All_reads* r, char* read_file_name)
|
||||
free(index_name);
|
||||
return 0;
|
||||
}
|
||||
// fprintf(stderr, "[M::%s]\tindex_name::%s\n", __func__, index_name);
|
||||
int local_adapterLen;
|
||||
int f_flag;
|
||||
f_flag = fread(&local_adapterLen, sizeof(local_adapterLen), 1, fp);
|
||||
@@ -201,6 +219,27 @@ int load_All_reads(All_reads* r, char* read_file_name)
|
||||
}
|
||||
///r->pb_regions = NULL;
|
||||
|
||||
uint64_t mm = 0;
|
||||
if (!feof(fp)) {
|
||||
if((fread(&mm, sizeof(mm), 1, fp)) && (mm == 1 || mm == 2)) {
|
||||
if(mm == 1) {
|
||||
mm = r->total_reads;
|
||||
} else {
|
||||
assert(mm == 2);
|
||||
fread(&mm, sizeof(mm), 1, fp);
|
||||
}
|
||||
r->tqn = mm;
|
||||
|
||||
MALLOC(r->rsc, r->tqn);
|
||||
for (i = 0; i < r->tqn; i++) {
|
||||
MALLOC(r->rsc[i], (r->read_length[i]/sc_bn) + ((r->read_length[i]%sc_bn)?1:0));
|
||||
f_flag += fread(r->rsc[i], sizeof(uint8_t), (r->read_length[i]/sc_bn) + ((r->read_length[i]%sc_bn)?1:0), fp);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
free(index_name);
|
||||
fclose(fp);
|
||||
fprintf(stderr, "Reads has been loaded.\n");
|
||||
@@ -209,6 +248,206 @@ int load_All_reads(All_reads* r, char* read_file_name)
|
||||
}
|
||||
|
||||
|
||||
void write_cc_v(cc_v* r, char* read_file_name)
|
||||
{
|
||||
fprintf(stderr, "Writing raw reads to disk... \n");
|
||||
char* index_name = (char*)malloc(strlen(read_file_name)+32);
|
||||
sprintf(index_name, "%s.bin", read_file_name);
|
||||
FILE* fp = fopen(index_name, "w"); free(index_name);
|
||||
|
||||
///typedef struct {size_t n, m; asg16_v *a; uint8_t *f; uint16_t *er; uint64_t bid;} cc_v;
|
||||
asg16_v *z;
|
||||
uint64_t k, rn = r->n; uint32_t zn; fwrite(&rn, sizeof(rn), 1, fp);
|
||||
for (k = 0; k < r->n; k++) {
|
||||
z = &(r->a[k]);
|
||||
zn = z->n;
|
||||
fwrite(&zn, sizeof(zn), 1, fp);
|
||||
fwrite(z->a, sizeof((*(z->a))), zn, fp);
|
||||
}
|
||||
|
||||
fflush(fp); fclose(fp);
|
||||
fprintf(stderr, "Raw reads has been written.\n");
|
||||
}
|
||||
|
||||
uint8_t load_cc_v(cc_v* r, char* read_file_name)
|
||||
{
|
||||
fprintf(stderr, "Loading raw reads... \n");
|
||||
char* index_name = (char*)malloc(strlen(read_file_name)+32);
|
||||
sprintf(index_name, "%s.bin", read_file_name);
|
||||
FILE* fp = fopen(index_name, "r"); free(index_name);
|
||||
if (!fp) {
|
||||
fprintf(stderr, "No raw read bin.\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
///typedef struct {size_t n, m; asg16_v *a; uint8_t *f; uint16_t *er; uint64_t bid;} cc_v;
|
||||
asg16_v *z; int f_flag = 0;
|
||||
uint64_t k, rn; uint32_t zn; f_flag += fread(&rn, sizeof(rn), 1, fp);
|
||||
r->n = r->m = rn; MALLOC(r->a, r->n);
|
||||
for (k = 0; k < r->n; k++) {
|
||||
z = &(r->a[k]);
|
||||
f_flag += fread(&zn, sizeof(zn), 1, fp);
|
||||
z->n = z->m = zn; MALLOC(z->a, z->n);
|
||||
f_flag += fread(z->a, sizeof((*(z->a))), zn, fp);
|
||||
}
|
||||
|
||||
fflush(fp); fclose(fp);
|
||||
fprintf(stderr, "Raw reads has been loaded.\n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
void read_ma(ma_hit_t* x, FILE* fp)
|
||||
{
|
||||
int f_flag;
|
||||
f_flag = fread(&(x->qns), sizeof(x->qns), 1, fp);
|
||||
f_flag += fread(&(x->qe), sizeof(x->qe), 1, fp);
|
||||
f_flag += fread(&(x->tn), sizeof(x->tn), 1, fp);
|
||||
f_flag += fread(&(x->ts), sizeof(x->ts), 1, fp);
|
||||
f_flag += fread(&(x->te), sizeof(x->te), 1, fp);
|
||||
f_flag += fread(&(x->el), sizeof(x->el), 1, fp);
|
||||
f_flag += fread(&(x->no_l_indel), sizeof(x->no_l_indel), 1, fp);
|
||||
|
||||
uint32_t t;
|
||||
f_flag += fread(&(t), sizeof(t), 1, fp);
|
||||
x->ml = t;
|
||||
|
||||
f_flag += fread(&(t), sizeof(t), 1, fp);
|
||||
x->rev = t;
|
||||
|
||||
f_flag += fread(&(t), sizeof(t), 1, fp);
|
||||
x->bl = t;
|
||||
|
||||
f_flag += fread(&(t), sizeof(t), 1, fp);
|
||||
x->del = t;
|
||||
}
|
||||
|
||||
|
||||
int append_All_reads(All_reads* r, char *idx, uint32_t id)
|
||||
{
|
||||
char *gfa_name = NULL; MALLOC(gfa_name, (strlen(idx)+100));
|
||||
FILE *fp = NULL, *fpo = NULL;
|
||||
|
||||
sprintf(gfa_name, "%s.ec%u.bin", idx, id);
|
||||
fp = fopen(gfa_name, "r");
|
||||
if (!fp) {free(gfa_name); return 0;}
|
||||
|
||||
sprintf(gfa_name, "%s.ovlp%u.source.bin", idx, id);
|
||||
fpo = fopen(gfa_name, "r");
|
||||
if (!fpo) {free(gfa_name); fclose(fp); return 0;}
|
||||
free(gfa_name);
|
||||
|
||||
int local_adapterLen, f_flag;
|
||||
f_flag = fread(&local_adapterLen, sizeof(local_adapterLen), 1, fp);
|
||||
if(local_adapterLen != asm_opt.adapterLen) {
|
||||
fprintf(stderr, "[M::%s] the adapterLen of index is: %d, but the adapterLen set by user is: %d\n",
|
||||
__func__, local_adapterLen, asm_opt.adapterLen);
|
||||
exit(1);
|
||||
}
|
||||
uint64_t index_size0, name_index_size0, total_reads0, total_reads_bases0, total_name_length0;
|
||||
index_size0 = r->index_size;
|
||||
total_reads0 = r->total_reads;
|
||||
total_reads_bases0 = r->total_reads_bases;
|
||||
total_name_length0 = r->total_name_length;
|
||||
name_index_size0 = ((total_reads0)?(total_reads0+1):(0));///r->name_index_size;
|
||||
|
||||
f_flag += fread(&r->index_size, sizeof(r->index_size), 1, fp);
|
||||
f_flag += fread(&r->name_index_size, sizeof(r->name_index_size), 1, fp);
|
||||
f_flag += fread(&r->total_reads, sizeof(r->total_reads), 1, fp);
|
||||
f_flag += fread(&r->total_reads_bases, sizeof(r->total_reads_bases), 1, fp);
|
||||
f_flag += fread(&r->total_name_length, sizeof(r->total_name_length), 1, fp);
|
||||
|
||||
r->index_size += index_size0;
|
||||
r->name_index_size += name_index_size0; if(name_index_size0) r->name_index_size--;
|
||||
r->total_reads += total_reads0;
|
||||
r->total_reads_bases += total_reads_bases0;
|
||||
r->total_name_length += total_name_length0;
|
||||
|
||||
|
||||
uint64_t i = 0, zero = 0, k;
|
||||
REALLOC(r->N_site, r->total_reads);
|
||||
|
||||
for (i = total_reads0; i < r->total_reads; i++) {
|
||||
f_flag += fread(&zero, sizeof(zero), 1, fp);
|
||||
|
||||
if (zero) {
|
||||
r->N_site[i] = (uint64_t*)malloc(sizeof(uint64_t)*(zero + 1));
|
||||
r->N_site[i][0] = zero;
|
||||
if (r->N_site[i][0]) {
|
||||
f_flag += fread(r->N_site[i]+1, sizeof(r->N_site[i][0]), r->N_site[i][0], fp);
|
||||
}
|
||||
} else {
|
||||
r->N_site[i] = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
REALLOC(r->read_length, r->total_reads);
|
||||
f_flag += fread(r->read_length + total_reads0, sizeof(uint64_t), r->total_reads-total_reads0, fp);
|
||||
|
||||
REALLOC(r->read_size, r->total_reads);
|
||||
memcpy (r->read_size + total_reads0, r->read_length + total_reads0, sizeof(uint64_t)*(r->total_reads-total_reads0));
|
||||
|
||||
REALLOC(r->read_sperate, r->total_reads);
|
||||
for (i = total_reads0; i < r->total_reads; i++) {
|
||||
r->read_sperate[i] = (uint8_t*)malloc(sizeof(uint8_t)*(r->read_length[i]/4+1));
|
||||
f_flag += fread(r->read_sperate[i], sizeof(uint8_t), r->read_length[i]/4+1, fp);
|
||||
}
|
||||
|
||||
|
||||
REALLOC(r->name, r->total_name_length);
|
||||
f_flag += fread(r->name + total_name_length0, sizeof(char), r->total_name_length - total_name_length0, fp);
|
||||
|
||||
REALLOC(r->name_index, r->name_index_size); uint64_t sft = 0;
|
||||
if(name_index_size0) sft = r->name_index[--name_index_size0];
|
||||
// fprintf(stderr, "name_index_size0::%lu, total_reads0::%lu, sft::%lu\n", name_index_size0, total_reads0, sft);
|
||||
f_flag += fread(r->name_index + name_index_size0, sizeof(uint64_t), r->name_index_size-name_index_size0, fp);
|
||||
if(sft) {
|
||||
for (i = name_index_size0; i < r->name_index_size; i++) r->name_index[i] += sft;
|
||||
}
|
||||
|
||||
/****************************may have bugs********************************/
|
||||
REALLOC(r->trio_flag, r->total_reads);
|
||||
f_flag += fread(r->trio_flag + total_reads0, sizeof(uint8_t), r->total_reads - total_reads0, fp);
|
||||
|
||||
int hom_cov0 = asm_opt.hom_cov, het_cov0 = asm_opt.het_cov;
|
||||
f_flag += fread(&(asm_opt.hom_cov), sizeof(asm_opt.hom_cov), 1, fp); asm_opt.hom_cov += hom_cov0;
|
||||
f_flag += fread(&(asm_opt.het_cov), sizeof(asm_opt.het_cov), 1, fp); asm_opt.het_cov += het_cov0;
|
||||
/****************************may have bugs********************************/
|
||||
|
||||
REALLOC(r->cigars, r->total_reads);
|
||||
REALLOC(r->second_round_cigar, r->total_reads);
|
||||
for (i = total_reads0; i < r->total_reads; i++) {
|
||||
r->second_round_cigar[i].size = r->cigars[i].size = 0;
|
||||
r->second_round_cigar[i].length = r->cigars[i].length = 0;
|
||||
r->second_round_cigar[i].record = r->cigars[i].record = NULL;
|
||||
|
||||
r->second_round_cigar[i].lost_base_size = r->cigars[i].lost_base_size = 0;
|
||||
r->second_round_cigar[i].lost_base_length = r->cigars[i].lost_base_length = 0;
|
||||
r->second_round_cigar[i].lost_base = r->cigars[i].lost_base = NULL;
|
||||
}
|
||||
///r->pb_regions = NULL;
|
||||
|
||||
REALLOC(r->paf, r->total_reads);
|
||||
memset(r->paf+total_reads0, 0, sizeof((*(r->paf)))*(r->total_reads - total_reads0));
|
||||
long long n_read; f_flag = fread(&n_read, sizeof(n_read), 1, fpo); ma_hit_t t;
|
||||
for (i = total_reads0; i < r->total_reads; i++) {
|
||||
f_flag += fread(&(r->paf[i].is_fully_corrected), sizeof(r->paf[i].is_fully_corrected), 1, fpo);
|
||||
f_flag += fread(&(r->paf[i].is_abnormal), sizeof(r->paf[i].is_abnormal), 1, fpo);
|
||||
f_flag += fread(&(r->paf[i].length), sizeof(r->paf[i].length), 1, fpo);
|
||||
|
||||
if(r->paf[i].length == 0) continue;
|
||||
for (k = 0; k < r->paf[i].length; k++) read_ma(&t, fpo);
|
||||
r->paf[i].length = 0;
|
||||
}
|
||||
|
||||
REALLOC(r->reverse_paf, r->total_reads);
|
||||
memset(r->reverse_paf+total_reads0, 0, sizeof((*(r->reverse_paf)))*(r->total_reads - total_reads0));
|
||||
|
||||
fclose(fp); fclose(fpo);
|
||||
fprintf(stderr, "Reads has been loaded.\n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
int destory_read_bin(All_reads* r)
|
||||
{
|
||||
|
||||
@@ -261,10 +500,18 @@ void malloc_All_reads(All_reads* r)
|
||||
memcpy(r->read_size, r->read_length, sizeof(uint64_t)*r->total_reads);
|
||||
|
||||
r->read_sperate = (uint8_t**)malloc(sizeof(uint8_t*)*r->total_reads);
|
||||
long long i = 0;
|
||||
for (i = 0; i < (long long)r->total_reads; i++)
|
||||
{
|
||||
r->read_sperate[i] = (uint8_t*)malloc(sizeof(uint8_t)*(r->read_length[i]/4+1));
|
||||
if(asm_opt.is_sc && r->tqn) MALLOC(r->rsc, r->tqn);
|
||||
assert(r->tqn <= r->total_reads);
|
||||
|
||||
uint64_t i = 0;
|
||||
if(r->rsc) {
|
||||
for (i = 0; i < r->tqn; i++) {
|
||||
MALLOC(r->read_sperate[i], (r->read_length[i]/4+1));
|
||||
MALLOC(r->rsc[i], (r->read_length[i]/sc_bn) + ((r->read_length[i]%sc_bn)?1:0));
|
||||
}
|
||||
}
|
||||
for (; i < r->total_reads; i++) {
|
||||
MALLOC(r->read_sperate[i], (r->read_length[i]/4+1));
|
||||
}
|
||||
|
||||
r->cigars = (Compressed_Cigar_record*)malloc(sizeof(Compressed_Cigar_record)*r->total_reads);
|
||||
@@ -273,8 +520,7 @@ void malloc_All_reads(All_reads* r)
|
||||
r->reverse_paf = (ma_hit_t_alloc*)malloc(sizeof(ma_hit_t_alloc)*r->total_reads);
|
||||
///r->pb_regions = (kvec_t_u64_warp*)malloc(r->total_reads*sizeof(kvec_t_u64_warp));
|
||||
|
||||
for (i = 0; i < (long long)r->total_reads; i++)
|
||||
{
|
||||
for (i = 0; i < r->total_reads; i++) {
|
||||
r->second_round_cigar[i].size = r->cigars[i].size = 0;
|
||||
r->second_round_cigar[i].length = r->cigars[i].length = 0;
|
||||
r->second_round_cigar[i].record = r->cigars[i].record = NULL;
|
||||
@@ -541,31 +787,27 @@ void recover_UC_Read(UC_Read* r, const All_reads *R_INF, uint64_t ID)
|
||||
r->length = Get_READ_LENGTH((*R_INF), ID);
|
||||
uint8_t* src = Get_READ((*R_INF), ID);
|
||||
|
||||
if (r->length + 4 > r->size)
|
||||
{
|
||||
if (r->length + 4 > r->size) {
|
||||
r->size = r->length + 4;
|
||||
r->seq = (char*)realloc(r->seq,sizeof(char)*(r->size));
|
||||
}
|
||||
|
||||
uint64_t i = 0;
|
||||
|
||||
while ((long long)i < r->length)
|
||||
{
|
||||
if(src) {
|
||||
while ((long long)i < r->length) {
|
||||
memcpy(r->seq+i, bit_t_seq_table[src[i>>2]], 4);
|
||||
i = i + 4;
|
||||
}
|
||||
|
||||
|
||||
if (R_INF->N_site[ID])
|
||||
{
|
||||
for (i = 1; i <= R_INF->N_site[ID][0]; i++)
|
||||
{
|
||||
r->seq[R_INF->N_site[ID][i]] = 'N';
|
||||
if (R_INF->N_site[ID]) {
|
||||
for (i = 1; i <= R_INF->N_site[ID][0]; i++) r->seq[R_INF->N_site[ID][i]] = 'N';
|
||||
}
|
||||
} else {///N
|
||||
memset(r->seq, 'N', r->length);
|
||||
}
|
||||
|
||||
r->RID = ID;
|
||||
|
||||
}
|
||||
|
||||
void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID)
|
||||
@@ -573,8 +815,7 @@ void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID)
|
||||
r->length = Get_READ_LENGTH((*R_INF), ID);
|
||||
uint8_t* src = Get_READ((*R_INF), ID);
|
||||
|
||||
if (r->length + 4 > r->size)
|
||||
{
|
||||
if (r->length + 4 > r->size) {
|
||||
r->size = r->length + 4;
|
||||
r->seq = (char*)realloc(r->seq,sizeof(char)*(r->size));
|
||||
}
|
||||
@@ -583,27 +824,29 @@ void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID)
|
||||
long long i = r->length / 4 - 1 + (last_chr != 0);
|
||||
long long index = 0;
|
||||
|
||||
if(last_chr!=0)
|
||||
{
|
||||
if(src) {
|
||||
if(last_chr!=0) {
|
||||
memcpy(r->seq + index, bit_t_seq_table_rc[src[i]] + 4 - last_chr, last_chr);
|
||||
index = last_chr;
|
||||
i--;
|
||||
}
|
||||
|
||||
while (i >= 0)
|
||||
{
|
||||
while (i >= 0) {
|
||||
memcpy(r->seq + index, bit_t_seq_table_rc[src[i]], 4);
|
||||
i--;
|
||||
index = index + 4;
|
||||
}
|
||||
|
||||
if (R_INF->N_site[ID])
|
||||
{
|
||||
for (i = 1; i <= (long long)R_INF->N_site[ID][0]; i++)
|
||||
{
|
||||
if (R_INF->N_site[ID]) {
|
||||
for (i = 1; i <= (long long)R_INF->N_site[ID][0]; i++) {
|
||||
r->seq[r->length - R_INF->N_site[ID][i] - 1] = 'N';
|
||||
}
|
||||
}
|
||||
} else {///N
|
||||
memset(r->seq, 'N', r->length);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
#define COMPRESS_BASE {c = seq_nt6_table[(uint8_t)src[i]];\
|
||||
@@ -675,6 +918,130 @@ void ha_compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_sit
|
||||
}
|
||||
}
|
||||
|
||||
void convert_qual(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitu, uint64_t rev, uint64_t sc_off)
|
||||
{
|
||||
uint64_t i = 0; uint8_t c = 0, sc;
|
||||
// fprintf(stderr, "\n[M::%s]\n", __func__);
|
||||
for (i = 0; i < src_l; i++) {
|
||||
for (c = 0, sc = ((uint8_t)src[i]) - sc_off; (c < bitu) && (sc_tb[c] < sc); c++);
|
||||
if(c >= bitu) c = bitu - 1;
|
||||
dest[(rev?(src_l-i-1):(i))] = c;
|
||||
// fprintf(stderr, "%u->%u\n", sc, c);
|
||||
}
|
||||
}
|
||||
|
||||
void ha_compress_qual_bit(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitn)
|
||||
{
|
||||
|
||||
uint64_t i = 0, k, bit_r = 8/bitn, dest_i = 0;
|
||||
uint8_t tmp = 0, c = 0;
|
||||
|
||||
for (i = 0; i + bit_r <= src_l;) {
|
||||
for (k = tmp = 0; k < bit_r; k++) {
|
||||
c = ((uint8_t)src[i]);
|
||||
tmp <<= bitn; tmp |= c; i++;
|
||||
}
|
||||
dest[dest_i++] = tmp;
|
||||
}
|
||||
|
||||
if(i < src_l) {
|
||||
for (k = tmp = 0; i < src_l; k++) {
|
||||
c = ((uint8_t)src[i]);
|
||||
tmp <<= bitn; tmp |= c; i++;
|
||||
}
|
||||
|
||||
dest[dest_i++] = (tmp<<(8-(bitn*k)));
|
||||
}
|
||||
}
|
||||
|
||||
void ha_compress_qual(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitn, uint64_t sc_off)
|
||||
{
|
||||
|
||||
uint64_t i = 0, k, bit_r = 8/bitn, dest_i = 0, bitu = (1<<bitn);
|
||||
uint8_t tmp = 0, c = 0, sc;
|
||||
|
||||
for (i = 0; i + bit_r <= src_l;) {
|
||||
for (k = tmp = 0; k < bit_r; k++) {
|
||||
for (c = 0, sc = ((uint8_t)src[i]) - sc_off; (c < bitu) && (sc_tb[c] < sc); c++);
|
||||
if(c >= bitu) c = bitu - 1;
|
||||
tmp <<= bitn; tmp |= c; i++;
|
||||
}
|
||||
dest[dest_i++] = tmp;
|
||||
}
|
||||
|
||||
if(i < src_l) {
|
||||
for (k = tmp = 0; i < src_l; k++) {
|
||||
for (c = 0, sc = ((uint8_t)src[i]) - sc_off; (c < bitu) && (sc_tb[c] < sc); c++);
|
||||
if(c >= bitu) c = bitu - 1;
|
||||
tmp <<= bitn; tmp |= c; i++;
|
||||
}
|
||||
|
||||
dest[dest_i++] = (tmp<<(8-(bitn*k)));
|
||||
}
|
||||
}
|
||||
|
||||
///[s, e)
|
||||
int64_t retrive_bqual(asg8_v *dv, uint8_t *ds, uint64_t id, int64_t s, int64_t e, uint8_t rev, int64_t bitn)
|
||||
{
|
||||
int64_t rl = Get_READ_LENGTH(R_INF, id), l;
|
||||
if(s < 0) {s = 0;} if(e < 0) {e = rl;}
|
||||
if(s >= e || e > rl) return -1;
|
||||
|
||||
uint8_t *da = NULL, *src = Get_QUAL(R_INF, id), mm = (((uint8_t)1)<<bitn)-1, mlf = 8 - bitn, mrf;
|
||||
int64_t bitr = 8/bitn, dk, sk, swk;
|
||||
l = e - s;
|
||||
if(dv) {
|
||||
kv_resize(uint8_t, *dv, ((uint64_t)l)); da = dv->a; dv->n = l;
|
||||
} else {
|
||||
da = ds;
|
||||
}
|
||||
|
||||
if(!rev) {
|
||||
dk = 0; sk = s;
|
||||
|
||||
mrf = ((s%bitr)*bitn);
|
||||
// if(s == 21519 && e == 22332) {
|
||||
// fprintf(stderr, "+[M::%s] id::%lu, in::[%ld, %ld), rev::%u, bitn::%ld, bitr::%ld, mrf::%u\n", __func__, id, s, e, rev, bitn, bitr, mrf);
|
||||
// }
|
||||
if(mrf) {
|
||||
for (swk = sk/bitr; mrf < 8 && sk < e; mrf += bitn, sk++) da[dk++] = ((src[swk]<<mrf)>>mlf)&mm;
|
||||
}
|
||||
|
||||
for (swk = sk/bitr; (sk + bitr) <= e; sk += bitr, swk++) {
|
||||
for (mrf = 0; mrf < 8; mrf += bitn) da[dk++] = ((src[swk]<<mrf)>>mlf)&mm;
|
||||
}
|
||||
|
||||
if(sk < e) {
|
||||
for (mrf = 0; sk < e; mrf += bitn, sk++) da[dk++] = ((src[swk]<<mrf)>>mlf)&mm;
|
||||
}
|
||||
// if(dk != l) {
|
||||
// fprintf(stderr, "+[M::%s] id::%lu, in::[%ld, %ld), rev::%u, bitn::%ld, bitr::%ld\n", __func__, id, s, e, rev, bitn, bitr);
|
||||
// }
|
||||
assert(dk == l);
|
||||
} else {
|
||||
sk = s; s = e; e = sk;
|
||||
s = rl - s; e = rl - e;
|
||||
dk = l; sk = s;
|
||||
|
||||
mrf = ((s%bitr)*bitn);
|
||||
if(mrf) {
|
||||
for (swk = sk/bitr; mrf < 8 && sk < e; mrf += bitn, sk++) da[--dk] = ((src[swk]<<mrf)>>mlf)&mm;
|
||||
}
|
||||
|
||||
for (swk = sk/bitr; (sk + bitr) <= e; sk += bitr, swk++) {
|
||||
for (mrf = 0; mrf < 8; mrf += bitn) da[--dk] = ((src[swk]<<mrf)>>mlf)&mm;
|
||||
}
|
||||
|
||||
if(sk < e) {
|
||||
for (mrf = 0; sk < e; mrf += bitn, sk++) da[--dk] = ((src[swk]<<mrf)>>mlf)&mm;
|
||||
}
|
||||
|
||||
assert(dk == 0);
|
||||
}
|
||||
|
||||
return l;
|
||||
}
|
||||
|
||||
void reverse_complement(char* pattern, uint64_t length)
|
||||
{
|
||||
uint64_t i = 0;
|
||||
@@ -697,6 +1064,24 @@ void reverse_complement(char* pattern, uint64_t length)
|
||||
}
|
||||
}
|
||||
|
||||
void print_fastq(FILE *fp, char *id, char *bs, char *qual, uint64_t bitu, uint64_t sc_off)
|
||||
{
|
||||
uint64_t i = 0, ql = strlen(qual); uint8_t c = 0, sc;
|
||||
|
||||
if(fp) fprintf(fp, "@%s\n%s\n+\n", id, bs);
|
||||
else fprintf(stdout, "@%s\n%s\n+\n", id, bs);
|
||||
|
||||
for (i = 0; i < ql; i++) {
|
||||
for (c = 0, sc = ((uint8_t)qual[i]) - sc_off; (c < bitu) && (sc_tb[c] < sc); c++);
|
||||
if(c >= bitu) c = bitu - 1;
|
||||
if(fp) fprintf(fp, "%u", c);
|
||||
else fprintf(stdout, "%u", c);
|
||||
}
|
||||
|
||||
if(fp) fprintf(fp, "\n");
|
||||
else fprintf(stdout, "\n");
|
||||
}
|
||||
|
||||
|
||||
void init_Debug_reads(Debug_reads* x, const char* file)
|
||||
{
|
||||
@@ -714,6 +1099,8 @@ void init_Debug_reads(Debug_reads* x, const char* file)
|
||||
}
|
||||
x->read_name = (char**)malloc(sizeof(char*)*x->query_num);
|
||||
x->candidate_count = (kvec_t_u64_warp*)malloc(sizeof(kvec_t_u64_warp)*x->query_num);
|
||||
x->read_id = NULL; MALLOC(x->read_id, x->query_num);
|
||||
memset(x->read_id, -1, sizeof((*(x->read_id)))*x->query_num);
|
||||
fseek(x->fp, 0, SEEK_SET);
|
||||
|
||||
i = 0;
|
||||
@@ -732,6 +1119,15 @@ void init_Debug_reads(Debug_reads* x, const char* file)
|
||||
sprintf(Name_Buffer, "%s.debug.stdout", file);
|
||||
x->fp = fopen(Name_Buffer,"w");
|
||||
fprintf(stderr, "Print debugging information to: %s\n", Name_Buffer);
|
||||
|
||||
sprintf(Name_Buffer, "%s.debug.r0.fa", file);
|
||||
x->fp_r0 = fopen(Name_Buffer,"w");
|
||||
fprintf(stderr, "Print raw reads to: %s\n", Name_Buffer);
|
||||
|
||||
sprintf(Name_Buffer, "%s.debug.r1.fa", file);
|
||||
x->fp_r1 = fopen(Name_Buffer,"w");
|
||||
fprintf(stderr, "Print corrected reads to: %s\n", Name_Buffer);
|
||||
|
||||
free(Name_Buffer);
|
||||
}
|
||||
|
||||
@@ -744,8 +1140,8 @@ void destory_Debug_reads(Debug_reads* x)
|
||||
kv_destroy(x->candidate_count[i].a);
|
||||
}
|
||||
|
||||
free(x->read_name);
|
||||
fclose(x->fp);
|
||||
free(x->read_name); free(x->read_id);
|
||||
fclose(x->fp); fclose(x->fp_r0); fclose(x->fp_r1);
|
||||
}
|
||||
|
||||
|
||||
@@ -1171,7 +1567,7 @@ void append_ul_t(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str,
|
||||
p->rlen = str_l;
|
||||
// fprintf(stderr, "str_l->%ld, str->%u\n", str_l, str?1:0);
|
||||
|
||||
if(o == NULL || on == 0) on = 0; en = 0;
|
||||
if(o == NULL || on == 0) {on = 0;} en = 0;
|
||||
for (i = on-1, st = et = str_l; i >= 0; i--) {
|
||||
z = &(o[i]);
|
||||
if(z->el) {
|
||||
@@ -1447,6 +1843,62 @@ void retrieve_u_seq(UC_Read* i_r, char* i_s, ma_utg_t *u, uint8_t strand, int64_
|
||||
}
|
||||
}
|
||||
|
||||
void retrieve_u_seq_fast(UC_Read* i_r, char* i_s, ma_utg_t *u, uint8_t strand, int64_t s, int64_t l, void *km)
|
||||
{
|
||||
if(u->m == 0 || u->n == 0) return;
|
||||
if(l < 0) l = u->len;
|
||||
char *r = NULL, *a = NULL;
|
||||
int64_t e = s + l, ssp, sep, rs, re, des_i;
|
||||
uint64_t k, rId, ori, r_l;
|
||||
if(i_r) {
|
||||
i_r->length = l; i_r->RID = 0;
|
||||
if(i_r->length > i_r->size) {
|
||||
i_r->size = i_r->length;
|
||||
if(!km) REALLOC(i_r->seq, i_r->size);
|
||||
else KREALLOC(km, i_r->seq, i_r->size);
|
||||
// i_r->seq = (char*)realloc(i_r->seq,sizeof(char)*(i_r->size));
|
||||
}
|
||||
r = i_r->seq;
|
||||
}
|
||||
if(i_s) r = i_s;
|
||||
|
||||
if(u->s) {
|
||||
memcpy(r, u->s + s, l * sizeof((*(u->s))));
|
||||
} else {
|
||||
if(strand == 1) {
|
||||
sep = u->len - s;
|
||||
ssp = u->len - e;
|
||||
s = ssp; e = sep;
|
||||
}
|
||||
for (k = l = des_i = 0; k < u->n; k++) {
|
||||
rId = u->a[k]>>33;
|
||||
ori = u->a[k]>>32&1;
|
||||
r_l = (uint32_t)u->a[k];
|
||||
if(r_l == 0) continue;
|
||||
ssp = l; sep = l + r_l;
|
||||
l += r_l;
|
||||
if(sep <= s) continue;
|
||||
if(ssp >= e) break;
|
||||
rs = MAX(ssp, s); re = MIN(sep, e);
|
||||
a = r + des_i; des_i += re - rs;
|
||||
recover_UC_Read_sub_region(a, rs-ssp, re-rs, ori, &R_INF, rId);
|
||||
}
|
||||
}
|
||||
|
||||
if(strand == 1) {
|
||||
char t;
|
||||
re = (e - s);
|
||||
l = re>>1;
|
||||
for (k = 0; k < (uint64_t)l; k++) {
|
||||
des_i = re - k - 1;
|
||||
t = r[des_i];
|
||||
r[des_i] = RC_CHAR(r[k]);
|
||||
r[k] = RC_CHAR(t);
|
||||
}
|
||||
if(re&1) r[l] = RC_CHAR(r[l]);
|
||||
}
|
||||
}
|
||||
|
||||
uint32_t retrieve_u_cov(const ul_idx_t *ul, uint64_t id, uint8_t strand, uint64_t pos, uint8_t dir, int64_t *pi)
|
||||
{
|
||||
uint64_t *a = ul->cc->interval.a + ul->cc->idx[id], cc = 0, ff = 0;
|
||||
@@ -1883,3 +2335,23 @@ int64_t load_compress_base_disk(FILE *fp, uint64_t *ul_rid, char *dest, uint32_t
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
scaf_res_t *init_scaf_res_t(uint32_t n)
|
||||
{
|
||||
scaf_res_t *p = NULL; CALLOC(p, 1);
|
||||
p->n = p->m = n; CALLOC(p->a, n);
|
||||
return p;
|
||||
}
|
||||
|
||||
void destroy_scaf_res_t(scaf_res_t *p)
|
||||
{
|
||||
if(p) {
|
||||
uint32_t k;
|
||||
for (k = 0; k < p->m; k++) {
|
||||
free(p->a[k].N_site.a); free(p->a[k].r_base.a); free(p->a[k].bb.a);
|
||||
}
|
||||
free(p->a);
|
||||
free(p);
|
||||
}
|
||||
}
|
||||
+46
-1
@@ -8,6 +8,7 @@
|
||||
#include <zlib.h>
|
||||
#include "Overlaps.h"
|
||||
#include "CommandLines.h"
|
||||
#include "Levenshtein_distance.h"
|
||||
///#include "Hash_Table.h"
|
||||
|
||||
#define READ_INIT_NUMBER 1000
|
||||
@@ -22,9 +23,11 @@
|
||||
#define Get_NAME_LENGTH(R_INF, ID) ((R_INF).name_index[(ID)+1] - (R_INF).name_index[(ID)])
|
||||
///#define Get_READ(R_INF, ID) R_INF.read + (R_INF.index[ID]>>2) + ID
|
||||
#define Get_READ(R_INF, ID) (R_INF).read_sperate[(ID)]
|
||||
#define Get_QUAL(R_INF, ID) (R_INF).rsc[(ID)]
|
||||
#define Get_NAME(R_INF, ID) ((R_INF).name + (R_INF).name_index[(ID)])
|
||||
#define CHECK_BY_NAME(R_INF, NAME, ID) (Get_NAME_LENGTH((R_INF),(ID))==strlen((NAME)) && \
|
||||
memcmp((NAME), Get_NAME((R_INF), (ID)), Get_NAME_LENGTH((R_INF),(ID))) == 0)
|
||||
#define IS_SCAF_READ(R_INF, ID) ((R_INF).read_sperate[(ID)] == NULL)
|
||||
|
||||
extern uint8_t seq_nt6_table[256];
|
||||
extern char bit_t_seq_table[256][4];
|
||||
@@ -37,6 +40,8 @@ extern char rc_Table[6];
|
||||
|
||||
void init_aux_table();
|
||||
|
||||
typedef struct { size_t n, m; uint8_t *a; } asg8_v;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
uint64_t x_id;
|
||||
@@ -106,6 +111,8 @@ typedef struct
|
||||
#define CHAIN_MATCH 1
|
||||
#define CHAIN_UNMATCH 0.334
|
||||
|
||||
#define NEC 1
|
||||
|
||||
typedef struct
|
||||
{
|
||||
uint64_t** N_site;
|
||||
@@ -116,6 +123,7 @@ typedef struct
|
||||
uint64_t* read_length;
|
||||
uint64_t* read_size;
|
||||
uint8_t* trio_flag;
|
||||
uint8_t** rsc;
|
||||
|
||||
///seq start pos in uint8_t* read
|
||||
///do not need it
|
||||
@@ -126,8 +134,10 @@ typedef struct
|
||||
uint64_t* name_index;
|
||||
uint64_t name_index_size;
|
||||
uint64_t total_reads;
|
||||
uint64_t tqn;
|
||||
uint64_t total_reads_bases;
|
||||
uint64_t total_name_length;
|
||||
uint64_t tr[2];
|
||||
|
||||
Compressed_Cigar_record* cigars;
|
||||
Compressed_Cigar_record* second_round_cigar;
|
||||
@@ -135,11 +145,16 @@ typedef struct
|
||||
ma_hit_t_alloc* paf;
|
||||
ma_hit_t_alloc* reverse_paf;
|
||||
|
||||
uint8_t is_syn;
|
||||
|
||||
///kvec_t_u64_warp* pb_regions;
|
||||
} All_reads;
|
||||
|
||||
extern All_reads R_INF;
|
||||
|
||||
typedef struct {size_t n, m; asg16_v *a; uint8_t *f; uint16_t *er; uint64_t bid;} cc_v;
|
||||
extern cc_v scb;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
char* seq;
|
||||
@@ -151,9 +166,10 @@ typedef struct
|
||||
typedef struct
|
||||
{
|
||||
char** read_name;
|
||||
uint64_t *read_id;
|
||||
uint64_t query_num;
|
||||
kvec_t_u64_warp* candidate_count;
|
||||
FILE* fp;
|
||||
FILE *fp, *fp_r0, *fp_r1;
|
||||
pthread_mutex_t OutputMutex;
|
||||
} Debug_reads;
|
||||
|
||||
@@ -203,14 +219,23 @@ typedef struct
|
||||
// uint32_t mm;
|
||||
} all_ul_t;
|
||||
|
||||
|
||||
typedef struct {
|
||||
ul_vec_t *a;
|
||||
size_t n, m;
|
||||
uint8_t dd;
|
||||
} scaf_res_t;
|
||||
|
||||
extern all_ul_t UL_INF;
|
||||
extern all_ul_t ULG_INF;
|
||||
// extern uint32_t *het_cnt;
|
||||
// extern uint32_t debug_out;
|
||||
|
||||
void init_All_reads(All_reads* r);
|
||||
void malloc_All_reads(All_reads* r);
|
||||
void ha_insert_read_len(All_reads *r, int read_len, int name_len);
|
||||
void ha_compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_lis, uint64_t N_site_occ);
|
||||
void ha_compress_qual(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitn, uint64_t sc_off);
|
||||
void init_UC_Read(UC_Read* r);
|
||||
void recover_UC_Read(UC_Read* r, const All_reads *R_INF, uint64_t ID);
|
||||
void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID);
|
||||
@@ -219,6 +244,9 @@ void destory_UC_Read(UC_Read* r);
|
||||
void reverse_complement(char* pattern, uint64_t length);
|
||||
void write_All_reads(All_reads* r, char* read_file_name);
|
||||
int load_All_reads(All_reads* r, char* read_file_name);
|
||||
uint8_t load_cc_v(cc_v* r, char* read_file_name);
|
||||
void write_cc_v(cc_v* r, char* read_file_name);
|
||||
int append_All_reads(All_reads* r, char *idx, uint32_t id);
|
||||
void destory_All_reads(All_reads* r);
|
||||
int destory_read_bin(All_reads* r);
|
||||
void init_Debug_reads(Debug_reads* x, const char* file);
|
||||
@@ -237,5 +265,22 @@ uint64_t retrieve_r_cov_region(const ul_idx_t *ul, uint64_t id, uint8_t strand,
|
||||
void append_ul_t_back(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str, int64_t str_l, ul_ov_t *o, int64_t on, float p_chain_rate);
|
||||
void write_compress_base_disk(FILE *fp, uint64_t ul_rid, char *str, uint32_t len, ul_vec_t *buf);
|
||||
int64_t load_compress_base_disk(FILE *fp, uint64_t *ul_rid, char *dest, uint32_t *len, ul_vec_t *buf);
|
||||
scaf_res_t *init_scaf_res_t(uint32_t n);
|
||||
void destroy_scaf_res_t(scaf_res_t *p);
|
||||
void read_ma(ma_hit_t* x, FILE* fp);
|
||||
void retrieve_u_seq_fast(UC_Read* i_r, char* i_s, ma_utg_t *u, uint8_t strand, int64_t s, int64_t l, void *km);
|
||||
|
||||
const uint64_t sc_tb[8] = {
|
||||
10, 20, 30, 40, 50, 60, 70, 80
|
||||
};
|
||||
|
||||
#define sc_bn 2
|
||||
#define sc_bm ((((uint64_t)1)<<sc_bn)-1)
|
||||
#define sc_wn 5
|
||||
|
||||
void convert_qual(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitu, uint64_t rev, uint64_t sc_off);
|
||||
int64_t retrive_bqual(asg8_v *dv, uint8_t *ds, uint64_t id, int64_t s, int64_t e, uint8_t rev, int64_t bitn);
|
||||
void print_fastq(FILE *fp, char *id, char *bs, char *qual, uint64_t bitu, uint64_t sc_off);
|
||||
void ha_compress_qual_bit(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitn);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -15,6 +15,9 @@ hifiasm -o CHM13.asm -t32 -l0 CHM13-HiFi.fa.gz 2> CHM13.asm.log
|
||||
# Assemble heterozygous genomes with built-in duplication purging
|
||||
hifiasm -o HG002.asm -t32 HG002-file1.fq.gz HG002-file2.fq.gz
|
||||
|
||||
# Assemble genomes with ONT R10 reads rather than PacBio HiFi reads using the latest release of hifiasm (>0.21.0-r686)
|
||||
hifiasm -o HG002.asm --ont -t32 HG002-ont.fq.gz
|
||||
|
||||
# Hi-C phasing with paired-end short reads in two FASTQ files
|
||||
hifiasm -o HG002.asm --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
||||
|
||||
@@ -22,6 +25,19 @@ hifiasm -o HG002.asm --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
||||
yak count -b37 -t16 -o pat.yak <(cat pat_1.fq.gz pat_2.fq.gz) <(cat pat_1.fq.gz pat_2.fq.gz)
|
||||
yak count -b37 -t16 -o mat.yak <(cat mat_1.fq.gz mat_2.fq.gz) <(cat mat_1.fq.gz mat_2.fq.gz)
|
||||
hifiasm -o HG002.asm -t32 -1 pat.yak -2 mat.yak HG002-HiFi.fa.gz
|
||||
|
||||
# Improve contiguity for diploid genome assembly by self-scaffolding (`--dual-scaf`)
|
||||
hifiasm -o HG002.asm --dual-scaf --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
||||
|
||||
# Preserve more telomeres for human genomes (`--telo-m CCCTAA`)
|
||||
hifiasm -o HG002.asm --telo-m CCCTAA --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
||||
|
||||
# Hybrid assembly with HiFi, ultralong and Hi-C reads
|
||||
hifiasm -o HG002.asm --h1 read1.fq.gz --h2 read2.fq.gz --ul ul.fq.gz HG002-HiFi.fq.gz
|
||||
|
||||
# Single-sample telomere-to-telomere assembly for diploid human genomes
|
||||
hifiasm -o HG002.asm --dual-scaf --telo-m CCCTAA --h1 read1.fq.gz --h2 read2.fq.gz --ul ul.fq.gz HG002-HiFi.fq.gz
|
||||
|
||||
```
|
||||
See [tutorial][tutorial] for more details.
|
||||
|
||||
@@ -32,6 +48,7 @@ See [tutorial][tutorial] for more details.
|
||||
- [Why Hifiasm?](#why)
|
||||
- [Usage](#use)
|
||||
- [Assembling HiFi reads without additional data types](#hifionly)
|
||||
- [Assembling ONT reads](#ontonly)
|
||||
- [Hi-C integration](#hic)
|
||||
- [Trio binning](#trio)
|
||||
- [Ultra-long ONT integration](#ul)
|
||||
@@ -43,15 +60,12 @@ See [tutorial][tutorial] for more details.
|
||||
|
||||
## <a name="intro"></a>Introduction
|
||||
|
||||
Hifiasm is a fast haplotype-resolved de novo assembler for PacBio HiFi reads.
|
||||
It can assemble a human genome in several hours and assemble a ~30Gb California
|
||||
redwood genome in a few days. Hifiasm emits partially phased assemblies of
|
||||
quality competitive with the best assemblers. Given parental short reads or
|
||||
Hi-C data, it produces arguably the best haplotype-resolved assemblies so far.
|
||||
Hifiasm is a fast haplotype-resolved de novo assembler initially designed for PacBio HiFi reads.
|
||||
Its latest release could support the telomere-to-telomere assembly by utilizing ultralong Oxford Nanopore reads. Hifiasm produces arguably the best single-sample telomere-to-telomere assemblies combing HiFi, ultralong and Hi-C reads, and it is one of the best haplotype-resolved assemblers for the trio-binning assembly given parental short reads. For a human genome, hifiasm can produce the telomere-to-telomere assembly in one day.
|
||||
|
||||
## <a name="why"></a>Why Hifiasm?
|
||||
|
||||
* Hifiasm delivers high-quality assemblies. It tends to generate longer contigs
|
||||
* Hifiasm delivers high-quality telomere-to-telomere assemblies. It tends to generate longer contigs
|
||||
and resolve more segmental duplications than other assemblers.
|
||||
|
||||
* Given Hi-C reads or short reads from the parents, hifiasm can produce overall the best
|
||||
@@ -106,6 +120,16 @@ bloom filter which takes 16GB memory at the beginning. For genomes much larger
|
||||
than human, applying `-f38` or even `-f39` is preferred to save memory on k-mer
|
||||
counting.
|
||||
|
||||
### <a name="ontonly"></a>Assembling ONT reads
|
||||
|
||||
Since version 0.21.0 (r686), hifiasm can support ONT assembly using ONT simplex R10 reads.
|
||||
To enable this feature, add the `--ont` option as shown below:
|
||||
```sh
|
||||
hifiasm -t64 --ont -o ONT.asm ONT.read.fastq.gz
|
||||
```
|
||||
Please note that this module requires input reads in FASTQ format.
|
||||
|
||||
|
||||
### <a name="hic"></a>Hi-C integration
|
||||
|
||||
Hifiasm can generate a pair of haplotype-resolved assemblies with paired-end
|
||||
@@ -146,11 +170,36 @@ The second command line will run much faster than the first.
|
||||
|
||||
### <a name="ul"></a>Ultra-long ONT integration
|
||||
|
||||
Hifiasm could integrate ultra-long ONT reads to improve the assembly quality:
|
||||
Hifiasm could integrate ultra-long ONT reads to produce the telomere-to-telomere assembly:
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t32 --ul ul.fq.gz HiFi-reads.fq.gz
|
||||
```
|
||||
Please note that this mode is not stable right now. We have only tested with >=100kb UL reads.
|
||||
For the single-sample telomere-to-telomere assembly with Hi-C reads:
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t32 --ul ul.fq.gz --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
|
||||
```
|
||||
For the trio-binning telomere-to-telomere assembly:
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t32 --ul ul.fq.gz -1 pat.yak -2 mat.yak HiFi-reads.fq.gz
|
||||
```
|
||||
|
||||
### <a name="ul"></a>Self-scaffolding
|
||||
|
||||
For diploid haplotype-resolved genome assembly, hifiasm can further enhance assembly contiguity
|
||||
by introducing scaffolding. It leverages the assemblies of the two haplotypes to scaffold each other.
|
||||
Specifically, if there is a gap within the haplotype 1 assembly, hifiasm will use the corresponding
|
||||
homologous region in haplotype 2 to scaffold haplotype 1. Below is an example using the `--dual-scaf` option.
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t32 --dual-scaf HiFi-reads.fq.gz
|
||||
```
|
||||
|
||||
### <a name="ul"></a>Preserve more telomeres for T2T assemblies
|
||||
|
||||
Hifiasm can preserve more telomeres by specifying the telomere motif using the `--telo-m` option.
|
||||
Below is an example applied to human genome assembly.
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t32 --telo-m CCCTAA HiFi-reads.fq.gz
|
||||
```
|
||||
|
||||
### <a name="output"></a>Output files
|
||||
|
||||
@@ -240,3 +289,8 @@ If you use hifiasm in your work, please cite:
|
||||
> Haplotype-resolved assembly of diploid genomes without parental data.
|
||||
> *Nature Biotechnology*, **40**:1332–1335.
|
||||
> https://doi.org/10.1038/s41587-022-01261-x
|
||||
|
||||
> Cheng, H., Asri, M., Lucas, J., Koren, S., Li, H. (2024)
|
||||
> Scalable telomere-to-telomere assembly for diploid and polyploid genomes with double graph.
|
||||
> *Nat Methods*, **21**:967-970.
|
||||
> https://doi.org/10.1038/s41592-024-02269-8
|
||||
|
||||
@@ -388,6 +388,31 @@ inline void phrase_hstatus(char *s, char **rname, uint32_t *hid)
|
||||
*hid = atoi(id);
|
||||
}
|
||||
|
||||
inline void phrase_hchar(char *s, char **rname, uint32_t *hid)
|
||||
{
|
||||
*rname = NULL; *hid = (uint32_t)-1;
|
||||
uint32_t tot = 0, k, l, z, sl = strlen(s);
|
||||
for (k = 1, l = 0; k <= sl; k++) {
|
||||
if((k == sl) || (s[k] == '\t') || (s[k] == ' ')) {
|
||||
if(k > l) {
|
||||
s[k] = 0;
|
||||
if(tot == 0) {
|
||||
*rname = s + l;
|
||||
} else if(tot == 1) {
|
||||
for (z = l; (z < k) && (s[z] >= '0') && (s[z] <= '9'); ++z);
|
||||
if(z < k) {
|
||||
fprintf(stderr, "ERROR: wrong hap id\n");
|
||||
return;
|
||||
}
|
||||
*hid = atoi(s + l);
|
||||
}
|
||||
tot++;
|
||||
}
|
||||
l = k + 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
uint32_t *ha_polybin_list(const hifiasm_opt_t *opt)
|
||||
{
|
||||
int64_t i;
|
||||
@@ -447,6 +472,74 @@ uint32_t *ha_polybin_list(const hifiasm_opt_t *opt)
|
||||
return ss;
|
||||
}
|
||||
|
||||
uint32_t *ha_charbin_list(const hifiasm_opt_t *opt, uint8_t **idx, uint32_t *idx_n)
|
||||
{
|
||||
int64_t i;
|
||||
khint_t k;
|
||||
cstr_ht_t *h; (*idx) = NULL; *idx_n = 0;
|
||||
assert(R_INF.total_reads < (uint32_t)-1);
|
||||
h = cstr_ht_init();
|
||||
for (i = 0; i < (int64_t)R_INF.total_reads; ++i) {
|
||||
int absent;
|
||||
char *str = (char*)calloc(Get_NAME_LENGTH(R_INF, i) + 1, 1);
|
||||
strncpy(str, Get_NAME(R_INF, i), Get_NAME_LENGTH(R_INF, i));
|
||||
k = cstr_ht_put(h, str, &absent);
|
||||
if (absent) kh_val(h, k) = i;
|
||||
}
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] created the hash table for read names\n", __func__, yak_realtime(), yak_cpu_usage());
|
||||
|
||||
gzFile fp;
|
||||
kstream_t *ks;
|
||||
kstring_t str = {0,0,0};
|
||||
char *rname = NULL;
|
||||
uint32_t hid, mhid = 0, *ss = NULL;
|
||||
int dret;
|
||||
int64_t n_tot = 0, n_bin = 0;
|
||||
fp = gzopen(opt->fn_chr_bin, "r");
|
||||
if (fp == 0) {
|
||||
fprintf(stderr, "ERROR: failed to open file '%s'\n", opt->fn_chr_bin);
|
||||
for (k = 0; k < kh_end(h); ++k)
|
||||
if (kh_exist(h, k))
|
||||
free((char*)kh_key(h, k));
|
||||
cstr_ht_destroy(h);
|
||||
return NULL;
|
||||
}
|
||||
MALLOC(ss, R_INF.total_reads); memset(ss, -1, sizeof((*ss))*R_INF.total_reads);
|
||||
ks = ks_init(fp);
|
||||
while (ks_getuntil(ks, KS_SEP_LINE, &str, &dret) >= 0) {
|
||||
khint_t k; ++n_tot;
|
||||
phrase_hchar(str.s, &rname, &hid);
|
||||
if((!(*rname)) || hid == (uint32_t)-1) {
|
||||
fprintf(stderr, "ERROR: wrong hap id\n");
|
||||
continue;
|
||||
}
|
||||
k = cstr_ht_get(h, rname);
|
||||
if (k != kh_end(h)) {
|
||||
ss[kh_val(h, k)] = hid;
|
||||
if(hid > mhid) mhid = hid;
|
||||
++n_bin;
|
||||
// fprintf(stderr, "%s\t%u\trid::%ld\n", rname, hid, kh_val(h, k));
|
||||
}
|
||||
}
|
||||
free(str.s);
|
||||
ks_destroy(ks);
|
||||
gzclose(fp);
|
||||
|
||||
for (k = 0; k < kh_end(h); ++k)
|
||||
if (kh_exist(h, k))
|
||||
free((char*)kh_key(h, k));
|
||||
cstr_ht_destroy(h);
|
||||
|
||||
mhid++; CALLOC((*idx), mhid);
|
||||
for (i = 0; i < (int64_t)R_INF.total_reads; ++i) {
|
||||
if(ss[i] >= mhid) continue;
|
||||
(*idx)[ss[i]] = 1;
|
||||
}
|
||||
*idx_n = mhid;
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> partitioned reads with external lists\n", __func__, yak_realtime(), yak_cpu_usage());
|
||||
return ss;
|
||||
}
|
||||
|
||||
void ha_triobin(const hifiasm_opt_t *opt)
|
||||
{
|
||||
memset(R_INF.trio_flag, AMBIGU, R_INF.total_reads * sizeof(uint8_t));
|
||||
|
||||
+2231
-5
File diff suppressed because it is too large
Load Diff
+9897
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,24 @@
|
||||
#ifndef __ECOVLP_PARSER__
|
||||
#define __ECOVLP_PARSER__
|
||||
|
||||
#define __STDC_LIMIT_MACROS
|
||||
#include <stdint.h>
|
||||
#include "Hash_Table.h"
|
||||
#include "Process_Read.h"
|
||||
#include "kdq.h"
|
||||
|
||||
|
||||
void prt_chain(overlap_region_alloc *o);
|
||||
void cal_ec_r(uint64_t n_thre, uint64_t round, uint64_t n_round, uint64_t n_a, uint64_t is_sv, uint64_t *tot_b, uint64_t *tot_e);
|
||||
void sl_ec_r(uint64_t n_thre, uint64_t n_a);
|
||||
void cal_ov_r(uint64_t n_thre, uint64_t n_a, uint64_t new_idx);
|
||||
void handle_chemical_r(uint64_t n_thre, uint64_t n_a);
|
||||
void handle_chemical_arc(uint64_t n_thre, uint64_t n_a);
|
||||
uint8_t* gen_chemical_arc_rf(uint64_t n_thre, uint64_t n_a);
|
||||
void cal_ec_r_dbg(uint64_t n_thre, uint64_t n_a);
|
||||
void write_ec_reads(const char *suffix_ou, cc_v *cvt, uint8_t is_rev);
|
||||
void destroy_cc_v(cc_v *z);
|
||||
void gen_gfa_bam(ma_ug_t *ug, uint64_t n_a);
|
||||
void clean_arc_rf(uint64_t n_thre, uint64_t n_a);
|
||||
|
||||
#endif
|
||||
+944
-180
File diff suppressed because it is too large
Load Diff
@@ -3,6 +3,7 @@
|
||||
#include "Overlaps.h"
|
||||
#include "hic.h"
|
||||
|
||||
#define is_contain_r(ri, z) (((z)<(ri).len)&&((ri).index[(z)]!=(uint32_t)(-1))&&(!((ri).index[(z)]>>31)))
|
||||
|
||||
typedef struct {
|
||||
asg_t *g;
|
||||
@@ -15,16 +16,16 @@ typedef struct {
|
||||
} sset_aux;
|
||||
|
||||
void ul_clean_gfa(ug_opt_t *uopt, asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, int64_t clean_round, double min_ovlp_drop_ratio, double max_ovlp_drop_ratio,
|
||||
double ou_drop_rate, int64_t max_tip, int64_t gap_fuzz, bub_label_t *b_mask_t, int32_t is_ou, int32_t is_trio, uint32_t ou_thres, char *o_file);
|
||||
uint32_t asg_arc_cut_tips(asg_t *g, uint32_t max_ext, asg64_v *in, uint32_t is_ou, R_to_U *ru);
|
||||
void asg_iterative_semi_circ(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t normal_len, uint32_t pop_chimer, asg64_v *dbg);
|
||||
void asg_arc_cut_chimeric(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t ou_thres);
|
||||
void asg_arc_cut_inexact(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, int32_t max_ext, uint32_t is_ou, uint32_t is_trio, float ou_rat/**, asg64_v *dbg**/);
|
||||
double ou_drop_rate, int64_t max_tip, int64_t gap_fuzz, bub_label_t *b_mask_t, int32_t is_ou, int32_t is_trio, uint32_t ou_thres, uint8_t *cmk, char *o_file);
|
||||
uint32_t asg_arc_cut_tips(asg_t *g, uint32_t max_ext, asg64_v *in, uint32_t is_ou, R_to_U *ru, telo_end_t *te);
|
||||
void asg_iterative_semi_circ(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t normal_len, uint32_t pop_chimer, asg64_v *dbg, telo_end_t *te);
|
||||
void asg_arc_cut_chimeric(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t ou_thres, telo_end_t *te);
|
||||
void asg_arc_cut_inexact(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, int32_t max_ext, uint32_t is_ou, uint32_t is_trio, uint32_t min_diff, float ou_rat/**, asg64_v *dbg**/);
|
||||
void asg_arc_cut_length(asg_t *g, asg64_v *in, int32_t max_ext, float len_rat, float ou_rat, uint32_t is_ou, uint32_t is_trio,
|
||||
uint32_t is_topo, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len);
|
||||
uint32_t is_topo, uint32_t min_diff, uint32_t min_ou, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len);
|
||||
void asg_arc_cut_bub_links(asg_t *g, asg64_v *in, float len_rat, float sec_len_rat, float ou_rat, uint32_t is_ou, uint64_t check_dist, ma_hit_t_alloc *rev, R_to_U* rI, int32_t max_ext);
|
||||
void asg_arc_cut_complex_bub_links(asg_t *g, asg64_v *in, float len_rat, float ou_rat, uint32_t is_ou, bub_label_t *b_mask_t);
|
||||
uint32_t asg_cut_large_indel(asg_t *g, asg64_v *in, int32_t max_ext, float ou_rat, uint32_t is_ou);
|
||||
uint32_t asg_cut_large_indel(asg_t *g, asg64_v *in, int32_t max_ext, float ou_rat, uint32_t is_ou, uint32_t min_diff);
|
||||
uint32_t asg_cut_semi_circ(asg_t *g, uint32_t lim_len, uint32_t is_clean);
|
||||
void ul_realignment_gfa(ug_opt_t *uopt, asg_t *sg, int64_t clean_round, double min_ovlp_drop_ratio,
|
||||
double max_ovlp_drop_ratio, int64_t max_tip, int64_t max_ul_tip, bub_label_t *b_mask_t, uint32_t is_trio, char *o_file, ul_renew_t *ropt, const char *bin_file, uint64_t free_uld, uint64_t is_bridg, uint64_t deep_clean);
|
||||
@@ -32,10 +33,13 @@ void recover_contain_g(asg_t *g, ma_hit_t_alloc *src, R_to_U* ruIndex, int64_t m
|
||||
void normalize_gou(asg_t *g);
|
||||
void prt_specfic_sge(asg_t *g, uint32_t src, uint32_t dst, const char* cmd);
|
||||
asg_t *gen_ng(ma_ug_t *ug, asg_t *sg, ug_opt_t *uopt, ma_sub_t **cov, R_to_U *ruI, uint64_t scaffold_len);
|
||||
void post_rescue(ug_opt_t *uopt, asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, bub_label_t *b_mask_t, long long no_trio_recover);
|
||||
void post_rescue(ug_opt_t *uopt, asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, bub_label_t *b_mask_t, long long no_trio_recover, uint8_t *cmk);
|
||||
// void print_raw_u2rgfa_seq(all_ul_t *aln, R_to_U* rI, uint32_t is_detail);
|
||||
bubble_type *gen_bubble_chain(asg_t *sg, ma_ug_t *ug, ug_opt_t *uopt, uint8_t **ir_het);
|
||||
bubble_type *gen_bubble_chain(asg_t *sg, ma_ug_t *ug, ug_opt_t *uopt, uint8_t **ir_het, uint8_t avoid_het);
|
||||
void filter_sg_by_ug(asg_t *rg, ma_ug_t *ug, ug_opt_t *uopt);
|
||||
void ug_ext_gfa(ug_opt_t *uopt, asg_t *sg, uint32_t max_len);
|
||||
void update_sg_uo(asg_t *g, ma_hit_t_alloc *src);
|
||||
uint32_t get_arcs(asg_t *g, uint32_t v, uint32_t* idx, uint32_t idx_n);
|
||||
uint64_t ug_occ_w(uint64_t is, uint64_t ie, ma_utg_t *u);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -2520,7 +2520,8 @@ void identify_bubbles(ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, kv_u_t
|
||||
// fprintf(stderr, "-bub->index[18759]: %u, bub->num.n: %u\n", (uint32_t)bub->index[18759], bub->num.n);
|
||||
}
|
||||
|
||||
uint32_t get_unitig_het_fly(ma_ug_t* ug, uint32_t uid, asg_t* sg, int64_t het_cov_thres,
|
||||
uint32_t get_unitig_het_fly(ma_ug_t* ug, uint32_t uid, asg_t* sg, /**int64_t het_cov_thres,**/
|
||||
int64_t m_het_cov_thres, int64_t m_hom_cov_thres,
|
||||
ma_hit_t_alloc* sources, R_to_U* ruIndex, uint8_t* r_flag, uint32_t m_het_occ, uint32_t m_het_label,
|
||||
uint32_t p_het_label, uint32_t n_het_label)
|
||||
{
|
||||
@@ -2596,8 +2597,10 @@ uint32_t p_het_label, uint32_t n_het_label)
|
||||
// fprintf(stderr, "[M::%s::uid->%u] u->n::%u, C_bases::%ld, R_bases::%ld, het_cov_thres::%ld, m_het_occ::%u\n",
|
||||
// __func__, uid, (uint32_t)u->n, C_bases, R_bases, het_cov_thres, m_het_occ);
|
||||
// }
|
||||
if((cov <= (het_cov_thres*1.333333)) && (u->n >= m_het_occ)) return m_het_label; ///must het
|
||||
if((cov >= (het_cov_thres*1.6))) return n_het_label; ///hom
|
||||
// if((cov <= (het_cov_thres*1.333333)) && (u->n >= m_het_occ)) return m_het_label; ///must het
|
||||
// if((cov >= (het_cov_thres*1.6))) return n_het_label; ///hom
|
||||
if((cov <= m_het_cov_thres) && (u->n >= m_het_occ)) return m_het_label; ///must het
|
||||
if(cov >= m_hom_cov_thres) return n_het_label; ///hom
|
||||
return p_het_label; ///potential het
|
||||
}
|
||||
|
||||
@@ -2608,9 +2611,20 @@ kv_u_trans_t *ref)
|
||||
if (!ug->g->is_symm) asg_symm(ug->g);
|
||||
uint32_t v, n_vtx = ug->g->n_seq * 2, i, k, mode = (((uint32_t)-1)<<2);
|
||||
uint32_t beg, sink, n, *a, n_occ;
|
||||
uint64_t pathLen;
|
||||
uint64_t pathLen, hom_cov, het_cov, m_het_cov, m_hom_cov;
|
||||
bub->ug = ug;
|
||||
bub->b_bub = bub->b_end_bub = bub->tangle_bub = bub->cross_bub = bub->mess_bub = 0;
|
||||
|
||||
if(asm_opt.hom_global_coverage_set) {
|
||||
hom_cov = asm_opt.hom_global_coverage;
|
||||
} else {
|
||||
hom_cov = ((double)asm_opt.hom_global_coverage)/((double)HOM_PEAK_RATE);
|
||||
}
|
||||
het_cov = hom_cov/asm_opt.polyploidy;
|
||||
m_het_cov = hom_cov - het_cov + (het_cov*0.333333);
|
||||
m_hom_cov = hom_cov - het_cov + (het_cov*0.6);
|
||||
// fprintf(stderr, "hom_cov::%lu, het_cov::%lu, m_het_cov::%lu, m_hom_cov::%lu\n", hom_cov, het_cov, m_het_cov, m_hom_cov);
|
||||
|
||||
if(bub->round_id == 0)
|
||||
{
|
||||
buf_t b; memset(&b, 0, sizeof(buf_t)); b.a = (binfo_t*)calloc(n_vtx, sizeof(binfo_t));
|
||||
@@ -2698,19 +2712,10 @@ kv_u_trans_t *ref)
|
||||
free(b.a); free(b.S.a); free(b.T.a); free(b.b.a); free(b.e.a);
|
||||
bub->f_bub = bub->num.n - 1; ///bub->s_bub = bub->num.n - 1;
|
||||
|
||||
|
||||
uint64_t dip_thre_max;
|
||||
memset(r_het_flag, 0, sizeof((*r_het_flag))*sg->n_seq);
|
||||
if(asm_opt.hom_global_coverage_set) {
|
||||
dip_thre_max = asm_opt.hom_global_coverage;
|
||||
} else {
|
||||
dip_thre_max = ((double)asm_opt.hom_global_coverage)/((double)HOM_PEAK_RATE);
|
||||
}
|
||||
// dip_thre_max = (double)(dip_thre_max) - (((double)(dip_thre_max)*0.5)/asm_opt.polyploidy);
|
||||
dip_thre_max = (double)(dip_thre_max) - ((double)(dip_thre_max)/asm_opt.polyploidy);
|
||||
for (i = 0; i < ug->g->n_seq; i++) {
|
||||
bub->index[i] = get_unitig_het_fly(ug, i, sg, dip_thre_max, sources, ruIndex,
|
||||
r_het_flag, 20, M_het(*bub), P_het(*bub), (uint32_t)-1);
|
||||
bub->index[i] = get_unitig_het_fly(ug, i, sg, m_het_cov, m_hom_cov,
|
||||
sources, ruIndex, r_het_flag, 20, M_het(*bub), P_het(*bub), (uint32_t)-1);
|
||||
}
|
||||
|
||||
for (i = 0; i < bub->f_bub; i++)
|
||||
@@ -2782,6 +2787,301 @@ kv_u_trans_t *ref)
|
||||
// fprintf(stderr, "-bub->index[18759]: %u, bub->num.n: %u\n", (uint32_t)bub->index[18759], bub->num.n);
|
||||
}
|
||||
|
||||
void reset_inner_bub_het_poy(asg_t* sg, ma_ug_t* ug, bubble_type* bub, uint32_t bid, uint64_t tLen, buf_t *b,
|
||||
uint64_t m_het_cov, uint64_t m_hom_cov, uint8_t *r_het_flag, ma_hit_t_alloc* sources, R_to_U* ruIndex)
|
||||
{
|
||||
uint32_t beg, sink, *ba, bn, m, v, z, socc;
|
||||
get_bubbles(bub, bid, &beg, &sink, &ba, &bn, NULL);
|
||||
for (m = 0; m < bn; m++) {
|
||||
v = ba[m];
|
||||
if(IF_HOM((v>>1), *bub)) continue;
|
||||
if(ug->g->seq[v>>1].del) continue;
|
||||
if(bub->index[v>>1] == (uint32_t)-1) continue;
|
||||
|
||||
if(get_unitig_het_fly(ug, v>>1, sg, m_het_cov, m_hom_cov, sources, ruIndex, r_het_flag, 20, 1, 0, (uint32_t)-1) == 1) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if(asg_arc_n(ug->g, v) < 2) continue;
|
||||
if(get_real_length(ug->g, v, NULL) < 2) continue;
|
||||
if(asg_bub_pop1_primary_trio(ug->g, NULL, v, tLen, b, (uint32_t)-1, (uint32_t)-1, 0, NULL, NULL, NULL, 0, 0, NULL)) {
|
||||
//beg is v, end is b.S.a[0]
|
||||
//note b.b include end, does not include beg
|
||||
for (z = socc = 0; z < b->b.n; z++) {
|
||||
if(b->b.a[z]==v || b->b.a[z]==b->S.a[0]) continue;
|
||||
socc += ug->u.a[b->b.a[z]>>1].n;
|
||||
}
|
||||
if((socc > 10) && (socc > (ug->u.a[v>>1].n*3))) {
|
||||
bub->index[v>>1] = (uint32_t)-1;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
v ^= 1;
|
||||
if(bub->index[v>>1] == (uint32_t)-1) continue;
|
||||
if(asg_arc_n(ug->g, v) < 2) continue;
|
||||
if(get_real_length(ug->g, v, NULL) < 2) continue;
|
||||
if(asg_bub_pop1_primary_trio(ug->g, NULL, v, tLen, b, (uint32_t)-1, (uint32_t)-1, 0, NULL, NULL, NULL, 0, 0, NULL)) {
|
||||
//beg is v, end is b.S.a[0]
|
||||
//note b.b include end, does not include beg
|
||||
for (z = socc = 0; z < b->b.n; z++) {
|
||||
if(b->b.a[z]==v || b->b.a[z]==b->S.a[0]) continue;
|
||||
socc += ug->u.a[b->b.a[z]>>1].n;
|
||||
}
|
||||
if((socc > 10) && (socc > (ug->u.a[v>>1].n*3))) {
|
||||
bub->index[v>>1] = (uint32_t)-1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
uint32_t deter_unitig_het_fly(ma_ug_t* ug, uint32_t uid, asg_t* sg, int64_t het_cov_thres,
|
||||
ma_hit_t_alloc* sources, R_to_U* ruIndex, uint8_t* r_flag)
|
||||
{
|
||||
ma_utg_t *u = &(ug->u.a[uid]);
|
||||
uint32_t k, i, j, rId, nv, tn, is_Unitig;
|
||||
asg_arc_t *av = NULL; ma_hit_t *h;
|
||||
int64_t R_bases = 0, C_bases = 0, cov;
|
||||
|
||||
///set
|
||||
u = &(ug->u.a[uid]);
|
||||
for (k = 0; k < u->n; k++) {
|
||||
rId = u->a[k]>>33;
|
||||
r_flag[rId] = 1;
|
||||
}
|
||||
for (i = 0; i < 2; i++) {
|
||||
nv = asg_arc_n(ug->g, (uid<<1)+i);
|
||||
av = asg_arc_a(ug->g, (uid<<1)+i);
|
||||
for (j = 0; j < nv; j++) {
|
||||
u = &(ug->u.a[av[j].v>>1]);
|
||||
for (k = 0; k < u->n; k++) {
|
||||
rId = u->a[k]>>33;
|
||||
r_flag[rId] = 2;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
u = &(ug->u.a[uid]);
|
||||
for (k = 0; k < u->n; k++) {
|
||||
if(u->a[k] == (uint64_t)-1) continue;
|
||||
rId = u->a[k]>>33;
|
||||
R_bases += sg->seq[rId].len;
|
||||
for (j = 0; j < (uint64_t)(sources[rId].length); j++) {
|
||||
h = &(sources[rId].buffer[j]);
|
||||
if(h->el != 1) continue;
|
||||
tn = Get_tn((*h));
|
||||
if(sg->seq[tn].del == 1) {
|
||||
///get the id of read that contains it
|
||||
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
|
||||
if(tn == (uint32_t)-1 || is_Unitig == 1 || sg->seq[tn].del == 1) continue;
|
||||
}
|
||||
if(sg->seq[tn].del == 1) continue;
|
||||
if(r_flag[tn] == 0) continue;
|
||||
if(r_flag[tn] == 1) {
|
||||
C_bases += (Get_qe((*h)) - Get_qs((*h)));
|
||||
}
|
||||
if(r_flag[tn] == 2) {
|
||||
C_bases += ((Get_qe((*h)) - Get_qs((*h)))/2);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
///reset
|
||||
u = &(ug->u.a[uid]);
|
||||
for (k = 0; k < u->n; k++) {
|
||||
rId = u->a[k]>>33;
|
||||
r_flag[rId] = 0;
|
||||
}
|
||||
for (i = 0; i < 2; i++) {
|
||||
nv = asg_arc_n(ug->g, (uid<<1)+i);
|
||||
av = asg_arc_a(ug->g, (uid<<1)+i);
|
||||
for (j = 0; j < nv; j++) {
|
||||
u = &(ug->u.a[av[j].v>>1]);
|
||||
for (k = 0; k < u->n; k++) {
|
||||
rId = u->a[k]>>33;
|
||||
r_flag[rId] = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
u = &(ug->u.a[uid]); cov = 0;
|
||||
if(R_bases > 0) cov = C_bases/R_bases;
|
||||
if(cov <= het_cov_thres) return 1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
void identify_bubbles_recal_poy(asg_t* sg, ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, ma_hit_t_alloc* sources, R_to_U* ruIndex,
|
||||
kv_u_trans_t *ref)
|
||||
{
|
||||
asg_cleanup(ug->g);
|
||||
if (!ug->g->is_symm) asg_symm(ug->g);
|
||||
uint32_t v, n_vtx = ug->g->n_seq * 2, i, k, mode = (((uint32_t)-1)<<2);
|
||||
uint32_t beg, sink, n, *a, n_occ;
|
||||
uint64_t pathLen, hom_cov, het_cov, m_het_cov, m_hom_cov;
|
||||
bub->ug = ug;
|
||||
bub->b_bub = bub->b_end_bub = bub->tangle_bub = bub->cross_bub = bub->mess_bub = 0;
|
||||
if(asm_opt.hom_global_coverage_set) {
|
||||
hom_cov = asm_opt.hom_global_coverage;
|
||||
} else {
|
||||
hom_cov = ((double)asm_opt.hom_global_coverage)/((double)HOM_PEAK_RATE);
|
||||
}
|
||||
het_cov = hom_cov/asm_opt.polyploidy;
|
||||
m_het_cov = het_cov + (het_cov*0.333333);///hom_cov - het_cov + (het_cov*0.333333);
|
||||
m_hom_cov = het_cov + (het_cov*0.6);///hom_cov - het_cov + (het_cov*0.6);
|
||||
// fprintf(stderr, "hom_cov::%lu, het_cov::%lu, m_het_cov::%lu, m_hom_cov::%lu\n", hom_cov, het_cov, m_het_cov, m_hom_cov);
|
||||
|
||||
buf_t b; memset(&b, 0, sizeof(buf_t)); b.a = (binfo_t*)calloc(n_vtx, sizeof(binfo_t));
|
||||
uint64_t tLen = get_bub_pop_max_dist_advance(ug->g, &b);
|
||||
kv_init(bub->list); kv_init(bub->num); kv_init(bub->pathLen);
|
||||
kv_init(bub->b_s_idx); kv_malloc(bub->b_s_idx, ug->g->n_seq);
|
||||
bub->b_ug = NULL; kv_init(bub->chain_weight);
|
||||
bub->b_s_idx.n = ug->g->n_seq;
|
||||
memset(bub->b_s_idx.a, -1, bub->b_s_idx.n * sizeof(uint64_t));
|
||||
CALLOC(bub->index, n_vtx);
|
||||
for (i = 0; i < ug->g->n_seq; i++) ug->g->seq[i].c = 0;
|
||||
for (v = 0; v < n_vtx; ++v)
|
||||
{
|
||||
if(ug->g->seq[v>>1].del) continue;
|
||||
if(asg_arc_n(ug->g, v) < 2) continue;
|
||||
if((bub->index[v]&(uint32_t)3) != 0) continue;
|
||||
if(asg_bub_pop1_primary_trio(ug->g, NULL, v, tLen, &b, (uint32_t)-1, (uint32_t)-1, 0, NULL, NULL, NULL, 0, 0, NULL))
|
||||
{
|
||||
//beg is v, end is b.S.a[0]
|
||||
//note b.b include end, does not include beg
|
||||
for (i = 0; i < b.b.n; i++)
|
||||
{
|
||||
if(b.b.a[i]==v || b.b.a[i]==b.S.a[0]) continue;
|
||||
bub->index[b.b.a[i]] &= mode; bub->index[b.b.a[i]] += 1;
|
||||
bub->index[b.b.a[i]^1] &= mode; bub->index[b.b.a[i]^1] += 1;
|
||||
}
|
||||
bub->index[v] &= mode; bub->index[v] += 2;
|
||||
bub->index[b.S.a[0]^1] &= mode; bub->index[b.S.a[0]^1] += 3;
|
||||
}
|
||||
}
|
||||
|
||||
kvec_t_u32_warp stack, result;
|
||||
kv_init(stack.a); kv_init(result.a);
|
||||
for (v = 0; v < n_vtx; ++v)
|
||||
{
|
||||
if((bub->index[v]&(uint32_t)3) !=2) continue;
|
||||
if(asg_bub_pop1_primary_trio(ug->g, ug, v, tLen, &b, (uint32_t)-1, (uint32_t)-1, 0, &pathLen, NULL, NULL, 0, 0, NULL))
|
||||
{
|
||||
//note b.b include end, does not include beg
|
||||
i = b.b.n + 1;
|
||||
if(b.b.n == 2 || b.b.n == 3 || b.b.n == 5)
|
||||
{
|
||||
for (i = 0; i < b.b.n; i++)
|
||||
{
|
||||
if(b.b.a[i]==v || b.b.a[i]==b.S.a[0]) continue;
|
||||
dfs_bubble(ug->g, &stack, &result, b.b.a[i]>>1, v>>1, b.S.a[0]>>1);
|
||||
if((result.a.n + 3) != b.b.n && (result.a.n + 2) != b.b.n) break;
|
||||
}
|
||||
}
|
||||
|
||||
if(i == b.b.n)
|
||||
{
|
||||
kv_push(uint32_t, bub->num, v);
|
||||
}
|
||||
else
|
||||
{
|
||||
kv_push(uint32_t, bub->num, v + (1<<31));
|
||||
}
|
||||
}
|
||||
}
|
||||
kv_destroy(stack.a); kv_destroy(result.a);
|
||||
radix_sort_u32(bub->num.a, bub->num.a + bub->num.n);
|
||||
bub->s_bub = 0;
|
||||
for (k = 0; k < bub->num.n; k++)
|
||||
{
|
||||
if((bub->num.a[k]>>31) == 0) bub->s_bub++;
|
||||
v = (bub->num.a[k]<<1)>>1;
|
||||
bub->num.a[k] = bub->list.n;
|
||||
if(asg_bub_pop1_primary_trio(ug->g, ug, v, tLen, &b, (uint32_t)-1, (uint32_t)-1, 0, &pathLen, NULL, NULL, 0, 0, NULL))
|
||||
{
|
||||
kv_push(uint64_t, bub->pathLen, pathLen);
|
||||
//beg is v, end is b.S.a[0]
|
||||
kv_push(uint32_t, bub->list, v);
|
||||
kv_push(uint32_t, bub->list, b.S.a[0]^1);
|
||||
|
||||
//note b.b include end, does not include beg
|
||||
for (i = 0; i < b.b.n; i++)
|
||||
{
|
||||
if(b.b.a[i]==v || b.b.a[i]==b.S.a[0]) continue;
|
||||
kv_push(uint32_t, bub->list, b.b.a[i]);
|
||||
}
|
||||
}
|
||||
}
|
||||
kv_push(uint32_t, bub->num, bub->list.n);
|
||||
// free(b.a); free(b.S.a); free(b.T.a); free(b.b.a); free(b.e.a);
|
||||
bub->f_bub = bub->num.n - 1; ///bub->s_bub = bub->num.n - 1;
|
||||
|
||||
|
||||
memset(r_het_flag, 0, sizeof((*r_het_flag))*sg->n_seq);
|
||||
for (i = 0; i < ug->g->n_seq; i++) {
|
||||
bub->index[i] = get_unitig_het_fly(ug, i, sg, m_het_cov, m_hom_cov,
|
||||
sources, ruIndex, r_het_flag, 20, M_het(*bub), P_het(*bub), (uint32_t)-1);
|
||||
}
|
||||
|
||||
for (i = 0; i < bub->f_bub; i++) {
|
||||
get_bubbles(bub, i, &beg, &sink, &a, &n, &pathLen);
|
||||
for (v = n_occ = 0; v < n; v++) {
|
||||
bub->index[(a[v]>>1)] = i;
|
||||
n_occ += ug->u.a[a[v]>>1].n;
|
||||
}
|
||||
|
||||
if(n_occ > 3) {
|
||||
if(bub->index[(beg>>1)] != M_het(*bub)) bub->index[(beg>>1)] = (uint32_t)-1;
|
||||
if(bub->index[(sink>>1)] != M_het(*bub)) bub->index[(sink>>1)] = (uint32_t)-1;
|
||||
}
|
||||
|
||||
|
||||
v = beg>>1;
|
||||
if(bub->b_s_idx.a[v] == (uint64_t)-1) {
|
||||
bub->b_s_idx.a[v] <<= 32;
|
||||
bub->b_s_idx.a[v] |= i;
|
||||
} else if((bub->b_s_idx.a[v] & 0xffffffff00000000) == 0xffffffff00000000) {
|
||||
bub->b_s_idx.a[v] <<= 32;
|
||||
bub->b_s_idx.a[v] |= i;
|
||||
}
|
||||
|
||||
|
||||
v = sink>>1;
|
||||
if(bub->b_s_idx.a[v] == (uint64_t)-1) {
|
||||
bub->b_s_idx.a[v] <<= 32;
|
||||
bub->b_s_idx.a[v] |= i;
|
||||
} else if((bub->b_s_idx.a[v] & 0xffffffff00000000) == 0xffffffff00000000) {
|
||||
bub->b_s_idx.a[v] <<= 32;
|
||||
bub->b_s_idx.a[v] |= i;
|
||||
}
|
||||
}
|
||||
for (i = 0; i < ug->g->n_seq; i++) {
|
||||
if(bub->index[i] == M_het(*bub)) bub->index[i] = P_het(*bub);
|
||||
}
|
||||
|
||||
bub->b_g = NULL;
|
||||
bub->b_ug = NULL;
|
||||
build_bub_graph(ug, bub);
|
||||
|
||||
|
||||
///make het nodes to be hom
|
||||
ma_utg_t *u = NULL;
|
||||
for (i = 0; i < bub->b_ug->u.n; i++) {
|
||||
u = &(bub->b_ug->u.a[i]);
|
||||
if(u->n == 0) continue;
|
||||
for (k = 0; k < u->n; k++) {
|
||||
reset_inner_bub_het_poy(sg, ug, bub, u->a[k]>>33, tLen, &b, m_het_cov, m_hom_cov, r_het_flag, sources, ruIndex);
|
||||
}
|
||||
}
|
||||
free(b.a); free(b.S.a); free(b.T.a); free(b.b.a); free(b.e.a);
|
||||
|
||||
for (i = 0; i < ug->g->n_seq; i++) {
|
||||
if(IF_HOM(i, *bub)) continue;
|
||||
if(deter_unitig_het_fly(ug, i, sg, het_cov*1.15, sources, ruIndex, r_het_flag)) continue;
|
||||
bub->index[i] = (uint32_t)-1;
|
||||
}
|
||||
// fprintf(stderr, "-bub->index[18759]: %u, bub->num.n: %u\n", (uint32_t)bub->index[18759], bub->num.n);
|
||||
}
|
||||
|
||||
void print_bubbles(ma_ug_t* ug, bubble_type* bub, kvec_pe_hit* hits, hc_links* link, ha_ug_index* idx)
|
||||
{
|
||||
uint64_t tLen, t_utg, i, k;
|
||||
@@ -6269,7 +6569,7 @@ min_cut_t* m, hc_links* link, G_partition* x)
|
||||
if(res->full_bub == 0)
|
||||
{
|
||||
res->a.n = 0;
|
||||
uint32_t v, u = 0, uv, k_n, pre_n = x->n;
|
||||
uint32_t v, u = 0, uv = UINT32_MAX, k_n, pre_n = x->n;
|
||||
hc_linkeage* t = NULL;
|
||||
x->n--;
|
||||
for (i = 0; i < n; i++)
|
||||
@@ -6634,7 +6934,7 @@ void get_bub_id(bubble_type* bub, uint32_t root, uint64_t* id0, uint64_t* id1, u
|
||||
if(check_het)
|
||||
{
|
||||
get_bubbles(bub, b_id0, &beg, &sink, NULL, NULL, NULL);
|
||||
if(IF_HET(beg>>1, *bub) && IF_HET(sink>>1, *bub)) b_id0 = (uint64_t)-1;
|
||||
if(((beg == (uint32_t)-1) || IF_HET(beg>>1, *bub)) && ((sink == (uint32_t)-1) || IF_HET(sink>>1, *bub))) b_id0 = (uint64_t)-1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6644,7 +6944,7 @@ void get_bub_id(bubble_type* bub, uint32_t root, uint64_t* id0, uint64_t* id1, u
|
||||
if(check_het)
|
||||
{
|
||||
get_bubbles(bub, b_id1, &beg, &sink, NULL, NULL, NULL);
|
||||
if(IF_HET(beg>>1, *bub) && IF_HET(sink>>1, *bub)) b_id1 = (uint64_t)-1;
|
||||
if(((beg == (uint32_t)-1) || IF_HET(beg>>1, *bub)) && ((sink == (uint32_t)-1) || IF_HET(sink>>1, *bub))) b_id1 = (uint64_t)-1;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -17130,8 +17430,280 @@ int hic_short_align_poy(const enzyme *fn1, const enzyme *fn2, ha_ug_index* idx,
|
||||
return 1;
|
||||
}
|
||||
|
||||
mmhap_t* gen_mmhap_t(ma_ug_t *ug, asg_t *rg, ma_hit_t_alloc *src)
|
||||
{
|
||||
uint64_t k, z, hom_cov, het_cov, s, *bs = NULL; uint8_t *ff; mmhap_t *p;
|
||||
if(asm_opt.hom_global_coverage_set) {
|
||||
hom_cov = asm_opt.hom_global_coverage;
|
||||
} else {
|
||||
hom_cov = ((double)asm_opt.hom_global_coverage)/((double)HOM_PEAK_RATE);
|
||||
}
|
||||
het_cov = hom_cov/asm_opt.polyploidy;
|
||||
CALLOC(ff, rg->n_seq); CALLOC(p, 1); CALLOC(bs, asm_opt.polyploidy+1);
|
||||
p->h.n = p->h.m = ug->u.n; CALLOC(p->h.a, p->h.n);
|
||||
for (k = p->a.n = s = 0; k < ug->u.n; k++) {
|
||||
p->h.a[k].a = p->a.n;
|
||||
p->h.a[k].n = 0;
|
||||
p->h.a[k].m = infer_mmhap_copy(ug, rg, src, ff, k, het_cov, asm_opt.polyploidy);
|
||||
if(p->h.a[k].m == (uint64_t)asm_opt.polyploidy) {
|
||||
p->h.a[k].n = p->h.a[k].m; s = 1;
|
||||
}
|
||||
p->a.n += p->h.a[k].m; bs[p->h.a[k].m] += ug->g->seq[k].len;
|
||||
}
|
||||
free(ff);
|
||||
p->a.m = p->a.n; MALLOC(p->a.a, p->a.n); memset(p->a.a, -1, sizeof((*(p->a.a)))*p->a.n);
|
||||
if(s) {
|
||||
for (k = p->a.n = 0; k < ug->u.n; k++) {
|
||||
if(p->h.a[k].n == p->h.a[k].m) {
|
||||
for (z = 0; z < p->h.a[k].m; z++) p->a.a[p->h.a[k].a+z] = z;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, uint32_t is_poy, kvec_pe_hit **rhits)
|
||||
for (k = 1; k <= (uint64_t)asm_opt.polyploidy; k++) {
|
||||
fprintf(stderr, "[M::stat] # %lu-copy bases: %lu\n", k, bs[k]);
|
||||
}
|
||||
free(bs);
|
||||
return p;
|
||||
}
|
||||
|
||||
bubble_type *gen_mmhap_bub(ma_ug_t* ug, uint8_t *r_het_flag, kv_u_trans_t *ref, mmhap_t *hh)
|
||||
{
|
||||
uint64_t k;
|
||||
bubble_type *p; CALLOC(p, 1); p->n_round = asm_opt.n_weight; p->round_id = 0;
|
||||
identify_bubbles(ug, p, r_het_flag, ref);
|
||||
for (k = 0; k < ug->g->n_seq; k++) {
|
||||
if(IF_BUB(k, (*p))) continue;
|
||||
if(IF_HOM(k, (*p))) {
|
||||
if(hh->h.a[k].n < hh->h.a[k].m) p->index[k] = p->num.n;
|
||||
} else if(IF_HET(k, (*p))) {
|
||||
if(hh->h.a[k].n >= hh->h.a[k].m) p->index[k] = (uint32_t)-1;
|
||||
}
|
||||
}
|
||||
return p;
|
||||
}
|
||||
|
||||
void purge_phase_0(ha_ug_index* idx, hc_links *link, bubble_type *bub, ps_t *s, kvec_pe_hit* hits, kv_u_trans_t *k_trans, uint8_t *del, mmhap_t *hh)
|
||||
{
|
||||
k_trans->idx.n = k_trans->n = 0; hits->idx.n = 0;
|
||||
for (bub->round_id = 0; bub->round_id < bub->n_round; bub->round_id++) {
|
||||
renew_kv_u_trans(k_trans, link, hits, &(idx->t_ch->k_trans), idx, bub, s->s, NULL, 1/**0**/);
|
||||
mc_solve(NULL, NULL, k_trans, idx->ug, idx->read_g, 0.8, R_INF.trio_flag,
|
||||
(bub->round_id==0?1:0), s->s, 1, bub, &(idx->t_ch->k_trans), 0,
|
||||
/**(((bub.round_id+1) == bub.n_round)?1:0)**/0);
|
||||
/*******************************for debug************************************/
|
||||
// label_unitigs_sm(s->s, NULL, idx->ug);
|
||||
}
|
||||
|
||||
uint64_t l[2], k, len; int64_t p; ma_ug_t *ug = idx->ug;
|
||||
for (k = l[0] = l[1] = 0; k < ug->g->n_seq; k++) {
|
||||
if((del[k] == 1) || (s->s[k] == 0)) continue;
|
||||
if(s->s[k] > 0) l[0] += ug->g->seq[k].len;
|
||||
else l[1] += ug->g->seq[k].len;
|
||||
}
|
||||
|
||||
p = (l[0]>=l[1])?1:-1;
|
||||
for (k = len = 0; k < ug->g->n_seq; k++) {
|
||||
if((del[k] == 1) || (s->s[k] == 0)) {
|
||||
if((del[k] == 2) ||
|
||||
((hh->h.a[k].n == hh->h.a[k].m) && (hh->h.a[k].m == ((uint64_t)asm_opt.polyploidy)))) {
|
||||
len += ug->g->seq[k].len;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if(s->s[k] == p) {
|
||||
del[k] = 2; len += ug->g->seq[k].len;
|
||||
} else {
|
||||
del[k] = 1; bub->index[k] = (uint32_t)-1;
|
||||
}
|
||||
s->s[k] = 0;///reset
|
||||
}
|
||||
fprintf(stderr, "[M::%s::stat] # remaining bases: %lu\n", __func__, len);
|
||||
}
|
||||
|
||||
void exchange_kv_u_trans_t(kv_u_trans_t *a, kv_u_trans_t *b)
|
||||
{
|
||||
uint64_t k, *ua; u_trans_t *u;
|
||||
k = a->n; a->n = b->n; b->n = k;
|
||||
k = a->m; a->m = b->m; b->m = k;
|
||||
u = a->a; a->a = b->a; b->a = u;
|
||||
|
||||
k = a->idx.n; a->idx.n = b->idx.n; b->idx.n = k;
|
||||
k = a->idx.m; a->idx.m = b->idx.m; b->idx.m = k;
|
||||
ua = a->idx.a; a->idx.a = b->idx.a; b->idx.a = ua;
|
||||
}
|
||||
|
||||
uint64_t cal_ave_ovlp(u_trans_t *a, uint64_t an, double top)
|
||||
{
|
||||
if(!an) return 0;
|
||||
uint64_t k, len, cut, occ, tot;
|
||||
for (k = len = 0; k < an; k++) {
|
||||
len += a[k].qe - a[k].qs;
|
||||
}
|
||||
cut = len - (len*top);
|
||||
for (k = occ = tot = 0; k < an; k++) {
|
||||
if(a[k].qe - a[k].qs < cut) continue;
|
||||
occ++; tot += a[k].qe - a[k].qs;
|
||||
}
|
||||
if(!occ) {
|
||||
occ = an; tot = len;
|
||||
}
|
||||
return tot/occ;
|
||||
}
|
||||
|
||||
|
||||
void clean_trans_ovlp(bubble_type *bub, kv_u_trans_t *des, kv_u_trans_t *src, asg64_v *srt, ma_ug_t *ug)
|
||||
{
|
||||
uint64_t k, st, i, m = 0, ncut;
|
||||
for (k = des->n = 0; k < src->n; k++) {
|
||||
if(IF_HOM(src->a[k].qn, (*bub))) continue;
|
||||
if(IF_HOM(src->a[k].tn, (*bub))) continue;
|
||||
kv_push(u_trans_t, *des, src->a[k]);
|
||||
}
|
||||
|
||||
kv_resize(uint64_t, des->idx, src->idx.n); des->idx.n = src->idx.n;
|
||||
memset(des->idx.a, 0, des->idx.n*sizeof((*(des->idx.a))));
|
||||
for (st = 0, i = 1; i <= des->n; ++i) {
|
||||
if (i == des->n || des->a[i].qn != des->a[st].qn) {
|
||||
des->idx.a[des->a[st].qn] = (((uint64_t)st)<<32)|(i-st); m++;
|
||||
st = i;
|
||||
}
|
||||
}
|
||||
|
||||
if(srt && m) {
|
||||
ncut = 0; kv_resize(uint64_t, *srt, m);
|
||||
for (k = srt->n = 0; k < des->idx.n; k++) {
|
||||
if(!((uint32_t)(des->idx.a[k]))) continue;
|
||||
m = cal_ave_ovlp(des->a+(des->idx.a[k]>>32), ((uint32_t)(des->idx.a[k])), 0.9);
|
||||
m = ((uint64_t)-1)-m; m <<= 32; m += k;
|
||||
kv_push(uint64_t, *srt, m);
|
||||
}
|
||||
radix_sort_b64(srt->a, srt->a + srt->n);
|
||||
for (k = 0; k < srt->n; k++) {
|
||||
ncut += trans_sec_cut0(des, srt, (uint32_t)(srt->a[k]), 0.2, 256, ug);
|
||||
}
|
||||
|
||||
if(ncut) {///renew idx
|
||||
for (i = k = 0; i < des->n; i++) {
|
||||
if(des->a[i].del) continue;
|
||||
des->a[k++] = des->a[i];
|
||||
}
|
||||
des->n = k;
|
||||
|
||||
memset(des->idx.a, 0, des->idx.n*sizeof((*(des->idx.a))));
|
||||
for (st = 0, i = 1; i <= des->n; ++i) {
|
||||
if (i == des->n || des->a[i].qn != des->a[st].qn) {
|
||||
des->idx.a[des->a[st].qn] = (((uint64_t)st)<<32)|(i-st); m++;
|
||||
st = i;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void purge_phase(ha_ug_index* idx, mmhap_t *hh, uint64_t hapid, uint64_t max_round, hc_links *link, kvec_pe_hit* hits,
|
||||
bubble_type *bub, ps_t *s, kv_u_trans_t *k_trans, uint8_t *del, uint32_t *bidx, kv_u_trans_t *ref, asg64_v *srt)
|
||||
{
|
||||
uint64_t k, len; uint32_t *bm;
|
||||
ma_ug_t *ug = idx->ug;
|
||||
if(max_round <= 0) {
|
||||
for (k = 0; k < ug->g->n_seq; k++) {
|
||||
if(hh->h.a[k].n >= hh->h.a[k].m) continue;
|
||||
hh->a.a[hh->h.a[k].a+hh->h.a[k].n] = hapid;
|
||||
hh->h.a[k].n++;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
for (k = 0; k < ug->g->n_seq; k++) {
|
||||
del[k] = 0; s->s[k] = 0;
|
||||
if(hh->h.a[k].n >= hh->h.a[k].m) {///all haplotypes have been set
|
||||
bub->index[k] = (uint32_t)-1; del[k] = 1;
|
||||
} else {
|
||||
if(IF_HOM(k, (*bub))) bub->index[k] = bub->num.n;
|
||||
}
|
||||
bidx[k] = bub->index[k];
|
||||
}
|
||||
bm = bub->index; bub->index = bidx; bidx = bm;
|
||||
|
||||
kv_resize(u_trans_t, *ref, idx->t_ch->k_trans.n); ref->n = idx->t_ch->k_trans.n;
|
||||
memcpy(ref->a, idx->t_ch->k_trans.a, ref->n*sizeof((*(ref->a))));
|
||||
kv_resize(uint64_t, ref->idx, idx->t_ch->k_trans.idx.n); ref->idx.n = idx->t_ch->k_trans.idx.n;
|
||||
memcpy(ref->idx.a, idx->t_ch->k_trans.idx.a, ref->idx.n*sizeof((*(ref->idx.a))));
|
||||
exchange_kv_u_trans_t(ref, &(idx->t_ch->k_trans));
|
||||
|
||||
for (k = 0; k < max_round; k++) {
|
||||
clean_trans_ovlp(bub, &(idx->t_ch->k_trans), ref, ((k+1)<max_round)?srt:NULL, ug);
|
||||
purge_phase_0(idx, link, bub, s, hits, k_trans, del, hh);
|
||||
}
|
||||
|
||||
for (k = len = 0; k < ug->g->n_seq; k++) {
|
||||
if(del[k] != 2) {
|
||||
if((hh->h.a[k].n == hh->h.a[k].m) && (hh->h.a[k].m == ((uint64_t)asm_opt.polyploidy))) {
|
||||
len += ug->g->seq[k].len;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
assert(hh->h.a[k].n < hh->h.a[k].m);
|
||||
hh->a.a[hh->h.a[k].a+hh->h.a[k].n] = hapid;
|
||||
hh->h.a[k].n++; len += ug->g->seq[k].len;
|
||||
}
|
||||
fprintf(stderr, "[M::%s::stat] # hap%lu bases: %lu\n", __func__, hapid+1, len);
|
||||
|
||||
bm = bub->index; bub->index = bidx; bidx = bm;
|
||||
exchange_kv_u_trans_t(ref, &(idx->t_ch->k_trans));
|
||||
}
|
||||
|
||||
int hic_short_align_mmhap(const enzyme *fn1, const enzyme *fn2, ha_ug_index* idx, ug_opt_t *opt, kvec_pe_hit **rhits, mmhap_t **rh)
|
||||
{
|
||||
if(rh) (*rh) = NULL;
|
||||
sldat_t sl;
|
||||
sl.idx = idx;
|
||||
sl.t_ch = idx->t_ch;
|
||||
sl.chunk_size = 20000000;
|
||||
sl.n_thread = asm_opt.thread_num;
|
||||
sl.total_base = sl.total_pair = 0;
|
||||
idx->hap_cnt = asm_opt.hap_occ;
|
||||
kv_init(sl.hits.a); kv_init(sl.hits.idx); kv_init(sl.hits.occ);
|
||||
|
||||
|
||||
if(!load_hc_hits(&sl.hits, idx->ug, asm_opt.output_file_name)) {
|
||||
alignment_worker_pipeline(&sl, fn1, fn2);
|
||||
write_hc_hits(&sl.hits, idx->ug, asm_opt.output_file_name);
|
||||
}
|
||||
sl.hits.uID_bits = idx->uID_bits; sl.hits.pos_mode = idx->pos_mode;
|
||||
|
||||
if(sl.hits.idx.n == 0) idx_hc_links(&(sl.hits), idx, NULL);
|
||||
|
||||
mmhap_t *hh = gen_mmhap_t(idx->ug, idx->read_g, opt->sources);
|
||||
bubble_type *bub = gen_mmhap_bub(idx->ug, idx->t_ch->ir_het, &(idx->t_ch->k_trans), hh);
|
||||
hc_links link; init_hc_links(&link, idx->ug->g->n_seq, idx->t_ch);
|
||||
measure_distance(idx, idx->ug, &sl.hits, &link, bub, &(idx->t_ch->k_trans));
|
||||
kv_u_trans_t k_trans; kv_init(k_trans); kv_init(k_trans.idx);
|
||||
ps_t *s = init_ps_t(11, idx->ug->g->n_seq); ///H_partition hap;
|
||||
uint64_t k, n_hap = asm_opt.polyploidy; asg64_v srt; kv_init(srt);
|
||||
uint8_t *ff; CALLOC(ff, idx->ug->g->n_seq);
|
||||
uint32_t *bidx; MALLOC(bidx, idx->ug->g->n_seq);
|
||||
kv_u_trans_t r_trans_buf; kv_init(r_trans_buf); kv_init(r_trans_buf.idx);
|
||||
|
||||
for (k = 0; k < n_hap; k++) {
|
||||
purge_phase(idx, hh, k, ((n_hap>k)?(n_hap-k-1):(0)), &link, &sl.hits, bub, s, &k_trans, ff, bidx, &r_trans_buf, &srt);
|
||||
}
|
||||
if(rh) (*rh) = hh;
|
||||
|
||||
// if(rhits) (*rhits) = get_r_hits_order(&sl.hits, idx->uID_bits, idx->pos_mode, idx->read_g, idx->ug, &bub);
|
||||
|
||||
kv_destroy(sl.hits.a); kv_destroy(sl.hits.idx); kv_destroy(sl.hits.occ);
|
||||
destory_hc_links(&link);
|
||||
kv_destroy(k_trans); kv_destroy(k_trans.idx);
|
||||
destory_ps_t(&s); destory_bubbles(bub); free(bub);
|
||||
kv_destroy(srt); free(ff); free(bidx);
|
||||
kv_destroy(r_trans_buf); kv_destroy(r_trans_buf.idx);
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, mmhap_t **rh, kvec_pe_hit **rhits)
|
||||
{
|
||||
ug_index = NULL;
|
||||
int exist = (asm_opt.load_index_from_disk?
|
||||
@@ -17142,13 +17714,198 @@ void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt,
|
||||
ug_index->read_g = read_g;
|
||||
ug_index->t_ch = t_ch;
|
||||
///test_unitig_index(ug_index, ug);
|
||||
if(!is_poy) hic_short_align(asm_opt.hic_reads[0], asm_opt.hic_reads[1], ug_index, opt, rhits);
|
||||
else hic_short_align_poy(asm_opt.hic_reads[0], asm_opt.hic_reads[1], ug_index, opt);
|
||||
if(!rh) hic_short_align(asm_opt.hic_reads[0], asm_opt.hic_reads[1], ug_index, opt, rhits);
|
||||
else hic_short_align_mmhap(asm_opt.hic_reads[0], asm_opt.hic_reads[1], ug_index, opt, rhits, rh);
|
||||
// else hic_short_align_poy(asm_opt.hic_reads[0], asm_opt.hic_reads[1], ug_index, opt);
|
||||
|
||||
|
||||
destory_hc_pt_index(ug_index);free(ug_index);
|
||||
}
|
||||
|
||||
uint64_t ug_occ_hap_w(uint64_t is, uint64_t ie, ma_utg_t *u)
|
||||
{
|
||||
uint64_t l, i, us, ue, occ;
|
||||
for (i = l = occ = 0; i < u->n; i++) {
|
||||
us = l; ue = l + Get_READ_LENGTH(R_INF, (u->a[i]>>33));
|
||||
if(is <= us && ie >= ue) {
|
||||
if((R_INF.trio_flag[u->a[i]>>33] == FATHER) || (R_INF.trio_flag[u->a[i]>>33] == MOTHER)) {
|
||||
occ++;
|
||||
}
|
||||
}
|
||||
if(us >= ie) break;
|
||||
l += (uint32_t)u->a[i];
|
||||
}
|
||||
return occ;
|
||||
}
|
||||
|
||||
void trio_phasing_refine(ma_ug_t *iug, asg_t* sg, kv_u_trans_t *ref, ug_opt_t *opt)
|
||||
{
|
||||
ma_ug_t *ug = copy_untig_graph(iug); uint8_t *bf = NULL;
|
||||
uint64_t ug_n0 = ug->g->n_seq, k, i, ul, cis_n, trans_n, w_n, tot_hap, tot_r, frid, mrid, flag;
|
||||
kv_u_trans_t in; memset(&in, 0, sizeof(in)); ma_utg_t *p; u_trans_t *z; int64_t fh, mh;
|
||||
bubble_type *bub = gen_bubble_chain(sg, ug, opt, &bf, 0); free(bf);
|
||||
clean_u_trans_t_idx_filter_adv(ref, ug, sg, 0.95, 0);
|
||||
|
||||
///update ug itself
|
||||
ul = FATHER; frid = ug->g->n_seq; asg_seq_set(ug->g, frid, ul, 0);
|
||||
kv_pushp(ma_utg_t, ug->u, &p); memset(p, 0, sizeof((*p)));
|
||||
p->s = 0; p->start = UINT32_MAX; p->end = UINT32_MAX; p->len = ul, p->n = p->m = 1; p->circ = 1;
|
||||
kv_roundup32(p->m); p->a = (uint64_t*)malloc(8 * p->m);
|
||||
p->a[0] = UINT32_MAX; p->a[0] <<= 32; p->a[0] |= ul;
|
||||
|
||||
ul = MOTHER; mrid = ug->g->n_seq; asg_seq_set(ug->g, mrid, ul, 0);
|
||||
kv_pushp(ma_utg_t, ug->u, &p); memset(p, 0, sizeof((*p)));
|
||||
p->s = 0; p->start = UINT32_MAX; p->end = UINT32_MAX; p->len = ul, p->n = p->m = 1; p->circ = 1;
|
||||
kv_roundup32(p->m); p->a = (uint64_t*)malloc(8 * p->m);
|
||||
p->a[0] = UINT32_MAX; p->a[0] <<= 32; p->a[0] |= ul;
|
||||
|
||||
free(ug->g->idx); ug->g->idx = 0; ug->g->is_srt = 0; asg_cleanup(ug->g);
|
||||
|
||||
///update bubble
|
||||
REALLOC(bub->index, ug->g->n_seq);
|
||||
for (k = ug_n0; k < ug->g->n_seq; k++) bub->index[k] = bub->f_bub+1;
|
||||
|
||||
bub->b_s_idx.n = ug->g->n_seq; kv_resize(uint64_t, bub->b_s_idx, ug->g->n_seq);
|
||||
for (k = ug_n0; k < bub->b_s_idx.m; k++) bub->b_s_idx.a[k] = (uint64_t)-1;
|
||||
|
||||
///check if a node has haplotype markers
|
||||
CALLOC(bf, ug->g->n_seq);
|
||||
for (k = cis_n = tot_hap = tot_r = 0; k < ug_n0; k++) {
|
||||
if(ug->g->seq[k].del) continue;
|
||||
p = &(ug->u.a[k]); tot_r += p->n;
|
||||
if(p->n == 0 || p->m == 0) continue;
|
||||
for (i = fh = mh = 0; i < p->n; i++) {
|
||||
if(R_INF.trio_flag[p->a[i]>>33] == FATHER) {
|
||||
tot_hap++; fh = 1;
|
||||
} else if(R_INF.trio_flag[p->a[i]>>33] == MOTHER) {
|
||||
tot_hap++; mh = 1;
|
||||
}
|
||||
if((R_INF.trio_flag[p->a[i]>>33] == FATHER) || (R_INF.trio_flag[p->a[i]>>33] == MOTHER)) {
|
||||
tot_hap++; bf[k] = 1;
|
||||
}
|
||||
}
|
||||
if(fh > 0) {
|
||||
bf[k] = 1; cis_n+=2;
|
||||
}
|
||||
if(mh > 0) {
|
||||
bf[k] = 1; cis_n+=2;
|
||||
}
|
||||
}
|
||||
for (; k < ug->g->n_seq; k++) bf[k] = 1;///for father/mother nodes
|
||||
|
||||
|
||||
///update trans overlaps
|
||||
kv_pushp(u_trans_t, *ref, &z); memset(z, 0, sizeof((*z)));
|
||||
z->f = RC_2; z->nw = (DBL_MAX/2); z->rev = 0;
|
||||
z->qn = ug_n0; z->tn = ug_n0+1; z->del = 0;
|
||||
z->qs = 0; z->qe = ug->g->seq[z->qn].len;
|
||||
z->ts = 0; z->te = ug->g->seq[z->tn].len;
|
||||
|
||||
kv_pushp(u_trans_t, *ref, &z); memset(z, 0, sizeof((*z)));
|
||||
z->f = RC_2; z->nw = (DBL_MAX/2); z->rev = 0;
|
||||
z->qn = ug_n0+1; z->tn = ug_n0; z->del = 0;
|
||||
z->qs = 0; z->qe = ug->g->seq[z->qn].len;
|
||||
z->ts = 0; z->te = ug->g->seq[z->tn].len;
|
||||
|
||||
kt_u_trans_t_idx(ref, ug->g->n_seq);
|
||||
// clean_u_trans_t_idx_adv(ref, ug, sg);
|
||||
// clean_u_trans_t_idx_filter_adv(ref, ug, sg);
|
||||
|
||||
///gen all links
|
||||
for (k = trans_n = 0; k < ref->n; k++) {
|
||||
if(ref->a[k].del) continue;
|
||||
if((IF_HOM(ref->a[k].qn, (*bub))) || (IF_HOM(ref->a[k].tn, (*bub)))) continue;
|
||||
///disable bf
|
||||
// if((!bf[ref->a[k].qn]) || (!bf[ref->a[k].tn])) continue;
|
||||
trans_n++;
|
||||
}
|
||||
kv_resize(u_trans_t, in, (trans_n+cis_n));
|
||||
|
||||
for (k = in.n = 0; k < ref->n; k++) {
|
||||
if(ref->a[k].del) continue;
|
||||
if((IF_HOM(ref->a[k].qn, (*bub))) || (IF_HOM(ref->a[k].tn, (*bub)))) continue;
|
||||
///disable bf
|
||||
// if((!bf[ref->a[k].qn]) || (!bf[ref->a[k].tn])) continue;
|
||||
|
||||
kv_pushp(u_trans_t, in, &z); memset(z, 0, sizeof((*z))); w_n = 0;
|
||||
if((ref->a[k].qn < ug_n0) && (ref->a[k].tn < ug_n0)) {
|
||||
w_n += ug_occ_hap_w(ref->a[k].qs, ref->a[k].qe, &(ug->u.a[ref->a[k].qn])) +
|
||||
ug_occ_hap_w(ref->a[k].ts, ref->a[k].te, &(ug->u.a[ref->a[k].tn]));
|
||||
w_n += ug_occ_w(ref->a[k].qs, ref->a[k].qe, &(ug->u.a[ref->a[k].qn])) +
|
||||
ug_occ_w(ref->a[k].ts, ref->a[k].te, &(ug->u.a[ref->a[k].tn]));
|
||||
} else {
|
||||
w_n += tot_hap + tot_r;
|
||||
}
|
||||
w_n >>= 1;///wn/4
|
||||
(*z) = ref->a[k]; z->nw = ((w_n)?(w_n):(1));
|
||||
}
|
||||
|
||||
for (k = 0; k < ug_n0; k++) {
|
||||
if(!bf[k]) continue;
|
||||
p = &(ug->u.a[k]);
|
||||
if(p->n == 0 || p->m == 0) continue;
|
||||
fh = mh = 0;
|
||||
for (i = 0; i < p->n; i++) {
|
||||
if(R_INF.trio_flag[p->a[i]>>33] == FATHER) fh--;
|
||||
if(R_INF.trio_flag[p->a[i]>>33] == MOTHER) mh--;
|
||||
}
|
||||
if(fh != 0) {
|
||||
w_n = frid;
|
||||
|
||||
kv_pushp(u_trans_t, in, &z); memset(z, 0, sizeof((*z)));
|
||||
z->f = RC_2; z->nw = fh; z->rev = 0;
|
||||
z->qn = w_n; z->tn = k; z->del = 0;
|
||||
z->qs = 0; z->qe = MIN(ug->g->seq[z->qn].len, ug->g->seq[z->tn].len);
|
||||
z->ts = 0; z->te = MIN(ug->g->seq[z->qn].len, ug->g->seq[z->tn].len);
|
||||
|
||||
kv_pushp(u_trans_t, in, &z); memset(z, 0, sizeof((*z)));
|
||||
z->f = RC_2; z->nw = fh; z->rev = 0;
|
||||
z->qn = k; z->tn = w_n; z->del = 0;
|
||||
z->qs = 0; z->qe = MIN(ug->g->seq[z->qn].len, ug->g->seq[z->tn].len);
|
||||
z->ts = 0; z->te = MIN(ug->g->seq[z->qn].len, ug->g->seq[z->tn].len);
|
||||
}
|
||||
|
||||
if(mh != 0) {
|
||||
w_n = mrid;
|
||||
|
||||
kv_pushp(u_trans_t, in, &z); memset(z, 0, sizeof((*z)));
|
||||
z->f = RC_2; z->nw = mh; z->rev = 0;
|
||||
z->qn = w_n; z->tn = k; z->del = 0;
|
||||
z->qs = 0; z->qe = MIN(ug->g->seq[z->qn].len, ug->g->seq[z->tn].len);
|
||||
z->ts = 0; z->te = MIN(ug->g->seq[z->qn].len, ug->g->seq[z->tn].len);
|
||||
|
||||
kv_pushp(u_trans_t, in, &z); memset(z, 0, sizeof((*z)));
|
||||
z->f = RC_2; z->nw = mh; z->rev = 0;
|
||||
z->qn = k; z->tn = w_n; z->del = 0;
|
||||
z->qs = 0; z->qe = MIN(ug->g->seq[z->qn].len, ug->g->seq[z->tn].len);
|
||||
z->ts = 0; z->te = MIN(ug->g->seq[z->qn].len, ug->g->seq[z->tn].len);
|
||||
}
|
||||
}
|
||||
free(bf);
|
||||
assert(in.n == (trans_n+cis_n));
|
||||
kt_u_trans_t_idx(&in, ug->g->n_seq);
|
||||
// clean_u_trans_t_idx_adv(&in, ug, sg);
|
||||
|
||||
ps_t *s = init_ps_t(11, ug->g->n_seq); s->s[frid] = 1; s->s[mrid] = -1;
|
||||
mc_solve(NULL, NULL, &in, ug, sg, 0.8, NULL, 0, s->s, 1, bub, ref, 0, 0);
|
||||
// fprintf(stderr, "[M::%s::] s[frid]::%d, s->s[mrid]::%d\n", __func__, s->s[frid], s->s[mrid]);
|
||||
if((s->s[frid] != 0) && (s->s[mrid] != 0) && (s->s[frid] != s->s[mrid])) {
|
||||
memset(R_INF.trio_flag, AMBIGU, R_INF.total_reads * sizeof(uint8_t));
|
||||
for (i = 0; i < ug_n0; i++) {
|
||||
if(ug->g->seq[i].del) continue;
|
||||
flag = AMBIGU;
|
||||
if(s->s[i] == 0) continue;
|
||||
flag = (s->s[i] == s->s[frid]? FATHER:MOTHER);
|
||||
p = &ug->u.a[i];
|
||||
if(p->m == 0) continue;
|
||||
for (k = 0; k < p->n; k++) R_INF.trio_flag[p->a[k]>>33] = flag;
|
||||
}
|
||||
}
|
||||
|
||||
destory_bubbles(bub); free(bub); ma_ug_destroy(ug);
|
||||
destory_ps_t(&s); kv_destroy(in); kv_destroy(in.idx);
|
||||
}
|
||||
|
||||
spg_t *hic_pre_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, kvec_pe_hit **rhits)
|
||||
{
|
||||
ug_index = NULL;
|
||||
|
||||
@@ -95,6 +95,8 @@ void destory_bubbles(bubble_type* bub);
|
||||
void identify_bubbles(ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, kv_u_trans_t *ref);
|
||||
void identify_bubbles_recal(asg_t* sg, ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, ma_hit_t_alloc* sources, R_to_U* ruIndex,
|
||||
kv_u_trans_t *ref);
|
||||
void identify_bubbles_recal_poy(asg_t* sg, ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, ma_hit_t_alloc* sources, R_to_U* ruIndex,
|
||||
kv_u_trans_t *ref);
|
||||
void resolve_bubble_chain_tangle(ma_ug_t* ug, bubble_type* bub);
|
||||
uint32_t connect_bub_occ(bubble_type* bub, uint32_t root_id, uint32_t check_het);
|
||||
void get_bub_id(bubble_type* bub, uint32_t root, uint64_t* id0, uint64_t* id1, uint32_t check_het);
|
||||
@@ -111,9 +113,10 @@ pdq* pqw, uint32_t* path_w, buf_t *resw, asg_t *sg, uint8_t *dest, uint8_t df, u
|
||||
long long *dis);
|
||||
void set_utg_by_dis(uint32_t v, pdq* pq, asg_t *g, kvec_t_u32_warp *res, uint32_t dis);
|
||||
void dedup_hits(kvec_pe_hit* hits, uint64_t is_dup);
|
||||
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, uint32_t is_poy, kvec_pe_hit **rhits);
|
||||
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, mmhap_t **rh, kvec_pe_hit **rhits);
|
||||
spg_t *hic_pre_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, kvec_pe_hit **rhits);
|
||||
void prt_bubble_gfa_adv(FILE *fp, bubble_type *bub, const char* utg_pre, const char* bub_pre, const char* chain_pre);
|
||||
void bp_solve(ug_opt_t *opt, kv_u_trans_t *ref, ma_ug_t *ug, asg_t *sg, bubble_type *bub, double cis_rate);
|
||||
void trio_phasing_refine(ma_ug_t *ug, asg_t* sg, kv_u_trans_t *ta, ug_opt_t *opt);
|
||||
|
||||
#endif
|
||||
|
||||
+371
-14
@@ -32,7 +32,7 @@ KRADIX_SORT_INIT(osg, osg_arc_t, osg_arc_key, member_size(osg_arc_t, u))
|
||||
#define BREAK_CUTOFF 0.1
|
||||
#define BREAK_BOUNDARY 0.015
|
||||
void reduce_hamming_error_adv(ma_ug_t *iug, asg_t *sg, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
int max_hang, int min_ovlp, long long gap_fuzz, R_to_U *ru, bubble_type* bub);
|
||||
int max_hang, int min_ovlp, long long gap_fuzz, R_to_U *ru, bubble_type* bub, uint32_t max_ext);
|
||||
|
||||
typedef struct {
|
||||
uint64_t ruid;
|
||||
@@ -714,7 +714,7 @@ void update_u_hits(kvec_pe_hit *u_hits, kvec_pe_hit *r_hits, ma_ug_t* ug, asg_t*
|
||||
kv_pushp(hit_aux_t, x, &p);
|
||||
p->ruid = u->a[i]>>32;
|
||||
p->ruid <<= 32;
|
||||
p->ruid |= v;
|
||||
p->ruid |= v;///rid|rev|uid
|
||||
p->off = offset;
|
||||
offset += (uint32_t)u->a[i];
|
||||
}
|
||||
@@ -725,8 +725,8 @@ void update_u_hits(kvec_pe_hit *u_hits, kvec_pe_hit *r_hits, ma_ug_t* ug, asg_t*
|
||||
}
|
||||
}
|
||||
|
||||
radix_sort_hit_aux_ruid(x.a, x.a + x.n);
|
||||
x.idx.n = x.idx.m = (x.n?(x.a[x.n-1].ruid>>33)+1:0);///how many unitigs
|
||||
radix_sort_hit_aux_ruid(x.a, x.a + x.n);///sort by (rid|rev|uid)
|
||||
x.idx.n = x.idx.m = (x.n?(x.a[x.n-1].ruid>>33)+1:0);///how many reads?
|
||||
CALLOC(x.idx.a, x.idx.n);
|
||||
for (k = 1, l = 0; k <= x.n; ++k)
|
||||
{
|
||||
@@ -781,7 +781,7 @@ ma_ug_t* get_trio_unitig_graph(asg_t *sg, uint8_t flag, ug_opt_t *opt)
|
||||
adjust_utg_by_trio(&ug, sg, flag, TRIO_THRES, opt->sources, opt->reverse_sources,
|
||||
opt->coverage_cut, opt->tipsLen, opt->tip_drop_ratio, opt->stops_threshold,
|
||||
opt->ruIndex, opt->chimeric_rate, opt->drop_ratio, opt->max_hang, opt->min_ovlp,
|
||||
&new_rtg_edges, opt->b_mask_t);
|
||||
opt->gap_fuzz, &new_rtg_edges, opt->b_mask_t);
|
||||
|
||||
kv_destroy(new_rtg_edges.a);
|
||||
return ug;
|
||||
@@ -2185,7 +2185,7 @@ h_covs *res, h_covs *cov_buf, h_covs *b_points, uint64_t local_bound, int unique
|
||||
span_e = MIN(span_e, len-1) + 1;
|
||||
//if(span_e - span_s <= ulen*BREAK_CUTOFF)//need it or not?
|
||||
{
|
||||
if(span_s >= sPos && span_e <= ePos)
|
||||
if(span_s >= sPos && span_e <= ePos)///test the density of this local region; bug -> should use any HiC pairs that are overlapped with [sPos, sPoe), instead of fully covered by [sPos, sPoe)
|
||||
{
|
||||
occ++;
|
||||
if(unique_only && hit->a.a[i].id == 0) continue;
|
||||
@@ -2590,18 +2590,18 @@ void update_h_w(h_w_t *e, dens_idx_t *idx, double *max_div)
|
||||
ii = 0;
|
||||
for (pi = l; pi < k; pi++)
|
||||
{
|
||||
pos = e->a[pi].d>>32;///
|
||||
pos = e->a[pi].d>>32;///loc of 3'-end
|
||||
while (ii < idn)
|
||||
{
|
||||
if(((uint32_t)id[ii]) == pos)
|
||||
{
|
||||
if(ori)
|
||||
if(ori)///the most left one
|
||||
{
|
||||
break;
|
||||
}
|
||||
else
|
||||
{
|
||||
while (ii < idn && (((uint32_t)id[ii]) == pos))
|
||||
while (ii < idn && (((uint32_t)id[ii]) == pos))///the most right one
|
||||
{
|
||||
ii++;
|
||||
}
|
||||
@@ -2612,7 +2612,7 @@ void update_h_w(h_w_t *e, dens_idx_t *idx, double *max_div)
|
||||
ii++;
|
||||
}
|
||||
if(ii >= idn) fprintf(stderr, "ERROR-1\n");
|
||||
e->a[pi].w += (ori? idn-ii: ii+1);
|
||||
e->a[pi].w += (ori? idn-ii: ii+1);///the smaller the better
|
||||
if(max_div) (*max_div) = MAX((*max_div), e->a[pi].w);
|
||||
}
|
||||
l = k;
|
||||
@@ -2785,7 +2785,7 @@ void update_scg(horder_t *h, trans_col_t *t_idx)
|
||||
if(!hits->a.a[i].id) continue;//hom hits
|
||||
suid = get_hit_suid(*hits, i);
|
||||
euid = get_hit_euid(*hits, i);
|
||||
if(suid == euid) continue;
|
||||
if(suid == euid) continue;//same unitig
|
||||
slen = ug->u.a[suid].len;
|
||||
elen = ug->u.a[euid].len;
|
||||
|
||||
@@ -2987,7 +2987,7 @@ void get_backbone_layout(horder_t *h, sc_lay_t *sl, osg_t *lg, uint8_t *vis)
|
||||
{
|
||||
///I guess this should be (!!(asg_arc_n(lg, k<<1)))^(!!(asg_arc_n(lg, (k<<1)+1)))?
|
||||
///no, since asg_arc_n is at most 1
|
||||
if((asg_arc_n(lg, k<<1))^(asg_arc_n(lg, (k<<1)+1)))
|
||||
if((asg_arc_n(lg, k<<1))^(asg_arc_n(lg, (k<<1)+1)))///end scaffolding
|
||||
{
|
||||
v = (asg_arc_n(lg, k<<1)?(k<<1):((k<<1)+1));
|
||||
if(vis[k<<1] || vis[(k<<1)+1]) continue;
|
||||
@@ -3512,7 +3512,7 @@ void generate_scaffold(ma_utg_t *su, lay_t *ly, ma_ug_t *pug, asg_t *rg)
|
||||
}
|
||||
if(i < ly->n - 2) kv_push(uint64_t, *su, (uint64_t)-1);
|
||||
}
|
||||
if(ly->n != 2) is_circle = 0;
|
||||
if(ly->n != 2) is_circle = 0;///ly-> == 2: single circle
|
||||
|
||||
for (i = 0, totalLen = 0; i < su->n-1; i++)
|
||||
{
|
||||
@@ -3936,7 +3936,7 @@ asg_t *i_rg, ma_ug_t* i_ug, bubble_type* bub, kv_u_trans_t *ref, ug_opt_t *opt,
|
||||
// output_hic_rtg(i_ug, h->r_g, opt, asm_opt.output_file_name);
|
||||
|
||||
// reduce_hamming_error(h->r_g, opt->sources, opt->coverage_cut, opt->max_hang, opt->min_ovlp, opt->gap_fuzz);
|
||||
reduce_hamming_error_adv(NULL, h->r_g, opt->sources, opt->coverage_cut, opt->max_hang, opt->min_ovlp, opt->gap_fuzz, opt->ruIndex, NULL);
|
||||
reduce_hamming_error_adv(NULL, h->r_g, opt->sources, opt->coverage_cut, opt->max_hang, opt->min_ovlp, opt->gap_fuzz, opt->ruIndex, NULL, (asm_opt.max_short_tip*2));
|
||||
/**
|
||||
scaffold_hap(h, t_idx, opt, round, asm_opt.output_file_name, FATHER);
|
||||
scaffold_hap(h, t_idx, opt, round, asm_opt.output_file_name, MOTHER);
|
||||
@@ -3967,6 +3967,363 @@ asg_t *i_rg, ma_ug_t* i_ug, bubble_type* bub, kv_u_trans_t *ref, ug_opt_t *opt,
|
||||
return h;
|
||||
}
|
||||
|
||||
int cmp_mc_edge_t_w(const void * a, const void * b)
|
||||
{
|
||||
if((*(mc_edge_t*)a).w == (*(mc_edge_t*)b).w) return 0;
|
||||
return (*(mc_edge_t*)a).w < (*(mc_edge_t*)b).w ? -1 : 1;
|
||||
}
|
||||
|
||||
void cal_chain_arch(scg_t *sg, const mc_match_t *ma, uint32_t v, uint32_t *va, uint32_t vn, uint64_t *idx, asg64_v *srt)
|
||||
{
|
||||
uint64_t z, n, j, t, w, o, l, k; double mw; osg_arc_t *p;
|
||||
srt->n = 0;
|
||||
for (z = 0; z < vn; z++) {
|
||||
assert((v == (idx[va[z]]>>32)) || v == ((uint32_t)idx[va[z]]));
|
||||
o = (ma->idx.a[va[z]]>>32);
|
||||
n = (uint32_t)ma->idx.a[va[z]];
|
||||
for (j = 0; j < n; ++j) {
|
||||
t = ((uint32_t)(ma->ma.a[o+j]).x);
|
||||
|
||||
w = idx[t]>>32;
|
||||
if((w != (uint32_t)-1) && ((w>>1) > (v>>1))) {
|
||||
kv_push(uint64_t, (*srt), ((w<<32)|(o+j)));
|
||||
}
|
||||
|
||||
w = (uint32_t)idx[t];
|
||||
if((w != (uint32_t)-1) && ((w>>1) > (v>>1))) {
|
||||
kv_push(uint64_t, (*srt), ((w<<32)|(o+j)));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
radix_sort_ho64(srt->a, srt->a + srt->n);
|
||||
for (l = 0, k = 1; k <= srt->n; k++) {
|
||||
if(k == srt->n || (srt->a[l]>>32) != (srt->a[k]>>32)) {
|
||||
w = srt->a[l]>>32; mw = 0;
|
||||
for (j = l; j < k; j++) {
|
||||
t = ((uint32_t)(ma->ma.a[(uint32_t)srt->a[j]]).x);
|
||||
assert((w == (idx[t]>>32)) || (w == ((uint32_t)idx[t])));
|
||||
mw += fabs(ma->ma.a[(uint32_t)srt->a[j]].w);
|
||||
}
|
||||
|
||||
p = osg_arc_pushp(sg->g);
|
||||
p->occ = p->del = 0;
|
||||
p->u = v; p->v = w; p->w = p->nw = mw;
|
||||
|
||||
p = osg_arc_pushp(sg->g);
|
||||
p->occ = p->del = 0;
|
||||
p->u = w; p->v = v; p->w = p->nw = mw;
|
||||
|
||||
l = k;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void prt_scg_t_arc(scg_t *sg)
|
||||
{
|
||||
uint64_t k;
|
||||
for (k = 0; k < sg->g->n_arc; k++) {
|
||||
fprintf(stderr, "[M::%s] k::%lu, v::%u, w::%u\n", __func__, k, sg->g->arc[k].u, sg->g->arc[k].v);
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t mc_clus_cut(scg_t *sg, mc_edge_t *sa, uint64_t sn, double cut_rate, uint64_t force_cut)
|
||||
{
|
||||
uint64_t i, k, kv, kw, v, w, nv, nw, cnt = 0, m = sn; double mm_ol, ol_max;
|
||||
osg_arc_t *av, *aw, *ve, *we;
|
||||
for (k = 0; k < sn; k++) {
|
||||
if(sa[k].x == ((uint64_t)-1)) {
|
||||
cnt++; continue;
|
||||
}
|
||||
if(sg->g->arc[sa[k].x].del) {
|
||||
sa[k].x = ((uint64_t)-1); cnt++;
|
||||
continue;
|
||||
}
|
||||
|
||||
v = sg->g->arc[sa[k].x].u;
|
||||
w = sg->g->arc[sa[k].x].v;
|
||||
|
||||
nv = asg_arc_n(sg->g, v); nw = asg_arc_n(sg->g, w);
|
||||
if(nv<=1 && nw <= 1) {
|
||||
sa[k].x = ((uint64_t)-1); cnt++;
|
||||
continue;
|
||||
}
|
||||
av = asg_arc_a(sg->g, v); aw = asg_arc_a(sg->g, w);
|
||||
|
||||
ve = &(sg->g->arc[sa[k].x]); we = NULL;
|
||||
for (i = 0; i < nw; ++i) {
|
||||
if (aw[i].v == v) {
|
||||
we = &(aw[i]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
// if(!((!we) && (!(we->del)))) {
|
||||
// fprintf(stderr, "[M::%s] sn::%lu, sg->g->n_arc::%u, sg->g->n_seq::%u, cut_rate::%f, v::%lu, w::%lu, x::%lu, v_beg::%lu, nv::%lu, w_beg::%lu, nw::%lu\n", __func__,
|
||||
// sn, sg->g->n_arc, sg->g->n_seq, cut_rate, v, w, sa[k].x,
|
||||
// (sg->g)->idx[(v)]>>32, nv, (sg->g)->idx[(w)]>>32, nw);
|
||||
// prt_scg_t_arc(sg);
|
||||
// }
|
||||
assert((we) && (!(we->del)));
|
||||
mm_ol = sg->g->arc[sa[k].x].nw;
|
||||
|
||||
for (i = kv = ol_max = 0; i < nv; ++i) {
|
||||
if(av[i].del) continue;
|
||||
kv++;
|
||||
if(ol_max < av[i].nw) ol_max = av[i].nw;
|
||||
}
|
||||
if (kv < 1) {
|
||||
sa[k].x = ((uint64_t)-1); cnt++;
|
||||
continue;
|
||||
}
|
||||
if (kv >= 2) {
|
||||
if ((!force_cut) && (mm_ol > (ol_max*cut_rate))) continue;
|
||||
}
|
||||
|
||||
|
||||
for (i = kw = ol_max = 0; i < nw; ++i) {
|
||||
if(aw[i].del) continue;
|
||||
kw++;
|
||||
if(ol_max < aw[i].nw) ol_max = aw[i].nw;
|
||||
}
|
||||
if (kw < 1) {
|
||||
sa[k].x = ((uint64_t)-1); cnt++;
|
||||
continue;
|
||||
}
|
||||
if (kw >= 2) {
|
||||
if ((!force_cut) && (mm_ol > (ol_max*cut_rate))) continue;
|
||||
}
|
||||
|
||||
if (kv <= 1 && kw <= 1) {
|
||||
sa[k].x = ((uint64_t)-1); cnt++;
|
||||
continue;
|
||||
}
|
||||
|
||||
ve->del = we->del = 1;
|
||||
sa[k].x = ((uint64_t)-1); cnt++;
|
||||
}
|
||||
|
||||
m = sn;
|
||||
if(cnt) {
|
||||
for (k = m = 0; k < sn; k++) {
|
||||
if(sa[k].x == ((uint64_t)-1)) continue;
|
||||
sa[m++] = sa[k];
|
||||
}
|
||||
}
|
||||
return m;
|
||||
}
|
||||
|
||||
void gen_mc_clus_backbone_layout(scg_t *sg, asg64_v *res, uint32_t *out, uint32_t out_n, uint32_t *buf)
|
||||
{
|
||||
uint64_t k, l, i, rn, n0, n1, v, z; int64_t m, s, e; osg_arc_t *t = NULL;
|
||||
kv_resize(uint64_t, *res, sg->g->n_seq); res->n = sg->g->n_seq;
|
||||
memset(res->a, 0, sizeof((*(res->a)))*res->n); rn = res->n;
|
||||
// fprintf(stderr, "[M::%s::]******Start******\n",__func__);
|
||||
|
||||
for (k = 0; k < sg->g->n_seq; k++) {
|
||||
///I guess this should be (!!(asg_arc_n(lg, k<<1)))^(!!(asg_arc_n(lg, (k<<1)+1)))?
|
||||
///no, since asg_arc_n is at most 1
|
||||
n0 = asg_arc_n(sg->g, (k<<1));
|
||||
n1 = asg_arc_n(sg->g, ((k<<1)+1));
|
||||
assert((n0 <= 1) && (n1 <= 1));
|
||||
///1&&0; 0&&0; 1&&1;
|
||||
if((n0^n1) || ((!n0) && (!n1))) {
|
||||
v = (n0?(k<<1):((k<<1)+1));
|
||||
// if(vis[k<<1] || vis[(k<<1)+1]) continue;
|
||||
// if((res->a[k]>>32) || ((uint32_t)res->a[k])) continue;
|
||||
if(res->a[k]) continue;
|
||||
|
||||
// kv_pushp(lay_t, *sl, &p);
|
||||
// kv_init(*p);
|
||||
// kv_push(uint32_t, *p, v^1);
|
||||
// kv_push(uint32_t, *p, v);
|
||||
// vis[v] = vis[v^1] = 1;
|
||||
// kv_push(uint64_t, *res, (((uint64_t)((v^1)<<32))|((uint64_t)(v))|((uint64_t)(0x8000000000000000))));
|
||||
kv_push(uint64_t, *res, (((uint64_t)((v^1)<<32))|((uint64_t)(v))));
|
||||
// res->a[v>>1] |= ((uint64_t)(1))<<32;
|
||||
// res->a[v>>1] |= ((uint64_t)(1));
|
||||
res->a[v>>1] = 1;
|
||||
|
||||
while (asg_arc_n(sg->g, v)) {
|
||||
v = (arc_first(sg->g, v).v)^1;
|
||||
// kv_push(uint32_t, *p, v^1);
|
||||
// kv_push(uint32_t, *p, v);
|
||||
// vis[v] = vis[v^1] = 1;
|
||||
kv_push(uint64_t, *res, ((uint64_t)((v^1)<<32))|((uint64_t)(v)));
|
||||
// res->a[v>>1] |= ((uint64_t)(1))<<32;
|
||||
// res->a[v>>1] |= ((uint64_t)(1));
|
||||
res->a[v>>1] = 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
for (k = 0; k < sg->g->n_seq; k++) {
|
||||
// if(vis[k<<1] || vis[(k<<1)+1]) continue;
|
||||
// if((res->a[k]>>32) || ((uint32_t)res->a[k])) continue;
|
||||
if(res->a[k]) continue;
|
||||
n0 = asg_arc_n(sg->g, (k<<1));
|
||||
n1 = asg_arc_n(sg->g, ((k<<1)+1));
|
||||
assert((n0 == 1) && (n1 == 1));///must within a circle
|
||||
|
||||
|
||||
v = k<<1; t = NULL;
|
||||
while (asg_arc_n(sg->g, v)) {
|
||||
if((!t) || (t->nw > arc_first(sg->g, v).nw)) {
|
||||
t = &(arc_first(sg->g, v));
|
||||
}
|
||||
v = (arc_first(sg->g, v).v)^1;
|
||||
if(v == (k<<1)) break;
|
||||
}
|
||||
|
||||
v = t->v^1;
|
||||
// kv_pushp(lay_t, *sl, &p);
|
||||
// kv_init(*p);
|
||||
// kv_push(uint32_t, *p, v^1);
|
||||
// kv_push(uint32_t, *p, v);
|
||||
// vis[v] = vis[v^1] = 1;
|
||||
// kv_push(uint64_t, *res, (((uint64_t)((v^1)<<32))|((uint64_t)(v))|((uint64_t)(0x8000000000000000))));
|
||||
kv_push(uint64_t, *res, ((uint64_t)((v^1)<<32))|((uint64_t)(v)));
|
||||
// res->a[v>>1] |= ((uint64_t)(1))<<32;
|
||||
// res->a[v>>1] |= ((uint64_t)(1));
|
||||
res->a[v>>1] = 1;
|
||||
|
||||
while (1) {
|
||||
v = (arc_first(sg->g, v).v)^1;
|
||||
// if(vis[v]) break;
|
||||
if(res->a[v>>1]) break;
|
||||
// kv_push(uint32_t, *p, v^1);
|
||||
// kv_push(uint32_t, *p, v);
|
||||
// vis[v] = vis[v^1] = 1;
|
||||
kv_push(uint64_t, *res, ((uint64_t)((v^1)<<32))|((uint64_t)(v)));
|
||||
// res->a[v>>1] |= ((uint64_t)(1))<<32;
|
||||
// res->a[v>>1] |= ((uint64_t)(1));
|
||||
res->a[v>>1] = 1;
|
||||
}
|
||||
}
|
||||
|
||||
// uint32_t db_on = 0, db_z = 0;
|
||||
assert((res->n-rn) == sg->g->n_seq);
|
||||
for (l = 0, k = 1, i = 0; k <= out_n; k++) {
|
||||
if(k == out_n || out[k] == (uint32_t)-1) {
|
||||
if(k > l) {
|
||||
res->a[i++] = ((l<<32)|(k)); ///db_on += k - l;
|
||||
} else {
|
||||
assert(k == out_n);
|
||||
}
|
||||
l = k + 1;
|
||||
}
|
||||
}
|
||||
assert(i == sg->g->n_seq && i == rn);
|
||||
for (k = rn, z = 0; k < res->n; k++) {
|
||||
v = (res->a[k]>>32);
|
||||
s = res->a[v>>1]>>32; e = (uint32_t)res->a[v>>1];
|
||||
if(!(v&1)) {
|
||||
for (m = s; m < e; m++) buf[z++] = out[m];
|
||||
} else {
|
||||
for (m = e-1; m >= s; m--) buf[z++] = out[m];
|
||||
}
|
||||
|
||||
// if(e > s) {
|
||||
// fprintf(stderr, "[M::%s::]\t#chain::%ld\tutg%.6ul->utg%.6ul\n",
|
||||
// __func__, (e-s), buf[z-(e-s)]+1, buf[z-1]+1);
|
||||
// }
|
||||
|
||||
buf[z++] = (uint32_t)-1;
|
||||
// db_z += e - s;
|
||||
}
|
||||
// if(!(z == out_n)) {
|
||||
// fprintf(stderr, "[M::%s] sg->g->n_arc::%u, sg->g->n_seq::%u, z::%lu, out_n::%u, db_z::%u, db_on::%u\n", __func__,
|
||||
// sg->g->n_arc, sg->g->n_seq, z, out_n, db_z, db_on);
|
||||
// }
|
||||
assert(z == out_n);
|
||||
memcpy(out, buf, sizeof((*out))*out_n);
|
||||
}
|
||||
|
||||
void layout_mc_clus_t(const mc_match_t *ma, uint32_t *a, uint32_t an, scg_t *sg, uint32_t *buf, uint64_t *idx, ma_ug_t* ug,
|
||||
double min_cut, double max_cut, uint64_t cut_round)
|
||||
{
|
||||
uint64_t i, k, l, len, z, cutoff, s, e, v, vn; mc_edge_t *sp;
|
||||
asg64_v srt; kv_init(srt); kvec_t(mc_edge_t) sm; kv_init(sm);
|
||||
osg_destroy(sg->g); sg->g = osg_init();
|
||||
memset(idx, -1, sizeof((*idx))*ug->g->n_seq);
|
||||
|
||||
// for (k = 0; k < an; k++) {
|
||||
// fprintf(stderr, "[M::%s] k::%lu, a[k]::%u, an::%u\n", __func__, k, a[k], an);
|
||||
// }
|
||||
for (l = 0, k = 1, i = 0; k <= an; k++) {
|
||||
if(k == an || a[k] == (uint32_t)-1) {
|
||||
// fprintf(stderr, "[M::%s] l::%lu, k::%lu\n", __func__, l, k);
|
||||
if(k > l) {
|
||||
osg_seq_set(sg->g, i, 0);
|
||||
for (z = l, len = 0; z < k; z++) len += ug->g->seq[a[z]].len;
|
||||
cutoff = len >> 1;
|
||||
for (z = l, len = 0; z < k; z++) {
|
||||
s = len; len += ug->g->seq[a[z]].len; e = len;
|
||||
if(s <= cutoff) {
|
||||
idx[a[z]] <<= 32; idx[a[z]] |= (i<<1);
|
||||
}
|
||||
if(e >= cutoff) {
|
||||
idx[a[z]] <<= 32; idx[a[z]] |= ((i<<1)+1);
|
||||
}
|
||||
}
|
||||
i++;
|
||||
} else {
|
||||
assert(k == an);
|
||||
}
|
||||
l = k+1;
|
||||
}
|
||||
}
|
||||
|
||||
for (l = 0, k = 1, i = 0; k <= an; k++) {
|
||||
if(k == an || a[k] == (uint32_t)-1) {
|
||||
if(k > l) {
|
||||
for (z = l, v = (i<<1), vn = 0; z < k; z++) {
|
||||
if((v == (idx[a[z]]>>32)) || v == ((uint32_t)idx[a[z]])) buf[vn++] = a[z];
|
||||
}
|
||||
cal_chain_arch(sg, ma, v, buf, vn, idx, &srt);
|
||||
|
||||
for (z = l, v = (i<<1)+1, vn = 0; z < k; z++) {
|
||||
if((v == (idx[a[z]]>>32)) || v == ((uint32_t)idx[a[z]])) buf[vn++] = a[z];
|
||||
}
|
||||
cal_chain_arch(sg, ma, v, buf, vn, idx, &srt);
|
||||
|
||||
i++;
|
||||
} else {
|
||||
assert(k == an);
|
||||
}
|
||||
l = k + 1;
|
||||
}
|
||||
}
|
||||
osg_cleanup(sg->g);
|
||||
|
||||
sm.n = 0;
|
||||
// kv_resize(mc_edge_t, sm, sg->g->n_arc);
|
||||
for (k = 0; k < sg->g->n_arc; k++) {
|
||||
if((asg_arc_n(sg->g, sg->g->arc[k].u)<=1) && (asg_arc_n(sg->g, sg->g->arc[k].v)<=1)) continue;
|
||||
kv_pushp(mc_edge_t, sm, &sp);
|
||||
sp->w = sg->g->arc[k].nw; sp->x = k;
|
||||
}
|
||||
// fprintf(stderr, "sm.n::%lu\n", (uint64_t)sm.n);
|
||||
qsort(sm.a, sm.n, sizeof(mc_edge_t), cmp_mc_edge_t_w);
|
||||
|
||||
|
||||
|
||||
double step = (cut_round==1?max_cut:((max_cut-min_cut)/(cut_round-1)));
|
||||
double drop = min_cut;
|
||||
|
||||
for (i = 0; i < cut_round; i++, drop += step) {
|
||||
if(drop > max_cut) drop = max_cut;
|
||||
sm.n = mc_clus_cut(sg, sm.a, sm.n, drop, 0);
|
||||
}
|
||||
mc_clus_cut(sg, sm.a, sm.n, 1.1, 1);
|
||||
osg_cleanup(sg->g);
|
||||
|
||||
gen_mc_clus_backbone_layout(sg, &srt, a, an, buf);
|
||||
|
||||
|
||||
kv_destroy(srt); kv_destroy(sm);
|
||||
}
|
||||
|
||||
void cpy_u_hits(kvec_pe_hit *u_hits, kvec_pe_hit *i_hits, uint32_t u_n)
|
||||
{
|
||||
uint64_t i;
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
#define __STDC_LIMIT_MACROS
|
||||
#include <stdint.h>
|
||||
#include "hic.h"
|
||||
#include "rcut.h"
|
||||
|
||||
#define get_hit_srev(x, k) ((x).a.a[(k)].s>>63)
|
||||
#define get_hit_slen(x, k) ((x).a.a[(k)].len>>32)
|
||||
@@ -71,4 +72,8 @@ void ha_aware_order(kvec_pe_hit *r_hits, asg_t *rg, ma_ug_t *ug_fa, ma_ug_t *ug_
|
||||
ug_opt_t *opt, uint32_t round);
|
||||
spg_t *horder_utg(kvec_pe_hit *i_hits, uint64_t i_hits_uid_bits, uint64_t i_hits_pos_mode,
|
||||
asg_t *i_rg, ma_ug_t* i_ug, bubble_type* bub, ug_opt_t *opt);
|
||||
void layout_mc_clus_t(const mc_match_t *ma, uint32_t *a, uint32_t an, scg_t *sg, uint32_t *buf, uint64_t *idx, ma_ug_t* ug,
|
||||
double min_cut, double max_cut, uint64_t cut_round);
|
||||
void osg_destroy(osg_t *g);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -385,7 +385,7 @@ ha_pt_t *ha_pt_gen(ha_ct_t *ct, int n_thread, int is_l)
|
||||
ha_ct_destroy_bf(ct);
|
||||
CALLOC(pt, 1);
|
||||
pt->k = ct->k, pt->pre = ct->pre, pt->tot = ct->tot;
|
||||
CALLOC(pt->h, 1<<pt->pre);
|
||||
CALLOC(pt->h, (((uint64_t)1)<<pt->pre));
|
||||
for (i = 0; i < 1<<pt->pre; ++i) {
|
||||
pt->h[i].h = yak_pt_init();
|
||||
yak_pt_resize(pt->h[i].h, kh_size(ct->h[i].h));
|
||||
@@ -422,7 +422,7 @@ ha_pt_t *ha_pt_gen_count(ha_ct_t *ct, int n_thread)
|
||||
ha_ct_destroy_bf(ct);
|
||||
CALLOC(pt, 1);
|
||||
pt->k = ct->k, pt->pre = ct->pre, pt->tot = ct->tot;
|
||||
CALLOC(pt->h, 1<<pt->pre);
|
||||
CALLOC(pt->h, (((uint64_t)1)<<pt->pre));
|
||||
for (i = 0; i < 1<<pt->pre; ++i) {
|
||||
pt->h[i].h = yak_pt_init();
|
||||
yak_pt_resize(pt->h[i].h, kh_size(ct->h[i].h));
|
||||
@@ -546,6 +546,21 @@ const int ha_pt_cnt(const ha_pt_t *h, uint64_t hash)
|
||||
return kh_key(g->h, k) & YAK_MAX_COUNT;
|
||||
}
|
||||
|
||||
inline uint64_t flt_quals(char *sc_a, uint64_t sc_l, uint64_t sc_off, int64_t sc_cut)
|
||||
{
|
||||
int64_t sc_min = sc_l * sc_cut, sc_tot; uint64_t k;
|
||||
for (k = sc_tot = 0; (k < sc_l) && (sc_tot < sc_min); k++) {
|
||||
sc_tot += (((uint8_t)sc_a[k]) - sc_off);
|
||||
}
|
||||
|
||||
// if(sc_tot < sc_min) {
|
||||
// fprintf(stderr, "[M::%s] sc_tot::%ld, sc_min::%ld, sc_l::%lu\n", __func__, sc_tot, sc_min, sc_l);
|
||||
// }
|
||||
|
||||
if(sc_tot < sc_min) return 0;
|
||||
return 1;
|
||||
}
|
||||
|
||||
/**********************************
|
||||
* Buffer for counting all k-mers *
|
||||
**********************************/
|
||||
@@ -563,7 +578,7 @@ KSEQ_INIT(gzFile, gzread)
|
||||
typedef struct { // global data structure for kt_pipeline()
|
||||
const yak_copt_t *opt;
|
||||
const void *flt_tab;
|
||||
int flag, create_new, is_store, uq;
|
||||
int flag, create_new, is_store, uq, ifq;
|
||||
uint64_t n_mz, n_seq; ///number of total reads
|
||||
kseq_t *ks;
|
||||
UC_Read ucr;
|
||||
@@ -692,6 +707,7 @@ static inline void sf##_pt_insert_buf(sf##_ch_buf_t *buf, int p, const HType *y)
|
||||
static void *sf##_worker_count(void *data, int step, void *in) /** callback for kt_pipeline()**/\
|
||||
{\
|
||||
pl_data_t *p = (pl_data_t*)data;\
|
||||
/**uint8_t src_a[1000000], des_a[1000000];**/\
|
||||
if (step == 0) { /** step 1: read a block of sequences**/\
|
||||
int ret;\
|
||||
sf##_st_data_t *s;\
|
||||
@@ -744,7 +760,8 @@ static void *sf##_worker_count(void *data, int step, void *in) /** callback for
|
||||
} else {\
|
||||
while ((ret = kseq_read(p->ks)) >= 0) {\
|
||||
int l = (int)(p->ks->seq.l) - (int)(p->opt->adaLen) - (int)(p->opt->adaLen);\
|
||||
if(l <= 0) continue;\
|
||||
if((l <= 0) || (l < asm_opt.rl_cut)) continue;\
|
||||
if((p->ifq) && (asm_opt.sc_cut > 0) && (!flt_quals(p->ks->qual.s+p->opt->adaLen, l, 33, asm_opt.sc_cut))) continue;\
|
||||
if (p->n_seq >= 1<<28) {\
|
||||
fprintf(stderr, "ERROR: this implementation supports no more than %d reads\n", 1<<28);\
|
||||
exit(1);\
|
||||
@@ -762,6 +779,16 @@ static void *sf##_worker_count(void *data, int step, void *in) /** callback for
|
||||
++n_N;\
|
||||
ha_compress_base(Get_READ(*p->rs_out, p->n_seq), p->ks->seq.s+p->opt->adaLen, l, &p->rs_out->N_site[p->n_seq], n_N);\
|
||||
memcpy(&p->rs_out->name[p->rs_out->name_index[p->n_seq]], p->ks->name.s, p->ks->name.l);\
|
||||
if(p->ifq) {\
|
||||
ha_compress_qual(Get_QUAL(*p->rs_out, p->n_seq), p->ks->qual.s+p->opt->adaLen, l, sc_bn, 33);\
|
||||
/**print_fastq(NULL, p->ks->name.s, p->ks->seq.s, p->ks->qual.s, (1<<sc_bn), 33);**/\
|
||||
/**if(l <= 1000000) {\
|
||||
convert_qual(src_a, p->ks->qual.s+p->opt->adaLen, l, (1<<sc_bn), 0, 33);\
|
||||
retrive_bqual(NULL, des_a, p->n_seq, -1, -1, 0, sc_bn);\
|
||||
if(memcmp(src_a, des_a, l)!=0) fprintf(stderr, "ERROR: incorrect qual values\n");\
|
||||
else fprintf(stderr, "Correct: correct qual values\n");\
|
||||
}**/\
|
||||
}\
|
||||
}\
|
||||
}\
|
||||
if (s->n_seq == s->m_seq) {\
|
||||
@@ -809,7 +836,7 @@ static void *sf##_worker_count(void *data, int step, void *in) /** callback for
|
||||
uint32_t j;\
|
||||
/**s->n_seq is how many reads at this buffer**/\
|
||||
/**s->mz && s->mz_buf are lists of minimzer vectors**/\
|
||||
CALLOC(s->mz, s->n_seq), CALLOC(s->mz_buf, p->opt->n_thread), CALLOC(s->mt, p->opt->n_thread);\
|
||||
CALLOC(s->mz, s->n_seq); CALLOC(s->mz_buf, p->opt->n_thread); CALLOC(s->mt, p->opt->n_thread);\
|
||||
/**calculate minimzers for each read, each read corresponds to one thread**/\
|
||||
kt_for(p->opt->n_thread, sf##_worker_for_mz, s, s->n_seq);\
|
||||
for (i = 0; i < p->opt->n_thread; ++i) free(s->mt[i].a), free(s->mz_buf[i].a);\
|
||||
@@ -900,7 +927,7 @@ void debug_adapter(const hifiasm_opt_t *asm_opt, All_reads *rs)
|
||||
exit(1);
|
||||
}
|
||||
|
||||
static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt_t *p0, ha_ct_t *c0, const void *flt_tab, All_reads *rs, ma_utg_v *us, int64_t *n_seq)
|
||||
static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt_t *p0, ha_ct_t *c0, const void *flt_tab, All_reads *rs, ma_utg_v *us, int64_t *n_seq, uint8_t ifq)
|
||||
{
|
||||
///for 0-th counting, flag = HAF_COUNT_ALL|HAF_RS_WRITE_LEN|HAF_CREATE_NEW
|
||||
int read_rs = (rs && (flag & HAF_RS_READ));
|
||||
@@ -908,7 +935,7 @@ static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt
|
||||
pl_data_t pl;
|
||||
gzFile fp = 0;
|
||||
memset(&pl, 0, sizeof(pl_data_t));
|
||||
pl.n_seq = *n_seq;
|
||||
pl.n_seq = *n_seq; pl.ifq = ifq;
|
||||
if(ug_rs) {
|
||||
pl.us_in = us;
|
||||
} else if (read_rs) {
|
||||
@@ -980,11 +1007,38 @@ ha_ct_t *ha_count(const hifiasm_opt_t *asm_o, int flag, int HPC, int k, int w, h
|
||||
opt.adaLen = (keep_adapter? asm_o->adapterLen : 0);
|
||||
opt.min_rcnt = (low_freq?*low_freq:-1);
|
||||
opt.uq = (unique_only?1:0);
|
||||
///asm_opt->num_reads is the number of fastq files
|
||||
for (i = n_bs = 0; i < (us?1:asm_o->num_reads); ++i){
|
||||
h = yak_count(&opt, asm_o->read_file_names[i], flag|HAF_CREATE_NEW, p0, h, flt_tab, rs, us, &n_seq);
|
||||
|
||||
n_bs = 0;
|
||||
///pending for integration
|
||||
/**
|
||||
if(rs && asm_o->ar && asm_o->ul_mod) {
|
||||
for (i = 0; i < asm_o->ar->n; ++i){
|
||||
h = yak_count(&opt, asm_o->ar->a[i], flag|HAF_CREATE_NEW, p0, h, flt_tab, rs, us, &n_seq);
|
||||
if(h) n_bs += h->bs;
|
||||
}
|
||||
}**/
|
||||
///asm_opt->num_reads is the number of fastq files
|
||||
for (i = 0; i < (us?1:asm_o->num_reads); ++i){
|
||||
h = yak_count(&opt, asm_o->read_file_names[i], flag|HAF_CREATE_NEW, p0, h, flt_tab, rs, us, &n_seq, asm_opt.is_sc);
|
||||
if(h) n_bs += h->bs;
|
||||
}
|
||||
|
||||
if((rs) && (flag & HAF_RS_WRITE_LEN) && (asm_opt.is_sc)) {
|
||||
rs->tqn = rs->total_reads; rs->tr[0] = rs->total_reads_bases;
|
||||
}
|
||||
|
||||
if((asm_o->hf) && (!us)) {
|
||||
for (i = 0; i < asm_o->hf->n; ++i) {
|
||||
h = yak_count(&opt, asm_o->hf->a[i], flag|HAF_CREATE_NEW, p0, h, flt_tab, rs, us, &n_seq, 0);
|
||||
if(h) n_bs += h->bs;
|
||||
}
|
||||
}
|
||||
if((rs) && (flag & HAF_RS_WRITE_LEN) && (asm_opt.is_sc)) {
|
||||
rs->tr[1] = rs->total_reads_bases - rs->tr[0];
|
||||
}
|
||||
|
||||
// fprintf(stderr, "[M::%s]\t# tqn::%lu, Ont base::%lu, # HiFi bases::%lu\n", __func__, R_INF.tqn, R_INF.tr[0], R_INF.tr[1]);
|
||||
|
||||
if(h) h->bs = n_bs;
|
||||
if (h && opt.bf_shift > 0)
|
||||
ha_ct_destroy_bf(h);
|
||||
@@ -1066,13 +1120,26 @@ void debug_ct_index(void* q_ct_idx, void* r_ct_idx)
|
||||
/*************************
|
||||
* High-level interfaces *
|
||||
*************************/
|
||||
void *ha_ft_ul_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int k, int w, int cutoff)
|
||||
|
||||
int64_t ha_ct_ug_cutoff(ha_ct_t *h, int64_t num_thre, double cut_rate)
|
||||
{
|
||||
yak_ft_t *flt_tab;
|
||||
ha_ct_t *h;
|
||||
int64_t cnt[YAK_N_COUNTS], k, tot_n = 0, tot_cutn = 0;
|
||||
ha_ct_hist(h, cnt, num_thre);
|
||||
for (k = tot_n = 0; k < YAK_N_COUNTS; k++) tot_n += cnt[k];
|
||||
tot_cutn = tot_n - (tot_n*cut_rate);
|
||||
|
||||
for (k = tot_n = 0; k < YAK_N_COUNTS && tot_n < tot_cutn; k++) tot_n += cnt[k];
|
||||
return k;
|
||||
}
|
||||
|
||||
void *ha_ft_ul_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int k, int w, int cutoff, int max_cutoff)
|
||||
{
|
||||
yak_ft_t *flt_tab; ha_ct_t *h;
|
||||
///HAF_COUNT_EXACT ---> no bf; HAF_COUNT_ALL ---> no minimizer
|
||||
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_UG_READ, !(asm_opt->flag&HA_F_NO_HPC), k, w, NULL, NULL, NULL, us, 0, NULL, 0);
|
||||
|
||||
if(cutoff < 0) cutoff = ha_ct_ug_cutoff(h, asm_opt->thread_num, 0.0002);
|
||||
if(cutoff > max_cutoff) cutoff = max_cutoff;
|
||||
// cutoff = (int)(asm_opt->hom_cov * asm_opt->high_factor);
|
||||
if (cutoff > YAK_MAX_COUNT - 1) cutoff = YAK_MAX_COUNT - 1;
|
||||
// fprintf(stderr, "[M::%s::] cutoff->%d\n\n", __func__, cutoff);
|
||||
@@ -1084,27 +1151,35 @@ void *ha_ft_ul_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int k, int w, int
|
||||
return (void*)flt_tab;
|
||||
}
|
||||
|
||||
void *ha_ft_ug_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int is_HPC, int k, int w, int min_freq, int max_freq)
|
||||
void *ha_ft_ug_gen(hifiasm_opt_t *asm_opt, ma_utg_v *us, int is_HPC, int k, int w, int min_freq, int max_freq)
|
||||
{
|
||||
yak_ft_t *flt_tab;
|
||||
ha_ct_t *h;
|
||||
ha_ct_t *h; ///int32_t b0;
|
||||
// if(min_freq < 0 || max_freq < 0) {
|
||||
// b0 = asm_opt->bf_shift; asm_opt->bf_shift = 0;
|
||||
// }
|
||||
///HAF_COUNT_EXACT ---> no bf; HAF_COUNT_ALL ---> no minimizer
|
||||
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_UG_READ|HAF_COUNT_EXACT, is_HPC, k, w, NULL, NULL, NULL, us, 0, NULL, 0);
|
||||
// if(min_freq < 0 || max_freq < 0) {
|
||||
// asm_opt->bf_shift = b0;
|
||||
|
||||
// }
|
||||
|
||||
ha_ct_shrink(h, min_freq, max_freq>YAK_MAX_COUNT-1?YAK_MAX_COUNT-1:max_freq, asm_opt->thread_num);
|
||||
flt_tab = gen_hh(h, YAK_MAX_COUNT);
|
||||
ha_ct_destroy(h);
|
||||
return (void*)flt_tab;
|
||||
}
|
||||
|
||||
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode)
|
||||
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode, int read_from_store)
|
||||
{
|
||||
yak_ft_t *flt_tab;
|
||||
int64_t cnt[YAK_N_COUNTS];
|
||||
int peak_hom, peak_het, cutoff = YAK_MAX_COUNT - 1, ex_flag = 0;
|
||||
if(is_hp_mode) ex_flag = HAF_RS_READ|HAF_SKIP_READ;
|
||||
ha_ct_t *h;
|
||||
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_RS_WRITE_LEN|ex_flag, !(asm_opt->flag&HA_F_NO_HPC), asm_opt->k_mer_length, asm_opt->mz_win, NULL, NULL, rs, NULL, 1, NULL, 0);
|
||||
if((asm_opt->flag & HA_F_VERBOSE_GFA))
|
||||
h = ha_count(asm_opt, HAF_COUNT_ALL|ex_flag|((read_from_store)?(HAF_RS_READ):(HAF_RS_WRITE_LEN)), !(asm_opt->flag&HA_F_NO_HPC), asm_opt->k_mer_length, asm_opt->mz_win, NULL, NULL, rs, NULL, 1, NULL, 0);
|
||||
if((asm_opt->flag & HA_F_VERBOSE_GFA) || (asm_opt->restart))
|
||||
{
|
||||
write_ct_index((void*)h, asm_opt->output_file_name);
|
||||
// load_ct_index(&ha_ct_table, asm_opt->output_file_name);
|
||||
@@ -1117,10 +1192,20 @@ void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int i
|
||||
{
|
||||
ha_ct_hist(h, cnt, asm_opt->thread_num);
|
||||
peak_hom = ha_analyze_count(YAK_N_COUNTS, asm_opt->min_hist_kmer_cnt, asm_opt->hg_size>0?(h->bs/asm_opt->hg_size):(-1), cnt, &peak_het);
|
||||
///r850
|
||||
fprintf(stderr, "[M::%s::auto] inferred peak_hom: %d; peak_het: %d\n", __func__, peak_hom, peak_het);
|
||||
if((asm_opt->het_cov_ss > 0) || (asm_opt->hmo_cov_ss > 0)) {
|
||||
fprintf(stderr, "[M::%s::user] override requested peak_hom: %ld; peak_het: %ld\n", __func__, asm_opt->hmo_cov_ss, asm_opt->het_cov_ss);
|
||||
if(asm_opt->hmo_cov_ss > 0) peak_hom = asm_opt->hmo_cov_ss;
|
||||
if(asm_opt->het_cov_ss > 0) peak_het = asm_opt->het_cov_ss;
|
||||
}
|
||||
if (hom_cov) *hom_cov = peak_hom;
|
||||
if (peak_hom > 0) fprintf(stderr, "[M::%s] peak_hom: %d; peak_het: %d\n", __func__, peak_hom, peak_het);
|
||||
if (peak_hom > 0) fprintf(stderr, "[M::%s::final] using peak_hom: %d; peak_het: %d\n", __func__, peak_hom, peak_het);
|
||||
|
||||
///in default, asm_opt->high_factor = 5.0
|
||||
///r833
|
||||
cutoff = (int)(peak_hom * asm_opt->high_factor);
|
||||
if(cutoff < asm_opt->hf_cutoff) cutoff = asm_opt->hf_cutoff;
|
||||
if (cutoff > YAK_MAX_COUNT - 1) cutoff = YAK_MAX_COUNT - 1;
|
||||
}
|
||||
ha_ct_shrink(h, cutoff, YAK_MAX_COUNT, asm_opt->thread_num);
|
||||
@@ -1215,9 +1300,16 @@ ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_f
|
||||
ha_ct_hist(ct, cnt, asm_opt->thread_num);
|
||||
fprintf(stderr, "[M::%s] count[%d] = %ld (for sanity check)\n", __func__, YAK_MAX_COUNT, (long)cnt[YAK_MAX_COUNT]);
|
||||
peak_hom = ha_analyze_count(YAK_N_COUNTS, asm_opt->min_hist_kmer_cnt, asm_opt->hg_size>0?(ct->bs/asm_opt->hg_size):(-1), cnt, &peak_het);
|
||||
///r850
|
||||
fprintf(stderr, "[M::%s::auto] inferred peak_hom: %d; peak_het: %d\n", __func__, peak_hom, peak_het);
|
||||
if((asm_opt->het_cov_ss > 0) || (asm_opt->hmo_cov_ss > 0)) {
|
||||
fprintf(stderr, "[M::%s::user] override requested peak_hom: %ld; peak_het: %ld\n", __func__, asm_opt->hmo_cov_ss, asm_opt->het_cov_ss);
|
||||
if(asm_opt->hmo_cov_ss > 0) peak_hom = asm_opt->hmo_cov_ss;
|
||||
if(asm_opt->het_cov_ss > 0) peak_het = asm_opt->het_cov_ss;
|
||||
}
|
||||
if (hom_cov) *hom_cov = peak_hom;
|
||||
if (het_cov) *het_cov = peak_het;
|
||||
if (peak_hom > 0) fprintf(stderr, "[M::%s] peak_hom: %d; peak_het: %d\n", __func__, peak_hom, peak_het);
|
||||
if (peak_hom > 0) fprintf(stderr, "[M::%s::final] using peak_hom: %d; peak_het: %d\n", __func__, peak_hom, peak_het);
|
||||
///here ha_ct_shrink is mostly used to remove k-mer appearing only 1 time
|
||||
if (flt_tab == 0) {
|
||||
int cutoff = (int)(peak_hom * asm_opt->high_factor);
|
||||
@@ -1329,8 +1421,9 @@ int load_ct_index(void **i_ct_idx, char* file_name)
|
||||
|
||||
int write_pt_index(void *flt_tab, ha_pt_t *ha_idx, All_reads* r, hifiasm_opt_t* opt, char* file_name)
|
||||
{
|
||||
char* gfa_name = (char*)malloc(strlen(file_name)+25);
|
||||
sprintf(gfa_name, "%s.pt_flt", file_name);
|
||||
char* gfa_name = (char*)malloc(strlen(file_name)+64);
|
||||
if(r) sprintf(gfa_name, "%s.pt_flt", file_name);
|
||||
else sprintf(gfa_name, "%s.pt_flt.bin", file_name);
|
||||
FILE* fp = fopen(gfa_name, "w");
|
||||
if (!fp) {
|
||||
free(gfa_name);
|
||||
@@ -1369,9 +1462,24 @@ int write_pt_index(void *flt_tab, ha_pt_t *ha_idx, All_reads* r, hifiasm_opt_t*
|
||||
fwrite(&opt->het_cov, sizeof(opt->het_cov), 1, fp);
|
||||
fwrite(&opt->max_n_chain, sizeof(opt->max_n_chain), 1, fp);
|
||||
|
||||
|
||||
if(r) {
|
||||
write_All_reads(r, gfa_name);
|
||||
|
||||
sprintf(gfa_name, "%s.pt_flt.paf.bin", file_name);
|
||||
fclose(fp); fp = fopen(gfa_name, "w"); uint64_t k;
|
||||
if (!fp) {
|
||||
free(gfa_name);
|
||||
return 0;
|
||||
}
|
||||
fwrite(&(r->total_reads), sizeof(r->total_reads), 1, fp);
|
||||
for (k = 0; k < r->total_reads; k++) {
|
||||
fwrite(&(r->paf[k].is_fully_corrected), sizeof(r->paf[k].is_fully_corrected), 1, fp);
|
||||
fwrite(&(r->paf[k].is_abnormal), sizeof(r->paf[k].is_abnormal), 1, fp);
|
||||
fwrite(&(r->paf[k].length), sizeof(r->paf[k].length), 1, fp);
|
||||
fwrite(r->paf[k].buffer, sizeof((*(r->paf[k].buffer))), r->paf[k].length, fp);
|
||||
}
|
||||
}
|
||||
|
||||
fprintf(stderr, "[M::%s] Index has been written.\n", __func__);
|
||||
free(gfa_name);
|
||||
fclose(fp);
|
||||
@@ -1380,8 +1488,10 @@ int write_pt_index(void *flt_tab, ha_pt_t *ha_idx, All_reads* r, hifiasm_opt_t*
|
||||
|
||||
int load_pt_index(void **r_flt_tab, ha_pt_t **r_ha_idx, All_reads *r, hifiasm_opt_t* opt, char* file_name)
|
||||
{
|
||||
char* gfa_name = (char*)malloc(strlen(file_name)+25);
|
||||
sprintf(gfa_name, "%s.pt_flt", file_name);
|
||||
char* gfa_name = (char*)malloc(strlen(file_name)+64);
|
||||
if(r) sprintf(gfa_name, "%s.pt_flt", file_name);
|
||||
else sprintf(gfa_name, "%s.pt_flt.bin", file_name);
|
||||
// fprintf(stderr, "[M::%s]\tgfa_name::%s\n", __func__, gfa_name);
|
||||
FILE* fp = fopen(gfa_name, "r");
|
||||
if (!fp) {
|
||||
free(gfa_name);
|
||||
@@ -1460,25 +1570,107 @@ int load_pt_index(void **r_flt_tab, ha_pt_t **r_ha_idx, All_reads* r, hifiasm_op
|
||||
f_flag += fread(&opt->max_n_chain, sizeof(opt->max_n_chain), 1, fp);
|
||||
|
||||
|
||||
// fclose(fp);
|
||||
if(r) {
|
||||
if(!load_All_reads(r, gfa_name)) {
|
||||
free(gfa_name);
|
||||
return 0;
|
||||
}
|
||||
|
||||
memset(r->trio_flag, AMBIGU, r->total_reads*sizeof(uint8_t));
|
||||
|
||||
sprintf(gfa_name, "%s.pt_flt.paf.bin", file_name);
|
||||
fclose(fp); fp = fopen(gfa_name, "r"); uint64_t k;
|
||||
if (!fp) {
|
||||
free(gfa_name);
|
||||
return 0;
|
||||
}
|
||||
f_flag += fread(&(r->total_reads), sizeof(r->total_reads), 1, fp);
|
||||
r->paf = (ma_hit_t_alloc*)malloc(sizeof(ma_hit_t_alloc)*r->total_reads);
|
||||
r->reverse_paf = (ma_hit_t_alloc*)malloc(sizeof(ma_hit_t_alloc)*r->total_reads);
|
||||
for (k = 0; k < r->total_reads; k++) {
|
||||
// init_ma_hit_t_alloc(&(r->paf[k]));
|
||||
init_ma_hit_t_alloc(&(r->reverse_paf[k]));
|
||||
|
||||
f_flag += fread(&(r->paf[k].is_fully_corrected), sizeof(r->paf[k].is_fully_corrected), 1, fp);
|
||||
f_flag += fread(&(r->paf[k].is_abnormal), sizeof(r->paf[k].is_abnormal), 1, fp);
|
||||
f_flag += fread(&(r->paf[k].length), sizeof(r->paf[k].length), 1, fp);
|
||||
r->paf[k].size = r->paf[k].length;
|
||||
|
||||
r->paf[k].buffer = NULL;
|
||||
if(r->paf[k].length == 0) continue;
|
||||
|
||||
r->paf[k].buffer = (ma_hit_t*)malloc(sizeof(ma_hit_t)*r->paf[k].length);
|
||||
fread(r->paf[k].buffer, sizeof((*(r->paf[k].buffer))), r->paf[k].length, fp);
|
||||
}
|
||||
}
|
||||
fclose(fp);
|
||||
|
||||
if(!load_All_reads(r, gfa_name))
|
||||
fprintf(stderr, "[M::%s] Index has been loaded.\n", __func__);
|
||||
|
||||
free(gfa_name);
|
||||
return 1;
|
||||
}
|
||||
|
||||
void refresh_pt_idx(void **flt_tab, ha_pt_t **ha_idx, All_reads *r, hifiasm_opt_t *opt, char *file_name, uint8_t is_w)
|
||||
{
|
||||
char* gfa_name = (char*)malloc(strlen(file_name)+64);
|
||||
sprintf(gfa_name, "%s.ad", file_name);
|
||||
|
||||
if(is_w) {
|
||||
write_pt_index(*flt_tab, *ha_idx, NULL, opt, gfa_name);
|
||||
} else {
|
||||
load_pt_index(flt_tab, ha_idx, NULL, opt, gfa_name);
|
||||
}
|
||||
|
||||
free(gfa_name);
|
||||
}
|
||||
|
||||
|
||||
uint64_t tmp_pt_pro(void **r_flt_tab, ha_pt_t **r_ha_idx, All_reads *r, hifiasm_opt_t *opt, char *file_name, uint64_t rr, uint64_t tot_rr, uint64_t is_load)
|
||||
{
|
||||
char* gfa_name = (char*)malloc(strlen(file_name)+64);
|
||||
FILE *fp = NULL; int f_flag = 0; uint64_t rr0 = (uint64_t)-1, tot_rr0 = (uint64_t)-1;
|
||||
|
||||
|
||||
if(is_load) {
|
||||
sprintf(gfa_name, "%s.r%lu.ht.bin", file_name, rr);
|
||||
fp = fopen(gfa_name, "r");
|
||||
if (!fp) {
|
||||
free(gfa_name);
|
||||
return 0;
|
||||
}
|
||||
f_flag += fread(&rr0, sizeof(rr0), 1, fp);
|
||||
f_flag += fread(&tot_rr0, sizeof(tot_rr0), 1, fp);
|
||||
fclose(fp);
|
||||
if(rr0 != rr || tot_rr0 != tot_rr) {
|
||||
free(gfa_name);
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
memset(r->trio_flag, AMBIGU, r->total_reads*sizeof(uint8_t));
|
||||
r->paf = (ma_hit_t_alloc*)malloc(sizeof(ma_hit_t_alloc)*r->total_reads);
|
||||
r->reverse_paf = (ma_hit_t_alloc*)malloc(sizeof(ma_hit_t_alloc)*r->total_reads);
|
||||
for (i = 0; i < (long long)r->total_reads; i++)
|
||||
{
|
||||
init_ma_hit_t_alloc(&(r->paf[i]));
|
||||
init_ma_hit_t_alloc(&(r->reverse_paf[i]));
|
||||
}
|
||||
|
||||
fprintf(stderr, "[M::%s] Index has been loaded.\n", __func__);
|
||||
sprintf(gfa_name, "%s.r%lu", file_name, rr);
|
||||
if(!load_pt_index(r_flt_tab, r_ha_idx, r, opt, gfa_name)) {
|
||||
free(gfa_name);
|
||||
return 0;
|
||||
}
|
||||
} else {
|
||||
sprintf(gfa_name, "%s.r%lu", file_name, rr);
|
||||
write_pt_index(*r_flt_tab, *r_ha_idx, r, opt, gfa_name);
|
||||
|
||||
|
||||
|
||||
sprintf(gfa_name, "%s.r%lu.ht.bin", file_name, rr);
|
||||
fp = fopen(gfa_name, "w");
|
||||
if (!fp) {
|
||||
free(gfa_name);
|
||||
return 0;
|
||||
}
|
||||
fwrite(&rr, sizeof(rr), 1, fp);
|
||||
fwrite(&tot_rr, sizeof(tot_rr), 1, fp);
|
||||
fclose(fp);
|
||||
}
|
||||
|
||||
free(gfa_name);
|
||||
return 1;
|
||||
|
||||
@@ -72,9 +72,9 @@ extern void *ha_flt_tab_hp;
|
||||
extern ha_pt_t *ha_idx_hp;
|
||||
extern void *ha_ct_table;
|
||||
|
||||
void *ha_ft_ul_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int k, int w, int cutoff);
|
||||
void *ha_ft_ug_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int is_HPC, int k, int w, int min_freq, int max_freq);
|
||||
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode);
|
||||
void *ha_ft_ul_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int k, int w, int cutoff, int max_cutoff);
|
||||
void *ha_ft_ug_gen(hifiasm_opt_t *asm_opt, ma_utg_v *us, int is_HPC, int k, int w, int min_freq, int max_freq);
|
||||
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode, int read_from_store);
|
||||
int32_t ha_ft_cnt(const void *hh, uint64_t y);
|
||||
void ha_ft_destroy(void *h);
|
||||
|
||||
@@ -88,11 +88,13 @@ const int ha_pt_cnt(const ha_pt_t *h, uint64_t hash);
|
||||
|
||||
int write_pt_index(void *flt_tab, ha_pt_t *ha_idx, All_reads* r, hifiasm_opt_t* opt, char* file_name);
|
||||
int load_pt_index(void **r_flt_tab, ha_pt_t **r_ha_idx, All_reads* r, hifiasm_opt_t* opt, char* file_name);
|
||||
void refresh_pt_idx(void **flt_tab, ha_pt_t **ha_idx, All_reads *r, hifiasm_opt_t *opt, char *file_name, uint8_t is_w);
|
||||
int uidx_write(void *flt_tab, ha_pt_t *ha_idx, char* file_name, ma_ug_t *ug);
|
||||
int uidx_load(void **r_flt_tab, ha_pt_t **r_ha_idx, char* file_name, ma_ug_t *ug);
|
||||
int write_ct_index(void *ct_idx, char* file_name);
|
||||
int load_ct_index(void **ct_idx, char* file_name);
|
||||
int query_ct_index(void* ct_idx, uint64_t hash);
|
||||
uint64_t tmp_pt_pro(void **r_flt_tab, ha_pt_t **r_ha_idx, All_reads *r, hifiasm_opt_t *opt, char *file_name, uint64_t rr, uint64_t tot_rr, uint64_t is_load);
|
||||
|
||||
ha_abuf_t *ha_abuf_init_buf(void *km);
|
||||
ha_abufl_t *ha_abufl_init_buf(void *km);
|
||||
@@ -108,6 +110,7 @@ uint64_t ha_abufl_mem(const ha_abufl_t *ab);
|
||||
|
||||
double yak_cputime(void);
|
||||
void yak_reset_realtime(void);
|
||||
double yak_realtime_0(void);
|
||||
double yak_realtime(void);
|
||||
long yak_peakrss(void);
|
||||
double yak_peakrss_in_gb(void);
|
||||
@@ -116,6 +119,7 @@ double yak_cpu_usage(void);
|
||||
void ha_triobin(const hifiasm_opt_t *opt);
|
||||
uint32_t test_yak_binning(char* fn, char *cmd);
|
||||
uint32_t *ha_polybin_list(const hifiasm_opt_t *opt);
|
||||
uint32_t *ha_charbin_list(const hifiasm_opt_t *opt, uint8_t **idx, uint32_t *idx_n);
|
||||
|
||||
void mz1_ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mz1_v *p, const void *hf, int sample_dist, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct, ha_pt_t *pt, int min_freq, int32_t dp_min_len, float dp_e, st_mt_t *mt, int32_t ws, int32_t is_unique, void *km);
|
||||
void mz2_ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mzl_v *p, const void *hf, int sample_dist, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct, ha_pt_t *pt, int min_freq, int32_t dp_min_len, float dp_e, st_mt_t *mt, int32_t ws, int32_t is_unique, void *km);
|
||||
@@ -124,6 +128,7 @@ int adj_m_peak_hom(int m_peak_hom, int max_i, int max2_i, int max3_i, int *peak_
|
||||
void print_hist_lines(int n_cnt, int start_cnt, const int64_t *cnt);
|
||||
void debug_adapter(const hifiasm_opt_t *asm_opt, All_reads *rs);
|
||||
|
||||
|
||||
inline int mz_low_b(int peak_hom, int peak_het)
|
||||
{
|
||||
int low_freq = 2;
|
||||
@@ -164,7 +169,7 @@ static inline uint64_t yak_hash_long(uint64_t x[4])
|
||||
return yak_hash64_64(x[j<<1|0]) + yak_hash64_64(x[j<<1|1]);
|
||||
}
|
||||
|
||||
#define CALLOC(ptr, len) ((ptr) = (__typeof__(ptr))calloc((len), sizeof(*(ptr))))
|
||||
#define CALLOC(ptr, len) ((ptr) = ((((len)*sizeof(*(ptr))) <= 9223372036854775807)?((__typeof__(ptr))calloc((len), sizeof(*(ptr)))):(NULL)))
|
||||
#define MALLOC(ptr, len) ((ptr) = (__typeof__(ptr))malloc((len) * sizeof(*(ptr))))
|
||||
#define REALLOC(ptr, len) ((ptr) = (__typeof__(ptr))realloc((ptr), (len) * sizeof(*(ptr))))
|
||||
#define MEMCPY(dest, src, len) (memcpy((dest), (src), (len) * sizeof(*(src))))
|
||||
|
||||
@@ -120,9 +120,17 @@ void clear_all_ul_t(all_ul_t *x);
|
||||
void trans_base_infer(ma_ug_t *ug, asg_t *sg, ug_opt_t *uopt, kv_u_trans_t *res, bubble_type *bub);
|
||||
hpc_re_t *gen_hpc_re_t(ma_ug_t *ug);
|
||||
idx_emask_t* graph_ovlp_binning(ma_ug_t *ug, asg_t *sg, const ug_opt_t *uopt);
|
||||
uint32_t gen_src_shared_interval_simple(uint32_t src, ma_ug_t *ug, kv_ul_ov_t *res);
|
||||
uint32_t gen_src_shared_interval_simple(uint32_t src, ma_ug_t *ug, uint64_t *flt, uint64_t flt_n, kv_ul_ov_t *res);
|
||||
uint64_t check_ul_ov_t_consist(ul_ov_t *x, ul_ov_t *y, int64_t ql, int64_t tl, double diff);
|
||||
uint32_t infer_se(uint32_t qs, uint32_t qe, uint32_t ts, uint32_t te, uint32_t rev,
|
||||
uint32_t rqs, uint32_t rqe, uint32_t *rts, uint32_t *rte);
|
||||
uint32_t clean_contain_g(const ug_opt_t *uopt, asg_t *sg, uint32_t push_trans);
|
||||
void dedup_contain_g(const ug_opt_t *uopt, asg_t *sg);
|
||||
void trans_base_mmhap_infer(ma_ug_t *ug, asg_t *sg, ug_opt_t *uopt, kv_u_trans_t *res);
|
||||
scaf_res_t *gen_contig_path(const ug_opt_t *uopt, asg_t *sg, ma_ug_t *ctg, ma_ug_t *ref);
|
||||
void gen_contig_trans(const ug_opt_t *uopt, asg_t *sg, ma_ug_t *qry, scaf_res_t *qry_sc, ma_ug_t *ref, scaf_res_t *ref_sc, ma_ug_t *gfa, kv_u_trans_t *ta, uint32_t qoff, uint32_t toff, bubble_type *bu, kv_u_trans_t *res);
|
||||
void gen_contig_self(const ug_opt_t *uopt, asg_t *sg, ma_ug_t *db, scaf_res_t *db_sc, ma_ug_t *gfa, kv_u_trans_t *ta, uint64_t soff, bubble_type *bu, kv_u_trans_t *res, uint32_t is_exact);
|
||||
void order_contig_trans(kv_u_trans_t *in);
|
||||
void sort_uc_block_qe(uc_block_t* a, uint64_t a_n);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -8,6 +8,7 @@
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
// setvbuf(stderr, NULL, _IONBF, 0);
|
||||
int i, ret;
|
||||
yak_reset_realtime();
|
||||
init_opt(&asm_opt);
|
||||
@@ -62,12 +63,40 @@ int main(int argc, char *argv[])
|
||||
// ed_band_cal_global_128bit((char*)"ACTTTTTT", 8, (char*)"AATTTT", 6, 3));
|
||||
// exit(1);
|
||||
|
||||
ret = ha_assemble();
|
||||
int8_t simd_auto = 0;
|
||||
#if defined(__x86_64__) || defined(__i386__)
|
||||
__builtin_cpu_init();
|
||||
if (__builtin_cpu_supports("avx512f")) simd_auto = 2;
|
||||
else if (__builtin_cpu_supports("avx2")) simd_auto = 1;
|
||||
else simd_auto = 0;
|
||||
#endif
|
||||
|
||||
if (simd_auto) {
|
||||
fprintf(stderr, "[M::%s::auto] detected CPU support for %s\n", __func__, (simd_auto == 2) ? "AVX-512" : "AVX2");
|
||||
} else {
|
||||
fprintf(stderr, "[M::%s::auto] no supported SIMD extension detected; falling back to non-SIMD\n", __func__);
|
||||
}
|
||||
|
||||
if (asm_opt.simd_mm == 0 || asm_opt.simd_mm == 1 || asm_opt.simd_mm == 2) {
|
||||
fprintf(stderr, "[M::%s::user] user requested %s mode\n", __func__, (asm_opt.simd_mm == 2) ? "AVX-512" : ((asm_opt.simd_mm == 1) ? "AVX2" : "non-SIMD"));
|
||||
if(asm_opt.simd_mm > simd_auto) asm_opt.simd_mm = simd_auto;
|
||||
} else {
|
||||
asm_opt.simd_mm = simd_auto;
|
||||
if(asm_opt.simd_mm >= 1) asm_opt.simd_mm = 1; ///use avx2 rather than avx512, looks like avx512 still has issues right now
|
||||
}
|
||||
|
||||
fprintf(stderr, "[M::%s::final] using %s mode\n", __func__, (asm_opt.simd_mm == 2) ? "AVX-512" : ((asm_opt.simd_mm == 1) ? "AVX2" : "non-SIMD"));
|
||||
|
||||
|
||||
if(asm_opt.sec_in) ret = ha_assemble_pair();
|
||||
else if(asm_opt.dbg_ovec_cal) ret = ha_ec_dbg();
|
||||
else ret = ha_assemble();
|
||||
|
||||
destory_opt(&asm_opt);
|
||||
fprintf(stderr, "[M::%s] Version: %s\n", __func__, HA_VERSION);
|
||||
fprintf(stderr, "[M::%s] CMD:", __func__);
|
||||
for (i = 0; i < argc; ++i)
|
||||
fprintf(stderr, " %s", argv[i]);
|
||||
fprintf(stderr, "\n[M::%s] Real time: %.3f sec; CPU: %.3f sec; Peak RSS: %.3f GB\n", __func__, yak_realtime(), yak_cputime(), yak_peakrss_in_gb());
|
||||
fprintf(stderr, "\n[M::%s] Real time: %.3f sec; CPU: %.3f sec; Peak RSS: %.3f GB; SIMD: %u\n", __func__, yak_realtime(), yak_cputime(), yak_peakrss_in_gb(), asm_opt.simd_mm);
|
||||
return ret;
|
||||
}
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#include "ksort.h"
|
||||
#include "kthread.h"
|
||||
#include "hic.h"
|
||||
#include "horder.h"
|
||||
|
||||
#define VERBOSE_CUT 0
|
||||
|
||||
@@ -116,6 +117,8 @@ typedef struct{
|
||||
uint32_t n, n_thread;
|
||||
clus_flip_aux *aux;
|
||||
mc_svaux_t *baux;
|
||||
asg64_v *asn;
|
||||
scg_t sg;
|
||||
} mc_clus_t;
|
||||
|
||||
typedef struct {
|
||||
@@ -803,8 +806,19 @@ uint32_t mc_edges_symm(mc_match_t *ma)
|
||||
|
||||
static void normalize_mb_edge(mb_edge_t *a, mb_edge_t *b)
|
||||
{
|
||||
if(a->w >= b->w)
|
||||
{
|
||||
uint8_t f = 0;
|
||||
if(a->w[0] > b->w[0]) {
|
||||
f = 1;
|
||||
} else if(a->w[0] == b->w[0] && a->w[1] > b->w[1]) {
|
||||
f = 1;
|
||||
} else if(a->w[0] == b->w[0] && a->w[1] == b->w[1] && a->w[2] > b->w[2]) {
|
||||
f = 1;
|
||||
} else if(a->w[0] == b->w[0] && a->w[1] == b->w[1] && a->w[2] == b->w[2] && a->w[3] > b->w[3]) {
|
||||
f = 1;
|
||||
}
|
||||
|
||||
// if(a->w >= b->w) {
|
||||
if(f) {
|
||||
b->x = (uint32_t)a->x;
|
||||
b->x <<= 32;
|
||||
b->x |= (a->x>>32);
|
||||
@@ -812,9 +826,7 @@ static void normalize_mb_edge(mb_edge_t *a, mb_edge_t *b)
|
||||
b->w[3] = a->w[3];
|
||||
b->w[1] = a->w[2];
|
||||
b->w[2] = a->w[1];
|
||||
}
|
||||
else
|
||||
{
|
||||
} else {
|
||||
a->x = (uint32_t)b->x;
|
||||
a->x <<= 32;
|
||||
a->x |= (b->x>>32);
|
||||
@@ -2340,13 +2352,13 @@ void renew_mc_clus_t(mc_clus_t *bc, uint32_t *a, uint32_t a_n)
|
||||
{
|
||||
if(!bc) return;
|
||||
// fprintf(stderr, "[M::%s] a_n::%u\n", __func__, a_n);
|
||||
uint32_t k, i, *ba, bn, m, cocc, iin, bub_occ = 0, bbn = 0; uint64_t *p; ma_utg_t *u = NULL;
|
||||
uint32_t k, i, *ba, bn, m, cocc, iin, /**bub_occ = 0,**/ bbn = 0; uint64_t *p; ma_utg_t *u = NULL;
|
||||
kvec_t(double) sc_l; kvec_t(double) sc_r; kvec_t(double) sc_m; kvec_t(uint32_t) tmp;
|
||||
kv_init(sc_l); kv_init(sc_r); kv_init(sc_m); kv_init(tmp);
|
||||
|
||||
bc->cc.ng.n = bc->cc.nn.n = 0;
|
||||
memset(bc->lock, 0, sizeof((*(bc->lock)))*bc->n);
|
||||
for (k = 0; k < a_n; k++) bc->lock[a[k]] = 1;
|
||||
memset(bc->lock, 0, sizeof((*(bc->lock)))*bc->n);///bc->n: number of all nodes
|
||||
for (k = 0; k < a_n; k++) bc->lock[a[k]] = 1;///a_n: number of nodes within the cluster
|
||||
|
||||
kv_resize(uint32_t, bc->cc.nn, a_n);
|
||||
// for (i = 0; i < bc->bub->chain_weight.n; i++) {
|
||||
@@ -2355,14 +2367,14 @@ void renew_mc_clus_t(mc_clus_t *bc, uint32_t *a, uint32_t a_n)
|
||||
for (i = 0; i < bc->bub->b_ug->u.n; i++) {
|
||||
u = &(bc->bub->b_ug->u.a[i]);
|
||||
if(u->n == 0) continue;
|
||||
bub_occ += u->n;
|
||||
// bub_occ += u->n;
|
||||
// fprintf(stderr, "[M::%s] i::%u, u->n::%u\n", __func__, i, (uint32_t)u->n);
|
||||
for (k = cocc = 0, iin = bc->cc.nn.n; k < u->n; k++) {
|
||||
get_bubbles(bc->bub, u->a[k]>>33, NULL, NULL, &ba, &bn, NULL);
|
||||
kv_pushp(uint64_t, bc->cc.ng, &p); bbn += bn;
|
||||
*p = bc->cc.nn.n;///a bubble
|
||||
for (m = 0; m < bn; m++) {
|
||||
if(!(bc->lock[ba[m]>>1])) continue;
|
||||
if(!(bc->lock[ba[m]>>1])) continue;///if the node of bubble is not at this cluster
|
||||
kv_push(uint32_t, bc->cc.nn, (ba[m]>>1));
|
||||
bc->lock[ba[m]>>1] = 2;
|
||||
}
|
||||
@@ -2372,7 +2384,7 @@ void renew_mc_clus_t(mc_clus_t *bc, uint32_t *a, uint32_t a_n)
|
||||
}
|
||||
*p <<= 32; *p |= (bc->cc.nn.n-((*p)>>32)); cocc++;
|
||||
}
|
||||
//split chains
|
||||
//split the chain if there is multipe bubbles
|
||||
if(cocc > 0) {//cocc: # of bubbles in this chain
|
||||
for (k = bc->cc.ng.n - cocc; k < bc->cc.ng.n; k++) {
|
||||
kv_resize(double, sc_l, clus_n((*bc), k));
|
||||
@@ -2385,8 +2397,8 @@ void renew_mc_clus_t(mc_clus_t *bc, uint32_t *a, uint32_t a_n)
|
||||
// prt_bub(clus_a((*bc), k), clus_n((*bc), k), "-1-");
|
||||
}
|
||||
for (k = iin; k < bc->cc.nn.n; k++) bc->lock[bc->cc.nn.a[k]] = 1;//reset
|
||||
kv_pushp(uint64_t, bc->cc.ng, &p); *p = (uint64_t)-1;
|
||||
kv_push(uint32_t, bc->cc.nn, ((uint32_t)-1)); ///split
|
||||
kv_pushp(uint64_t, bc->cc.ng, &p); *p = (uint64_t)-1; ///cluster
|
||||
kv_push(uint32_t, bc->cc.nn, ((uint32_t)-1)); ///split, node id
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2396,6 +2408,117 @@ void renew_mc_clus_t(mc_clus_t *bc, uint32_t *a, uint32_t a_n)
|
||||
// a_n, (uint32_t)bc->cc.nn.n, (uint32_t)bc->cc.ng.n, bbn, bub_occ, (uint32_t)bc->bub->b_ug->u.n);
|
||||
}
|
||||
|
||||
|
||||
void renew_mc_clus_t_adv(mc_clus_t *bc, uint32_t *a, uint32_t a_n)
|
||||
{
|
||||
if(!bc) return;
|
||||
// fprintf(stderr, "[M::%s] a_n::%u\n", __func__, a_n);
|
||||
uint32_t k, i, *ba, bn, m, cocc, iin, /**bub_occ = 0,**/ bbn = 0, bid, bk, bl;
|
||||
uint64_t *p; ma_utg_t *u = NULL; asg64_v *asn = bc->asn;
|
||||
kvec_t(double) sc_l; kvec_t(double) sc_r; kvec_t(double) sc_m; kvec_t(uint32_t) tmp;
|
||||
kv_init(sc_l); kv_init(sc_r); kv_init(sc_m); kv_init(tmp);
|
||||
|
||||
bc->cc.ng.n = bc->cc.nn.n = 0;
|
||||
memset(bc->lock, 0, sizeof((*(bc->lock)))*bc->n);///bc->n: number of all nodes
|
||||
for (k = 0; k < a_n; k++) bc->lock[a[k]] = 1;///a_n: number of nodes within the cluster
|
||||
|
||||
kv_resize(uint32_t, bc->cc.nn, a_n);
|
||||
for (i = bid = bk = 0; i < bc->bub->b_ug->u.n; i++) {
|
||||
u = &(bc->bub->b_ug->u.a[i]);
|
||||
if(u->n == 0) continue;
|
||||
// bub_occ += u->n;
|
||||
// fprintf(stderr, "[M::%s] i::%u, u->n::%u\n", __func__, i, (uint32_t)u->n);
|
||||
for (k = cocc = 0, iin = bc->cc.nn.n; k < u->n; k++, bid++) {
|
||||
get_bubbles(bc->bub, u->a[k]>>33, NULL, NULL, &ba, &bn, NULL);
|
||||
kv_pushp(uint64_t, bc->cc.ng, &p); bbn += bn;
|
||||
*p = bc->cc.nn.n;///a bubble
|
||||
for (m = 0; m < bn; m++) {
|
||||
if(!(bc->lock[ba[m]>>1])) continue;///if the node of bubble is not at this cluster
|
||||
kv_push(uint32_t, bc->cc.nn, (ba[m]>>1));
|
||||
bc->lock[ba[m]>>1] = 2;
|
||||
}
|
||||
if(asn) {
|
||||
for (; bk < asn->n && (asn->a[bk]>>32) < bid; bk++);
|
||||
for (; bk < asn->n && (asn->a[bk]>>32) == bid; bk++) {
|
||||
if(!(bc->lock[((uint32_t)asn->a[bk])])) continue;///if the node of bubble is not at this cluster
|
||||
kv_push(uint32_t, bc->cc.nn, ((uint32_t)asn->a[bk]));
|
||||
bc->lock[((uint32_t)asn->a[bk])] = 2;
|
||||
}
|
||||
}
|
||||
|
||||
if(bc->cc.nn.n <= (*p)) {///no node in this bubble
|
||||
bc->cc.ng.n--;
|
||||
continue;
|
||||
}
|
||||
*p <<= 32; *p |= (bc->cc.nn.n-((*p)>>32)); cocc++;
|
||||
}
|
||||
//split the chain if there is multipe bubbles
|
||||
if(cocc > 0) {//cocc: # of bubbles in this chain
|
||||
for (k = bc->cc.ng.n - cocc; k < bc->cc.ng.n; k++) {
|
||||
kv_resize(double, sc_l, clus_n((*bc), k));
|
||||
kv_resize(double, sc_r, clus_n((*bc), k));
|
||||
kv_resize(double, sc_m, clus_n((*bc), k));
|
||||
kv_resize(uint32_t, tmp, clus_n((*bc), k));
|
||||
// prt_bub(clus_a((*bc), k), clus_n((*bc), k), "-0-");
|
||||
reorder_bub(bc->mg->e, clus_a((*bc), k), clus_n((*bc), k), bc->lock, 3, 2, 4, 5,
|
||||
sc_l.a, sc_r.a, sc_m.a, tmp.a);
|
||||
// prt_bub(clus_a((*bc), k), clus_n((*bc), k), "-1-");
|
||||
}
|
||||
for (k = iin; k < bc->cc.nn.n; k++) bc->lock[bc->cc.nn.a[k]] = 1;//reset
|
||||
kv_pushp(uint64_t, bc->cc.ng, &p); *p = (uint64_t)-1; ///cluster
|
||||
kv_push(uint32_t, bc->cc.nn, ((uint32_t)-1)); ///split, node id
|
||||
}
|
||||
}
|
||||
|
||||
if(asn) {
|
||||
for (bl = bk, bk = bk + 1; bk <= asn->n; bk++) {
|
||||
if(bk == asn->n || (asn->a[bk]>>32) != (asn->a[bl]>>32)) {
|
||||
cocc = 0; iin = bc->cc.nn.n;
|
||||
kv_pushp(uint64_t, bc->cc.ng, &p); bbn += bn;
|
||||
*p = bc->cc.nn.n;///a bubble
|
||||
for (m = bl; m < bk; m++) {
|
||||
if(!(bc->lock[((uint32_t)asn->a[m])])) continue;///if the node of bubble is not at this cluster
|
||||
kv_push(uint32_t, bc->cc.nn, ((uint32_t)asn->a[m]));
|
||||
bc->lock[((uint32_t)asn->a[m])] = 2;
|
||||
}
|
||||
if(bc->cc.nn.n <= (*p)) {///no node in this bubble
|
||||
bc->cc.ng.n--;
|
||||
bl = bk;
|
||||
continue;
|
||||
}
|
||||
*p <<= 32; *p |= (bc->cc.nn.n-((*p)>>32)); cocc++;
|
||||
if(cocc > 0) {//cocc: # of bubbles in this chain
|
||||
for (k = bc->cc.ng.n - cocc; k < bc->cc.ng.n; k++) {
|
||||
kv_resize(double, sc_l, clus_n((*bc), k));
|
||||
kv_resize(double, sc_r, clus_n((*bc), k));
|
||||
kv_resize(double, sc_m, clus_n((*bc), k));
|
||||
kv_resize(uint32_t, tmp, clus_n((*bc), k));
|
||||
// prt_bub(clus_a((*bc), k), clus_n((*bc), k), "-0-");
|
||||
reorder_bub(bc->mg->e, clus_a((*bc), k), clus_n((*bc), k), bc->lock, 3, 2, 4, 5,
|
||||
sc_l.a, sc_r.a, sc_m.a, tmp.a);
|
||||
// prt_bub(clus_a((*bc), k), clus_n((*bc), k), "-1-");
|
||||
}
|
||||
for (k = iin; k < bc->cc.nn.n; k++) bc->lock[bc->cc.nn.a[k]] = 1;//reset
|
||||
kv_pushp(uint64_t, bc->cc.ng, &p); *p = (uint64_t)-1; ///cluster
|
||||
kv_push(uint32_t, bc->cc.nn, ((uint32_t)-1)); ///split, node id
|
||||
}
|
||||
|
||||
bl = bk;
|
||||
}
|
||||
}
|
||||
}
|
||||
///need enable it later
|
||||
kv_resize(uint32_t, tmp, bc->cc.nn.n);
|
||||
kv_resize(uint64_t, bc->cc.ng, bc->cc.ng.n+bc->bub->ug->g->n_seq);
|
||||
layout_mc_clus_t(bc->mg->e, bc->cc.nn.a, bc->cc.nn.n, &(bc->sg), tmp.a, bc->cc.ng.a+bc->cc.ng.n, bc->bub->ug, 0.199999, 0.800001, 4);
|
||||
|
||||
for (k = 0; k < a_n; k++) bc->lock[a[k]] = 0;
|
||||
kv_destroy(sc_l); kv_destroy(sc_r); kv_destroy(sc_m); kv_destroy(tmp);
|
||||
// fprintf(stderr, "[M::%s] a_n::%u, bc->cc.nn.n::%u, bc->cc.ng.n::%u, bbn::%u, bub_occ::%u, bc->bub->b_ug->u.n::%u\n", __func__,
|
||||
// a_n, (uint32_t)bc->cc.nn.n, (uint32_t)bc->cc.ng.n, bbn, bub_occ, (uint32_t)bc->bub->b_ug->u.n);
|
||||
}
|
||||
|
||||
|
||||
void clean_clus_flip_aux(clus_flip_aux *z)
|
||||
{
|
||||
z->w = -1; z->occ = z->off = (uint32_t)-1;
|
||||
@@ -2630,7 +2753,8 @@ uint32_t mc_solve_cc_adv(const mc_opt_t *opt, const mc_g_t *mg, mc_svaux_t *b, u
|
||||
b->s_opt[b->cc_node[j]] = b->s[b->cc_node[j]]; ///hap status of each unitig
|
||||
b->z_opt[b->cc_node[j]] = b->z[b->cc_node[j]]; ///z[0]: positive weight; z[1]: positive weight
|
||||
}
|
||||
renew_mc_clus_t(bc, b->cc_node, b->cc_size);
|
||||
// renew_mc_clus_t(bc, b->cc_node, b->cc_size);
|
||||
renew_mc_clus_t_adv(bc, b->cc_node, b->cc_size);
|
||||
// fprintf(stderr, "\ncc_size: %u, cc_off: %u\n", b->cc_size, b->cc_off);
|
||||
// print_sc(opt, mg->e, b, sc_opt, n_iter);
|
||||
sc = mc_optimize_local(opt, mg->e, b, &n_iter);
|
||||
@@ -3231,7 +3355,7 @@ mc_clus_t *init_mc_clus_t(const mc_opt_t *opt, mc_g_t *mg, bubble_type* bub, uin
|
||||
mc_clus_t *p; CALLOC(p, 1);
|
||||
p->bub = bub; p->opt = opt; p->mg = mg; p->n = bub->ug->g->n_seq;
|
||||
CALLOC(p->lock, p->n);
|
||||
if(n_thread > 64) n_thread = 64; p->n_thread = n_thread;
|
||||
if(n_thread > 64) {n_thread = 64;} p->n_thread = n_thread;
|
||||
CALLOC(p->aux, p->n_thread);
|
||||
uint32_t k, ss = (p->n>>3)+(!!(p->n&7));
|
||||
for (k = 0; k < p->n_thread; k++) {
|
||||
@@ -3244,13 +3368,98 @@ mc_clus_t *init_mc_clus_t(const mc_opt_t *opt, mc_g_t *mg, bubble_type* bub, uin
|
||||
void des_mc_clus_t(mc_clus_t *p)
|
||||
{
|
||||
if((!p)) return;
|
||||
uint32_t k; free(p->lock);
|
||||
uint32_t k; free(p->lock); osg_destroy(p->sg.g);
|
||||
if(p->asn) {
|
||||
free(p->asn->a); free(p->asn);
|
||||
}
|
||||
for (k = 0; k < p->n_thread; k++) free(p->aux[k].vis.a);
|
||||
free(p->aux); free(p->cc.ng.a); free(p->cc.nn.a);
|
||||
free(p);
|
||||
}
|
||||
|
||||
void mc_solve_core_adv(const mc_opt_t *opt, mc_g_t *mg, bubble_type* bub)
|
||||
asg64_v* gen_ref_bub(mc_clus_t *bc, kv_u_trans_t *ref)
|
||||
{
|
||||
if(!ref) return NULL;
|
||||
asg64_v *aux; CALLOC(aux, 1); aux->n = aux->m = bc->n; double w, mmw;
|
||||
MALLOC(aux->a, bc->n); memset(aux->a, -1, sizeof((*(aux->a)))*bc->n);
|
||||
uint32_t k, l, i, *ba, bn, m, on, bid; ma_utg_t *u = NULL; uint64_t p, mm, st, j;
|
||||
u_trans_t *o = NULL; asg64_v buf; kv_init(buf);
|
||||
|
||||
for (i = bid = 0; i < bc->bub->b_ug->u.n; i++) {///set nodes within bubbles
|
||||
u = &(bc->bub->b_ug->u.a[i]);
|
||||
for (k = 0; k < u->n; k++, bid++) {
|
||||
get_bubbles(bc->bub, u->a[k]>>33, NULL, NULL, &ba, &bn, NULL);
|
||||
for (m = 0; m < bn; m++) aux->a[ba[m]>>1] = bid;
|
||||
}
|
||||
}
|
||||
// bin = bid;
|
||||
|
||||
for (i = 0; i < bc->bub->ug->g->n_seq; i++) {
|
||||
if(aux->a[i] != ((uint64_t)-1)) continue;
|
||||
o = u_trans_a(*ref, i); on = u_trans_n(*ref, i);
|
||||
if(!on) continue;
|
||||
buf.n = 0; kv_resize(uint64_t, buf, on);
|
||||
for (k = 0; k < on; k++) {
|
||||
p = aux->a[o[k].tn]; p<<= 32; p += k;
|
||||
kv_push(uint64_t, buf, p);
|
||||
}
|
||||
|
||||
radix_sort_mc64(buf.a, buf.a + buf.n);
|
||||
mm = (uint32_t)-1; mmw = -1;
|
||||
for (l = 0, k = 1; k <= buf.n; k++) {
|
||||
if((k == buf.n) || ((buf.a[l]>>32) == ((uint32_t)-1)) || ((buf.a[l]>>32) != (buf.a[k]>>32))) {
|
||||
for (m = l, w = 0; m < k; m++) w += fabs(o[((uint32_t)buf.a[m])].nw);
|
||||
if(mmw < w) {
|
||||
mmw = w; mm = buf.a[l]>>32;
|
||||
}
|
||||
l = k;
|
||||
}
|
||||
}
|
||||
if(mm != ((uint32_t)-1)) aux->a[i] = mm;
|
||||
}
|
||||
|
||||
|
||||
///final cluster
|
||||
for (i = 0; i < bc->bub->ug->g->n_seq; i++) {
|
||||
if(aux->a[i] != ((uint64_t)-1)) continue;
|
||||
buf.n = 0; kv_push(uint64_t, buf, i);
|
||||
while (buf.n > 0) {
|
||||
k = buf.a[--buf.n];
|
||||
if(aux->a[k] != ((uint64_t)-1)) continue;
|
||||
aux->a[k] = bid;///group id
|
||||
o = u_trans_a(*ref, k); on = u_trans_n(*ref, k);
|
||||
for (st = 0, j = 1; j <= on; ++j) {
|
||||
if(j == on || o[j].tn != o[st].tn) {
|
||||
if(aux->a[o[st].tn] == ((uint64_t)-1)) {
|
||||
kv_push(uint64_t, buf, o[st].tn);
|
||||
}
|
||||
st = j;
|
||||
}
|
||||
}
|
||||
}
|
||||
bid++;
|
||||
}
|
||||
|
||||
assert(aux->n == bc->bub->ug->g->n_seq);
|
||||
|
||||
for (i = 0; i < bc->bub->b_ug->u.n; i++) {///no need nodes in bubbles
|
||||
u = &(bc->bub->b_ug->u.a[i]);
|
||||
for (k = 0; k < u->n; k++) {
|
||||
get_bubbles(bc->bub, u->a[k]>>33, NULL, NULL, &ba, &bn, NULL);
|
||||
for (m = 0; m < bn; m++) aux->a[ba[m]>>1] = (uint64_t)-1;
|
||||
}
|
||||
}
|
||||
for (i = aux->n = 0; i < bc->bub->ug->g->n_seq; i++) {
|
||||
if(aux->a[i] == ((uint64_t)-1)) continue;
|
||||
aux->a[i] <<= 32; aux->a[i] |= i;//bub_id|node_id
|
||||
aux->a[aux->n++] = aux->a[i];
|
||||
}
|
||||
radix_sort_mc64(aux->a, aux->a + aux->n);
|
||||
kv_destroy(buf);
|
||||
return aux;
|
||||
}
|
||||
|
||||
void mc_solve_core_adv(const mc_opt_t *opt, mc_g_t *mg, bubble_type* bub, kv_u_trans_t *ref)
|
||||
{
|
||||
double index_time = yak_realtime();
|
||||
uint32_t st, i;
|
||||
@@ -3259,6 +3468,7 @@ void mc_solve_core_adv(const mc_opt_t *opt, mc_g_t *mg, bubble_type* bub)
|
||||
mc_g_cc(mg->e);
|
||||
b = mc_svaux_init(mg, opt->seed);
|
||||
bc = init_mc_clus_t(opt, mg, bub, asm_opt.thread_num, b, 16);
|
||||
if(ref && bc) bc->asn = gen_ref_bub(bc, ref);
|
||||
// bc = gen_mc_clus_t(mg->e, b, bub, ref, asm_opt.thread_num);
|
||||
// if(bub) bp = mc_bp_t_init(mg->e, b, bub, asm_opt.thread_num);
|
||||
/*******************************for debug************************************/
|
||||
@@ -3524,7 +3734,7 @@ void mc_solve(hap_overlaps_list* ovlp, trans_chain* t_ch, kv_u_trans_t *ta, ma_u
|
||||
///debug_mc_g_t(mg);
|
||||
// if(renew_s == 0) write_mc_g_t(&opt, mg, MC_NAME);
|
||||
// mc_solve_core(&opt, mg, bub);
|
||||
mc_solve_core_adv(&opt, mg, bub);
|
||||
mc_solve_core_adv(&opt, mg, bub, ref);
|
||||
|
||||
if((asm_opt.flag & HA_F_PARTITION) && t_ch)
|
||||
{
|
||||
|
||||
Reference in New Issue
Block a user