Compare commits

..
Author SHA1 Message Date
chhylp123 1ac574adc7 Merge pull request #553 from chhylp123/hifiasm_dev_debug
fix bubble issue
2023-11-06 00:10:06 -05:00
chhylp123 7f6d36b2c4 Merge pull request #540 from chhylp123/hifiasm_dev_debug
fix bug for gen_contain_consensus_chain
2023-10-22 13:24:28 -04:00
chhylp123 40e2a48706 Merge pull request #535 from chhylp123/hifiasm_dev_debug
add scaffolding
2023-10-10 12:41:10 -04:00
chhylp123 e974b22ac0 Merge pull request #521 from chhylp123/hifiasm_dev_debug
checkpoint for scaffolding
2023-09-15 13:00:13 -04:00
chhylp123 94a284b430 Merge pull request #503 from chhylp123/hifiasm_dev_debug
disable multiple assertions
2023-08-17 19:59:13 -04:00
chhylp123 30d2ee065e Merge pull request #482 from chhylp123/hifiasm_dev_debug
fix filter_short_ulalignments
2023-07-19 19:12:10 -04:00
chhylp123 d91fc50058 Merge pull request #462 from chhylp123/hifiasm_dev_debug
trio-dual/bug for hic phasing
2023-05-31 14:55:16 -04:00
chhylp123 abff5fae04 Merge pull request #457 from chhylp123/hifiasm_dev_debug
remove ksw2 from makefile
2023-05-12 20:57:45 -04:00
chhylp123 2cc990ed22 Merge pull request #455 from chhylp123/hifiasm_dev_debug
fix memory leak of print_utg
2023-05-12 11:32:03 -04:00
chhylp123 5bddae4ae9 Merge pull request #446 from chhylp123/hifiasm_dev_debug
fixed missing/false duplication for diploid asm
2023-04-17 19:58:31 -04:00
chhylp123 fbfcf72eb3 Merge pull request #445 from chhylp123/hifiasm_dev_debug
Hifiasm dev debug
2023-04-14 16:44:10 -04:00
chhylp123 b49b4a6cc7 Merge pull request #438 from chhylp123/hifiasm_dev_debug
Hifiasm dev debug
2023-04-06 23:18:26 -04:00
chhylp123 00959dfdd9 Merge pull request #430 from chhylp123/hifiasm_dev_debug
0.19.2->0.19.3
2023-03-22 12:26:41 -04:00
chhylp123 00ad7458c3 Merge pull request #429 from chhylp123/hifiasm_dev_debug
avoid misassemblies; better polyploidy graph
2023-03-22 12:24:31 -04:00
chhylp123 b763e1ff76 Merge pull request #422 from chhylp123/hifiasm_dev_debug
disable postjoin for the ul assembly
2023-03-13 15:44:45 -04:00
chhylp123 7bb366688e Merge pull request #421 from chhylp123/hifiasm_dev_debug
Hifiasm dev debug
2023-03-13 11:14:38 -04:00
chhylp123 f69166ee3b Merge pull request #420 from chhylp123/hifiasm_dev_debug
Hifiasm dev debug
2023-03-10 23:44:34 -05:00
chhylp123 fa663a9680 Merge pull request #419 from chhylp123/hifiasm_dev_debug
Hifiasm dev debug
2023-03-10 19:34:50 -05:00
chhylp123 8a00c6c5f4 Merge pull request #414 from chhylp123/hifiasm_dev_debug
better resolution for complex regions
2023-02-27 09:36:46 -05:00
chhylp123 ad4ff5550b Merge pull request #411 from chhylp123/hifiasm_dev_debug
bug fixed for mask calculation
2023-02-24 00:37:36 -05:00
chhylp123 dd5f368fe6 Merge pull request #409 from chhylp123/hifiasm_dev_debug
Hifiasm dev debug
2023-02-21 20:23:31 -05:00
chhylp123 27f77dca29 Merge pull request #401 from chhylp123/hifiasm_dev_debug
0.18.6 -> 0.18.7
2023-02-20 23:42:20 -05:00
chhylp123 026cf97d61 Merge pull request #400 from chhylp123/hifiasm_dev_debug
Hifiasm dev debug
2023-02-16 00:09:39 -05:00
chhylp123 1dd82e7729 Merge pull request #387 from chhylp123/hifiasm_dev_debug
r500
2023-01-23 15:29:16 -05:00
chhylp123 6c91e8b0a7 Merge pull request #386 from chhylp123/hifiasm_dev_debug
Hifiasm dev debug
2023-01-23 15:26:38 -05:00
chhylp123 7280f132b1 Merge pull request #383 from chhylp123/hifiasm_dev_debug
Merge pull request #348 from chhylp123/master
2023-01-17 10:28:29 -05:00
3 changed files with 11 additions and 14 deletions
+7 -10
View File
@@ -290,7 +290,7 @@ int* r_extra_begin, int* r_extra_end, long long* r_y_start, long long* r_y_lengt
///since Window_Len == x_len + (threshold << 1) ///since Window_Len == x_len + (threshold << 1)
if(y_start < 0 || currentIDLen <= y_start || if(y_start < 0 || currentIDLen <= y_start ||
currentIDLen - y_start + 2 * threshold + THRESHOLD_MAX_SIZE < Window_Len)///if ylen is too small currentIDLen - y_start + 2 * threshold + THRESHOLD_MAX_SIZE < Window_Len)
{ {
return 0; return 0;
} }
@@ -1770,9 +1770,9 @@ inline int move_gap_greedy(char* path, int path_i, int path_length, char* x, int
/** /**
* *
GGCG-TGTGCCTGT [Y] GGCG-TGTGCCTGT
* *
GGCAATGTGCCTGT [X] GGCAATGTGCCTGT
* *
00013000000000 00013000000000
**/ **/
@@ -1782,7 +1782,7 @@ inline int move_gap_greedy(char* path, int path_i, int path_length, char* x, int
char oper = path[path_i]; char oper = path[path_i];
if(oper == 3)///there are more x; path[] is in reverse order if(oper == 3)///there are more x
{ {
path_i++; path_i++;
y_i--; y_i--;
@@ -3197,7 +3197,7 @@ inline void recalcate_window_advance(overlap_region_alloc* overlap_list, All_rea
// z->w_list.a[i].x_start, z->w_list.a[i].x_end, w_e); // z->w_list.a[i].x_start, z->w_list.a[i].x_end, w_e);
// } // }
assert(z->w_list.a[i].x_end == w_e); assert(z->w_list.a[i].x_end == w_e);
total_y_start = z->w_list.a[i].y_end + 1 - z->w_list.a[i].extra_begin;///since y_end has extra_begin total_y_start = z->w_list.a[i].y_end + 1 - z->w_list.a[i].extra_begin;
for (k = w_id + 1; k < nw; k++) { for (k = w_id + 1; k < nw; k++) {
if(w_idx[k] != (uint64_t)-1) break; if(w_idx[k] != (uint64_t)-1) break;
w_s = w_e + 1; w_s = w_e + 1;
@@ -9287,7 +9287,7 @@ int insert_snp_ee(haplotype_evdience_alloc* h, haplotype_evdience* a, uint64_t a
uint32_t is_homopolymer = (uint32_t)-1; uint32_t is_homopolymer = (uint32_t)-1;
if(occ_0 == 0 || diff <= 1) return 0; if(occ_0 == 0 || diff <= 1) return 0;
for (i = m = 0; i < 4; i++) { for (i = m = 0; i < 4; i++) {
if(occ_1[i] >= 2){///Improve: it would be better to double check the bases around this snp at different overlapped reads to make sure this snp is real, instead of by chance if(occ_1[i] >= 2){
if(!km) kv_pushp(SnpStats, h->snp_stat, &p); if(!km) kv_pushp(SnpStats, h->snp_stat, &p);
else kv_pushp_km(km, SnpStats, h->snp_stat, &p); else kv_pushp_km(km, SnpStats, h->snp_stat, &p);
p->id = h->snp_stat.n-1; p->id = h->snp_stat.n-1;
@@ -9674,7 +9674,7 @@ void correct_overlap(overlap_region_alloc* overlap_list, All_reads* R_INF,
int flag = 0; int flag = 0;
while(get_Window(&w_inf, &window_start, &window_end) && flag != -2)///Improve: discard alignments with too many diff early while(get_Window(&w_inf, &window_start, &window_end) && flag != -2)
{ {
dumy->length = 0; dumy->length = 0;
dumy->lengthNT = 0; dumy->lengthNT = 0;
@@ -9700,12 +9700,9 @@ void correct_overlap(overlap_region_alloc* overlap_list, All_reads* R_INF,
// recalcate_window(overlap_list, R_INF, g_read, dumy, overlap_read); // recalcate_window(overlap_list, R_INF, g_read, dumy, overlap_read);
// partition_overlaps(overlap_list, R_INF, g_read, dumy, hap, force_repeat); // partition_overlaps(overlap_list, R_INF, g_read, dumy, hap, force_repeat);
///Improve: we should also use the base quality to avoid additional base-pair dp or even for the chaining
//////Improve: we may also consider to do quick conesensus among candidate overlapped reads to make sure if inforamtive sites are real
recalcate_window_advance(overlap_list, R_INF, NULL, g_read, dumy, overlap_read, v_idx, w_inf.window_length, asm_opt.max_ov_diff_ec, asm_opt.max_ov_diff_final); recalcate_window_advance(overlap_list, R_INF, NULL, g_read, dumy, overlap_read, v_idx, w_inf.window_length, asm_opt.max_ov_diff_ec, asm_opt.max_ov_diff_final);
// fprintf(stderr, "[M::%s-beg] occ[0]->%lu, occ[1]->%lu, occ[2]->%lu, occ[3]->%lu\n", __func__, // fprintf(stderr, "[M::%s-beg] occ[0]->%lu, occ[1]->%lu, occ[2]->%lu, occ[3]->%lu\n", __func__,
// ovlp_occ(overlap_list, 0), ovlp_occ(overlap_list, 1), ovlp_occ(overlap_list, 2), ovlp_occ(overlap_list, 3)); // ovlp_occ(overlap_list, 0), ovlp_occ(overlap_list, 1), ovlp_occ(overlap_list, 2), ovlp_occ(overlap_list, 3));
///Improve: only keeps specific number of trans overlaps to save memory
partition_overlaps_advance(overlap_list, R_INF, g_read, overlap_read, dumy, hap, force_repeat); partition_overlaps_advance(overlap_list, R_INF, g_read, overlap_read, dumy, hap, force_repeat);
// fprintf(stderr, "[M::%s-after] occ[0]->%lu, occ[1]->%lu, occ[2]->%lu, occ[3]->%lu\n", __func__, // fprintf(stderr, "[M::%s-after] occ[0]->%lu, occ[1]->%lu, occ[2]->%lu, occ[3]->%lu\n", __func__,
// ovlp_occ(overlap_list, 0), ovlp_occ(overlap_list, 1), ovlp_occ(overlap_list, 2), ovlp_occ(overlap_list, 3)); // ovlp_occ(overlap_list, 0), ovlp_occ(overlap_list, 1), ovlp_occ(overlap_list, 2), ovlp_occ(overlap_list, 3));
+1 -1
View File
@@ -1134,7 +1134,7 @@ void calculate_overlap_region_by_chaining(Candidates_list* candidates, overlap_r
{ {
continue; continue;
} }
///Improve: direct filter out too sparse chain before chaining
chain_DP(candidates->list + sub_region_beg, chain_DP(candidates->list + sub_region_beg,
sub_region_end - sub_region_beg + 1, &(candidates->chainDP), f_cigar, band_width_threshold, sub_region_end - sub_region_beg + 1, &(candidates->chainDP), f_cigar, band_width_threshold,
25, /**Get_READ_LENGTH((*R_INF), (*f_cigar).x_id)**/readLength, 25, /**Get_READ_LENGTH((*R_INF), (*f_cigar).x_id)**/readLength,
+3 -3
View File
@@ -126,7 +126,7 @@ void ha_get_new_candidates(ha_abuf_t *ab, int64_t rid, UC_Read *ucr, overlap_reg
ha_mz1_t *z = &ab->mz.a[i]; ha_mz1_t *z = &ab->mz.a[i];
seed1_t *s = &ab->seed[i]; seed1_t *s = &ab->seed[i];
for (j = 0; j < s->n; ++j) { for (j = 0; j < s->n; ++j) {
const ha_idxpos_t *y = &s->a[j];///Improve: filter out y->rid == rid const ha_idxpos_t *y = &s->a[j];
anchor1_t *an = &ab->a[k++]; anchor1_t *an = &ab->a[k++];
uint8_t rev = z->rev == y->rev? 0 : 1; uint8_t rev = z->rev == y->rev? 0 : 1;
an->other_off = y->pos; an->other_off = y->pos;
@@ -152,7 +152,7 @@ void ha_get_new_candidates(ha_abuf_t *ab, int64_t rid, UC_Read *ucr, overlap_reg
cl->size = ab->m_a; cl->size = ab->m_a;
REALLOC(cl->list, cl->size); REALLOC(cl->list, cl->size);
} }
for (k = 0; k < ab->n_a; ++k) {///Improve: direct use ab->a instead of cl->list for (k = 0; k < ab->n_a; ++k) {
k_mer_hit *p = &cl->list[k]; k_mer_hit *p = &cl->list[k];
p->readID = ab->a[k].srt >> 33; p->readID = ab->a[k].srt >> 33;
p->strand = ab->a[k].srt >> 32 & 1; p->strand = ab->a[k].srt >> 32 & 1;
@@ -184,7 +184,7 @@ void ha_get_new_candidates(ha_abuf_t *ab, int64_t rid, UC_Read *ucr, overlap_reg
} }
#endif #endif
if ((int)overlap_list->length > max_n_chain) {///Improve: use HEAP to directly avoid chaining with too small scores if ((int)overlap_list->length > max_n_chain) {
int32_t w, n[4], s[4]; int32_t w, n[4], s[4];
n[0] = n[1] = n[2] = n[3] = 0, s[0] = s[1] = s[2] = s[3] = 0; n[0] = n[1] = n[2] = n[3] = 0, s[0] = s[1] = s[2] = s[3] = 0;
ks_introsort_or_ss(overlap_list->length, overlap_list->list); ks_introsort_or_ss(overlap_list->length, overlap_list->list);