mirror of
https://github.com/chhylp123/hifiasm.git
synced 2026-09-23 12:38:13 +08:00
Compare commits
2
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
284cd0784a | ||
|
|
ceeb4562af |
@@ -1,21 +0,0 @@
|
||||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
matrix:
|
||||
compiler: [gcc]
|
||||
|
||||
steps:
|
||||
- name: Checkout minimap2
|
||||
uses: actions/checkout@v2
|
||||
|
||||
- name: Compile with ${{ matrix.compiler }}
|
||||
run: make CC=${{ matrix.compiler }}
|
||||
@@ -10,12 +10,9 @@
|
||||
#include "Correct.h"
|
||||
#include "htab.h"
|
||||
#include "kthread.h"
|
||||
#include "rcut.h"
|
||||
|
||||
void ha_get_candidates_interface(ha_abuf_t *ab, int64_t rid, UC_Read *ucr, overlap_region_alloc *overlap_list, overlap_region_alloc *overlap_list_hp, Candidates_list *cl, double bw_thres,
|
||||
int max_n_chain, int keep_whole_chain, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* chain_idx, ma_hit_t_alloc* paf, ma_hit_t_alloc* rev_paf, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct);
|
||||
void ha_get_ug_candidates(ha_abuf_t *ab, int64_t rid, ma_utg_t *u, ma_utg_v *ua, overlap_region_alloc *overlap_list, Candidates_list *cl, double bw_thres, int max_n_chain, int keep_whole_chain, kvec_t_u8_warp* k_flag,
|
||||
kvec_t_u64_warp* chain_idx, void *ha_flt_tab, ha_pt_t *ha_idx, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct, double chain_match_rate);
|
||||
void ha_sort_list_by_anchor(overlap_region_alloc *overlap_list);
|
||||
|
||||
All_reads R_INF;
|
||||
@@ -437,7 +434,6 @@ typedef struct {
|
||||
kvec_t_u64_warp r_buf;
|
||||
kvec_t_u8_warp k_flag;
|
||||
overlap_region tmp_region;
|
||||
ma_utg_v *ua;
|
||||
} ha_ovec_buf_t;
|
||||
|
||||
ha_ovec_buf_t *ha_ovec_init(int is_final, int save_ov)
|
||||
@@ -1605,62 +1601,10 @@ void ha_overlap_final(void)
|
||||
asm_opt.het_cov = het_cov;
|
||||
}
|
||||
|
||||
static void worker_ov_utg(void *data, long i, int tid)
|
||||
{
|
||||
ha_ovec_buf_t *b = ((ha_ovec_buf_t**)data)[tid];
|
||||
if(b->ua->a[i].len == 0) return;
|
||||
|
||||
ha_get_ug_candidates(b->ab, i, &(b->ua->a[i]), b->ua, &b->olist, &b->clist,
|
||||
0.3, asm_opt.polyploidy*5, 0, &(b->k_flag), &b->r_buf, ha_flt_tab, ha_idx,
|
||||
&(b->tmp_region), NULL, /**0.3**/0);
|
||||
|
||||
overlap_region_sort_y_id(b->olist.list, b->olist.length);
|
||||
ma_hit_sort_tn(R_INF.paf[i].buffer, R_INF.paf[i].length);
|
||||
ma_hit_sort_tn(R_INF.reverse_paf[i].buffer, R_INF.reverse_paf[i].length);
|
||||
|
||||
update_overlaps(&b->olist, &(R_INF.paf[i]), &b->self_read, &b->ovlp_read, 1, 1);
|
||||
update_overlaps(&b->olist, &(R_INF.reverse_paf[i]), &b->self_read, &b->ovlp_read, 2, 0);
|
||||
///recover missing exact overlaps
|
||||
update_exact_overlaps(&b->olist, &b->self_read, &b->ovlp_read);
|
||||
|
||||
///Final_phasing(&overlap_list, &cigarline, &g_read, &overlap_read, c2n);
|
||||
push_final_overlaps(&(R_INF.paf[i]), R_INF.reverse_paf, &b->olist, 1);
|
||||
push_final_overlaps(&(R_INF.reverse_paf[i]), R_INF.reverse_paf, &b->olist, 2);
|
||||
}
|
||||
|
||||
|
||||
void ug_idx_build(ma_ug_t *ug, int hap_n)
|
||||
{
|
||||
int flag = asm_opt.flag&HA_F_NO_HPC, i;
|
||||
asm_opt.flag -= flag;
|
||||
ha_flt_tab = ha_ft_ug_gen(&asm_opt, &(ug->u), hap_n);
|
||||
ha_idx = ha_pt_ug_gen(&asm_opt, ha_flt_tab, &(ug->u), hap_n);
|
||||
|
||||
ha_ovec_buf_t **b = NULL;
|
||||
// overlap and correct reads
|
||||
CALLOC(b, asm_opt.thread_num);
|
||||
for (i = 0; i < asm_opt.thread_num; ++i)
|
||||
{
|
||||
b[i] = ha_ovec_init(1, 1);
|
||||
b[i]->ua = &(ug->u);
|
||||
}
|
||||
|
||||
kt_for(asm_opt.thread_num, worker_ov_utg, b, R_INF.total_reads);
|
||||
|
||||
for (i = 0; i < asm_opt.thread_num; ++i)
|
||||
ha_ovec_destroy(b[i]);
|
||||
free(b);
|
||||
|
||||
ha_ft_destroy(ha_flt_tab);
|
||||
ha_pt_destroy(ha_idx);
|
||||
asm_opt.flag += flag;
|
||||
exit(1);
|
||||
}
|
||||
|
||||
int ha_assemble(void)
|
||||
{
|
||||
// debug_mc_g_t(MC_NAME);
|
||||
// debug_mc_gg_t(MC_NAME, 0, 0);
|
||||
extern void ha_extract_print_list(const All_reads *rs, int n_rounds, const char *o);
|
||||
int r, hom_cov = -1, ovlp_loaded = 0;
|
||||
if (asm_opt.load_index_from_disk && load_all_data_from_disk(&R_INF.paf, &R_INF.reverse_paf, asm_opt.output_file_name)) {
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
#ifndef __ASSEMBLY__
|
||||
#define __ASSEMBLY__
|
||||
#include "CommandLines.h"
|
||||
#include "Overlaps.h"
|
||||
|
||||
#define FORWARD 0
|
||||
#define REVERSE_COMPLEMENT (0x8000000000000000)
|
||||
@@ -15,6 +14,5 @@
|
||||
#define RESEED_HP_RATE 0.9
|
||||
|
||||
int ha_assemble(void);
|
||||
void ug_idx_build(ma_ug_t *ug, int hap_n);
|
||||
|
||||
#endif
|
||||
|
||||
+71
-142
@@ -23,22 +23,13 @@ static ko_longopt_t long_options[] = {
|
||||
{ "ex-iter", ko_required_argument, 308 },
|
||||
{ "purge-cov", ko_required_argument, 309 },
|
||||
{ "pri-range", ko_required_argument, 310 },
|
||||
{ "high-het", ko_no_argument, 311 },
|
||||
{ "lowQ", ko_required_argument, 312 },
|
||||
{ "min-hist-cnt", ko_required_argument, 313 },
|
||||
{ "h1", ko_required_argument, 314 },
|
||||
{ "h2", ko_required_argument, 315 },
|
||||
{ "enzyme", ko_required_argument, 316 },
|
||||
{ "b-cov", ko_required_argument, 317 },
|
||||
{ "h-cov", ko_required_argument, 318 },
|
||||
{ "m-rate", ko_required_argument, 319 },
|
||||
{ "primary", ko_no_argument, 320 },
|
||||
{ "t-occ", ko_required_argument, 321 },
|
||||
{ "seed", ko_required_argument, 322 },
|
||||
{ "n-perturb", ko_required_argument, 323 },
|
||||
{ "f-perturb", ko_required_argument, 324 },
|
||||
{ "n-hap", ko_required_argument, 325 },
|
||||
{ "n-weight", ko_required_argument, 326 },
|
||||
{ "l-msjoin", ko_required_argument, 327 },
|
||||
{ 0, 0, 0 }
|
||||
};
|
||||
|
||||
@@ -54,77 +45,55 @@ void Print_H(hifiasm_opt_t* asm_opt)
|
||||
fprintf(stderr, "Usage: hifiasm [options] <in_1.fq> <in_2.fq> <...>\n");
|
||||
fprintf(stderr, "Options:\n");
|
||||
fprintf(stderr, " Input/Output:\n");
|
||||
fprintf(stderr, " -o STR prefix of output files [%s]\n", asm_opt->output_file_name);
|
||||
fprintf(stderr, " -i ignore saved read correction and overlaps\n");
|
||||
fprintf(stderr, " -t INT number of threads [%d]\n", asm_opt->thread_num);
|
||||
fprintf(stderr, " -z INT length of adapters that should be removed [%d]\n", asm_opt->adapterLen);
|
||||
fprintf(stderr, " --version show version number\n");
|
||||
fprintf(stderr, " -o STR prefix of output files [%s]\n", asm_opt->output_file_name);
|
||||
fprintf(stderr, " -i ignore saved read correction and overlaps\n");
|
||||
fprintf(stderr, " -t INT number of threads [%d]\n", asm_opt->thread_num);
|
||||
fprintf(stderr, " -z INT length of adapters that should be removed [%d]\n", asm_opt->adapterLen);
|
||||
fprintf(stderr, " --version show version number\n");
|
||||
fprintf(stderr, " Overlap/Error correction:\n");
|
||||
fprintf(stderr, " -k INT k-mer length (must be <64) [%d]\n", asm_opt->k_mer_length);
|
||||
fprintf(stderr, " -w INT minimizer window size [%d]\n", asm_opt->mz_win);
|
||||
fprintf(stderr, " -f INT number of bits for bloom filter; 0 to disable [%d]\n", asm_opt->bf_shift);
|
||||
fprintf(stderr, " -D FLOAT drop k-mers occurring >FLOAT*coverage times [%.1f]\n", asm_opt->high_factor);
|
||||
fprintf(stderr, " -N INT consider up to max(-D*coverage,-N) overlaps for each oriented read [%d]\n", asm_opt->max_n_chain);
|
||||
fprintf(stderr, " -r INT round of correction [%d]\n", asm_opt->number_of_round);
|
||||
fprintf(stderr, " -k INT k-mer length (must be <64) [%d]\n", asm_opt->k_mer_length);
|
||||
fprintf(stderr, " -w INT minimizer window size [%d]\n", asm_opt->mz_win);
|
||||
fprintf(stderr, " -f INT number of bits for bloom filter; 0 to disable [%d]\n", asm_opt->bf_shift);
|
||||
fprintf(stderr, " -D FLOAT drop k-mers occurring >FLOAT*coverage times [%.1f]\n", asm_opt->high_factor);
|
||||
fprintf(stderr, " -N INT consider up to max(-D*coverage,-N) overlaps for each oriented read [%d]\n", asm_opt->max_n_chain);
|
||||
fprintf(stderr, " -r INT round of correction [%d]\n", asm_opt->number_of_round);
|
||||
fprintf(stderr, " Assembly:\n");
|
||||
fprintf(stderr, " -a INT round of assembly cleaning [%d]\n", asm_opt->clean_round);
|
||||
fprintf(stderr, " -m INT pop bubbles of <INT in size in contig graphs [%lld]\n", asm_opt->large_pop_bubble_size);
|
||||
fprintf(stderr, " -p INT pop bubbles of <INT in size in unitig graphs [%lld]\n", asm_opt->small_pop_bubble_size);
|
||||
fprintf(stderr, " -n INT remove tip unitigs composed of <=INT reads [%d]\n", asm_opt->max_short_tip);
|
||||
fprintf(stderr, " -x FLOAT max overlap drop ratio [%.2g]\n", asm_opt->max_drop_rate);
|
||||
fprintf(stderr, " -y FLOAT min overlap drop ratio [%.2g]\n", asm_opt->min_drop_rate);
|
||||
fprintf(stderr, " -u disable post join contigs step which may improve N50\n");
|
||||
fprintf(stderr, " --lowQ INT\n");
|
||||
fprintf(stderr, " output contig regions with >=INT%% inconsistency in BED format; 0 to disable [%d]\n", asm_opt->bed_inconsist_rate);
|
||||
fprintf(stderr, " --b-cov INT\n");
|
||||
fprintf(stderr, " break contigs at positions with <INT-fold coverage; work with '--m-rate'; 0 to disable [%d]\n", asm_opt->b_low_cov);
|
||||
fprintf(stderr, " --h-cov INT\n");
|
||||
fprintf(stderr, " break contigs at positions with >INT-fold coverage; work with '--m-rate'; -1 to disable [%d]\n", asm_opt->b_high_cov);
|
||||
fprintf(stderr, " --m-rate FLOAT\n");
|
||||
fprintf(stderr, " break contigs at positions with <=FLOAT*coverage exact overlaps;\n");
|
||||
fprintf(stderr, " only work with '--b-cov' or '--h-cov'[%.2f]\n", asm_opt->m_rate);
|
||||
fprintf(stderr, " --primary output a primary assembly and an alternate assembly\n");
|
||||
fprintf(stderr, " -a INT round of assembly cleaning [%d]\n", asm_opt->clean_round);
|
||||
fprintf(stderr, " -m INT pop bubbles of <INT in size in contig graphs [%lld]\n", asm_opt->large_pop_bubble_size);
|
||||
fprintf(stderr, " -p INT pop bubbles of <INT in size in unitig graphs [%lld]\n", asm_opt->small_pop_bubble_size);
|
||||
fprintf(stderr, " -n INT remove tip unitigs composed of <=INT reads [%d]\n", asm_opt->max_short_tip);
|
||||
fprintf(stderr, " -x FLOAT max overlap drop ratio [%.2g]\n", asm_opt->max_drop_rate);
|
||||
fprintf(stderr, " -y FLOAT min overlap drop ratio [%.2g]\n", asm_opt->min_drop_rate);
|
||||
fprintf(stderr, " -u disable post join contigs step which may improve N50\n");
|
||||
fprintf(stderr, " --lowQ INT\n");
|
||||
fprintf(stderr, " output contig regions with >=INT%% inconsistency in BED format; 0 to disable [%d]\n", asm_opt->bed_inconsist_rate);
|
||||
fprintf(stderr, " --b-cov INT\n");
|
||||
fprintf(stderr, " break contigs at breakpoints with coverage drop at <INT-fold coverage [%d]\n", asm_opt->break_cov);
|
||||
|
||||
// fprintf(stderr, " --pri-range INT1[,INT2]\n");
|
||||
// fprintf(stderr, " keep contigs with coverage in this range in p_ctg.gfa; -1 to disable [auto,inf]\n");
|
||||
|
||||
fprintf(stderr, " Trio-partition:\n");
|
||||
fprintf(stderr, " -1 FILE hap1/paternal k-mer dump generated by \"yak count\" []\n");
|
||||
fprintf(stderr, " -2 FILE hap2/maternal k-mer dump generated by \"yak count\" []\n");
|
||||
fprintf(stderr, " -c INT lower bound of the binned k-mer's frequency [%d]\n", asm_opt->min_cnt);
|
||||
fprintf(stderr, " -d INT upper bound of the binned k-mer's frequency [%d]\n", asm_opt->mid_cnt);
|
||||
fprintf(stderr, " -3 FILE list of hap1/paternal read names []\n");
|
||||
fprintf(stderr, " -4 FILE list of hap2/maternal read names []\n");
|
||||
fprintf(stderr, " --t-occ INT\n");
|
||||
fprintf(stderr, " force remove unitigs with >INT unexpected haplotype-specific reads;\n");
|
||||
fprintf(stderr, " ignore graph topology; [%d]\n", asm_opt->trio_flag_occ_thres);
|
||||
fprintf(stderr, " -1 FILE hap1/paternal k-mer dump generated by \"yak count\" []\n");
|
||||
fprintf(stderr, " -2 FILE hap2/maternal k-mer dump generated by \"yak count\" []\n");
|
||||
fprintf(stderr, " -c INT lower bound of the binned k-mer's frequency [%d]\n", asm_opt->min_cnt);
|
||||
fprintf(stderr, " -d INT upper bound of the binned k-mer's frequency [%d]\n", asm_opt->mid_cnt);
|
||||
fprintf(stderr, " -3 FILE list of hap1/paternal read names []\n");
|
||||
fprintf(stderr, " -4 FILE list of hap2/maternal read names []\n");
|
||||
|
||||
fprintf(stderr, " Purge-dups:\n");
|
||||
fprintf(stderr, " -l INT purge level. 0: no purging; 1: light; 2/3: aggressive [0 for trio; 3 for unzip]\n");
|
||||
fprintf(stderr, " -s FLOAT similarity threshold for duplicate haplotigs [%g for -l1/-l2, %g for -l3]\n",
|
||||
asm_opt->purge_simi_rate_l2, asm_opt->purge_simi_rate_l3);
|
||||
fprintf(stderr, " -O INT min number of overlapped reads for duplicate haplotigs [%d]\n",
|
||||
asm_opt->purge_overlap_len);
|
||||
fprintf(stderr, " --purge-cov INT\n");
|
||||
fprintf(stderr, " coverage upper bound of Purge-dups [auto]\n");
|
||||
fprintf(stderr, " --n-hap INT\n");
|
||||
fprintf(stderr, " number of haplotypes [%d]\n", asm_opt->polyploidy);
|
||||
fprintf(stderr, " -l INT purge level. 0: no purging; 1: light; 2: aggressive [0 for trio; 2 for unzip]\n");
|
||||
fprintf(stderr, " -s FLOAT similarity threshold for duplicate haplotigs [%g]\n",
|
||||
asm_opt->purge_simi_rate);
|
||||
fprintf(stderr, " -O INT min number of overlapped reads for duplicate haplotigs [%d]\n",
|
||||
asm_opt->purge_overlap_len);
|
||||
fprintf(stderr, " --purge-cov INT\n");
|
||||
fprintf(stderr, " coverage upper bound of Purge-dups [auto]\n");
|
||||
fprintf(stderr, " --high-het enable this mode for high heterozygosity sample [experimental, not stable]\n");
|
||||
|
||||
// fprintf(stderr, " Hi-C-partition [experimental, not stable]:\n");
|
||||
fprintf(stderr, " Hi-C-partition:\n");
|
||||
fprintf(stderr, " Hi-C-partition [experimental, not stable]:\n");
|
||||
fprintf(stderr, " --h1 FILEs file names of Hi-C R1 [r1_1.fq,r1_2.fq,...]\n");
|
||||
fprintf(stderr, " --h2 FILEs file names of Hi-C R2 [r2_1.fq,r2_2.fq,...]\n");
|
||||
fprintf(stderr, " --seed INT RNG seed [%lu]\n", asm_opt->seed);
|
||||
|
||||
|
||||
fprintf(stderr, " --n-weight INT\n");
|
||||
fprintf(stderr, " rounds of reweighting Hi-C links [%d]\n", asm_opt->n_weight);
|
||||
fprintf(stderr, " --n-perturb INT\n");
|
||||
fprintf(stderr, " rounds of perturbation [%d]\n", asm_opt->n_perturb);
|
||||
fprintf(stderr, " --f-perturb FLOAT\n");
|
||||
fprintf(stderr, " fraction to flip for perturbation [%.3g]\n", asm_opt->f_perturb);
|
||||
fprintf(stderr, " --l-msjoin INT\n");
|
||||
fprintf(stderr, " detect misjoined unitigs of >=INT in size; 0 to disable [%lu]\n", asm_opt->misjoin_len);
|
||||
|
||||
fprintf(stderr, "Example: ./hifiasm -o NA12878.asm -t 32 NA12878.fq.gz\n");
|
||||
fprintf(stderr, "See `man ./hifiasm.1' for detailed description of these command-line options.\n");
|
||||
@@ -133,8 +102,7 @@ void Print_H(hifiasm_opt_t* asm_opt)
|
||||
void init_opt(hifiasm_opt_t* asm_opt)
|
||||
{
|
||||
memset(asm_opt, 0, sizeof(hifiasm_opt_t));
|
||||
///asm_opt->flag = 0;
|
||||
asm_opt->flag = HA_F_PARTITION;
|
||||
asm_opt->flag = 0;
|
||||
asm_opt->coverage = -1;
|
||||
asm_opt->num_reads = 0;
|
||||
asm_opt->read_file_names = NULL;
|
||||
@@ -160,8 +128,7 @@ void init_opt(hifiasm_opt_t* asm_opt)
|
||||
asm_opt->number_of_round = 3;
|
||||
asm_opt->adapterLen = 0;
|
||||
asm_opt->clean_round = 4;
|
||||
///asm_opt->small_pop_bubble_size = 100000;
|
||||
asm_opt->small_pop_bubble_size = 0;
|
||||
asm_opt->small_pop_bubble_size = 100000;
|
||||
asm_opt->large_pop_bubble_size = 10000000;
|
||||
asm_opt->min_drop_rate = 0.2;
|
||||
asm_opt->max_drop_rate = 0.8;
|
||||
@@ -173,32 +140,20 @@ void init_opt(hifiasm_opt_t* asm_opt)
|
||||
asm_opt->max_short_tip = 3;
|
||||
asm_opt->min_cnt = 2;
|
||||
asm_opt->mid_cnt = 5;
|
||||
asm_opt->purge_level_primary = 3;
|
||||
asm_opt->purge_level_primary = 2;
|
||||
asm_opt->purge_level_trio = 0;
|
||||
asm_opt->purge_simi_rate_l2 = 0.75;
|
||||
asm_opt->purge_simi_rate_l3 = 0.55;
|
||||
asm_opt->purge_simi_rate = 0.75;
|
||||
asm_opt->purge_simi_rate_hic = 0.85;
|
||||
asm_opt->purge_overlap_len = 1;
|
||||
///asm_opt->purge_overlap_len_hic = 50;
|
||||
asm_opt->purge_overlap_len_hic = 50;
|
||||
asm_opt->recover_atg_cov_min = -1024;
|
||||
asm_opt->recover_atg_cov_max = INT_MAX;
|
||||
asm_opt->hom_global_coverage = -1;
|
||||
asm_opt->hom_global_coverage_set = 0;
|
||||
asm_opt->bed_inconsist_rate = 70;
|
||||
asm_opt->hic_inconsist_rate = 30;
|
||||
///asm_opt->bub_mer_length = 3;
|
||||
asm_opt->bub_mer_length = 1000000;
|
||||
asm_opt->b_low_cov = 0;
|
||||
asm_opt->b_high_cov = -1;
|
||||
asm_opt->m_rate = 0.75;
|
||||
asm_opt->hap_occ = 1;
|
||||
asm_opt->polyploidy = 2;
|
||||
asm_opt->trio_flag_occ_thres = 60;
|
||||
asm_opt->seed = 11;
|
||||
asm_opt->n_perturb = 10000;
|
||||
asm_opt->f_perturb = 0.1;
|
||||
asm_opt->n_weight = 3;
|
||||
asm_opt->is_alt = 0;
|
||||
asm_opt->misjoin_len = 500000;
|
||||
asm_opt->break_cov = 0;
|
||||
}
|
||||
|
||||
void destory_enzyme(enzyme* f)
|
||||
@@ -395,9 +350,9 @@ int check_option(hifiasm_opt_t* asm_opt)
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->purge_level_primary < 0 || asm_opt->purge_level_primary > 3)
|
||||
if(asm_opt->purge_level_primary < 0 || asm_opt->purge_level_primary > 2)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] the level of purge-dup should be [0, 3] (-l)\n");
|
||||
fprintf(stderr, "[ERROR] the level of purge-dup should be [0, 2] (-l)\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -464,35 +419,25 @@ int check_option(hifiasm_opt_t* asm_opt)
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->b_low_cov < 0)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] must >= 0 (--b-cov)\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->b_high_cov != -1 && asm_opt->b_high_cov < 0)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] must >= 0 (--h-cov)\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->m_rate < 0)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] must >= 0 (--m-rate)\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->b_high_cov != -1 && asm_opt->b_high_cov <= asm_opt->b_low_cov)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] [--h-cov] must >= [--b-cov]\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->purge_simi_thres < 0)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] [-s] must >= 0\n");
|
||||
return 0;
|
||||
}
|
||||
// fprintf(stderr, "input file num: %d\n", asm_opt->num_reads);
|
||||
// fprintf(stderr, "output file: %s\n", asm_opt->output_file_name);
|
||||
// fprintf(stderr, "number of threads: %d\n", asm_opt->thread_num);
|
||||
// fprintf(stderr, "number of rounds for correction: %d\n", asm_opt->number_of_round);
|
||||
// fprintf(stderr, "number of rounds for assembly cleaning: %d\n", asm_opt->clean_round);
|
||||
// fprintf(stderr, "length of removed adapters: %d\n", asm_opt->adapterLen);
|
||||
// fprintf(stderr, "length of k_mer: %d\n", asm_opt->k_mer_length);
|
||||
// fprintf(stderr, "min overlap drop ratio: %.2g\n", asm_opt->min_drop_rate);
|
||||
// fprintf(stderr, "max overlap drop ratio: %.2g\n", asm_opt->max_drop_rate);
|
||||
// fprintf(stderr, "size of popped small bubbles: %lld\n", asm_opt->small_pop_bubble_size);
|
||||
// fprintf(stderr, "size of popped large bubbles: %lld\n", asm_opt->large_pop_bubble_size);
|
||||
// fprintf(stderr, "small removed unitig threshold: %d\n", asm_opt->max_short_tip);
|
||||
// fprintf(stderr, "small removed unitig threshold: %d\n", asm_opt->max_short_tip);
|
||||
// fprintf(stderr, "min_cnt: %d\n", asm_opt->min_cnt);
|
||||
// fprintf(stderr, "mid_cnt: %d\n", asm_opt->mid_cnt);
|
||||
// fprintf(stderr, "purge_level_primary: %d\n", asm_opt->purge_level_primary);
|
||||
// fprintf(stderr, "purge_level_trio: %d\n", asm_opt->purge_level_trio);
|
||||
// fprintf(stderr, "purge_simi_rate: %f\n", asm_opt->purge_simi_rate);
|
||||
// fprintf(stderr, "purge_overlap_len: %d\n", asm_opt->purge_overlap_len);
|
||||
|
||||
return 1;
|
||||
}
|
||||
@@ -630,11 +575,7 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
else if (c == 306) asm_opt->max_ov_diff_final = atof(opt.arg);
|
||||
else if (c == 307) asm_opt->extract_list = opt.arg;
|
||||
else if (c == 308) asm_opt->extract_iter = atoi(opt.arg);
|
||||
else if (c == 309)
|
||||
{
|
||||
asm_opt->hom_global_coverage = atoi(opt.arg);
|
||||
asm_opt->hom_global_coverage_set = 1;
|
||||
}
|
||||
else if (c == 309) asm_opt->hom_global_coverage = atoi(opt.arg);
|
||||
else if (c == 310)
|
||||
{
|
||||
char* s = NULL;
|
||||
@@ -645,28 +586,18 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
asm_opt->recover_atg_cov_min = asm_opt->recover_atg_cov_max = -1;
|
||||
}
|
||||
}
|
||||
///else if (c == 311) asm_opt->flag |= HA_F_HIGH_HET;
|
||||
else if (c == 311) asm_opt->flag |= HA_F_HIGH_HET;
|
||||
else if (c == 312) asm_opt->bed_inconsist_rate = atoi(opt.arg);
|
||||
else if (c == 313) asm_opt->min_hist_kmer_cnt = atoi(opt.arg);
|
||||
else if (c == 314) get_hic_enzymes(opt.arg, &(asm_opt->hic_reads[0]), 0);
|
||||
else if (c == 315) get_hic_enzymes(opt.arg, &(asm_opt->hic_reads[1]), 0);
|
||||
else if (c == 316) get_hic_enzymes(opt.arg, &(asm_opt->hic_enzymes), 1);
|
||||
else if (c == 317) asm_opt->b_low_cov = atoi(opt.arg);
|
||||
else if (c == 318) asm_opt->b_high_cov = atoi(opt.arg);
|
||||
else if (c == 319) asm_opt->m_rate = atof(opt.arg);
|
||||
else if (c == 320) asm_opt->flag -= HA_F_PARTITION, asm_opt->is_alt = 1;
|
||||
else if (c == 321) asm_opt->trio_flag_occ_thres = atoi(opt.arg);
|
||||
else if (c == 322) asm_opt->seed = atol(opt.arg);
|
||||
else if (c == 323) asm_opt->n_perturb = atoi(opt.arg);
|
||||
else if (c == 324) asm_opt->f_perturb = atof(opt.arg);
|
||||
else if (c == 325) asm_opt->polyploidy = atoi(opt.arg);
|
||||
else if (c == 326) asm_opt->n_weight = atoi(opt.arg);
|
||||
else if (c == 327) asm_opt->misjoin_len = atol(opt.arg);
|
||||
else if (c == 317) asm_opt->break_cov = atoi(opt.arg);
|
||||
else if (c == 'l')
|
||||
{ ///0: disable purge_dup; 1: purge containment; 2: purge overlap
|
||||
asm_opt->purge_level_primary = asm_opt->purge_level_trio = atoi(opt.arg);
|
||||
}
|
||||
else if (c == 's') asm_opt->purge_simi_rate_l2 = asm_opt->purge_simi_rate_l3 = atof(opt.arg);
|
||||
else if (c == 's') asm_opt->purge_simi_rate = atof(opt.arg);
|
||||
else if (c == 'O') asm_opt->purge_overlap_len = atoll(opt.arg);
|
||||
else if (c == ':')
|
||||
{
|
||||
@@ -680,9 +611,6 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
}
|
||||
}
|
||||
|
||||
if(asm_opt->purge_level_primary > 2) asm_opt->purge_simi_thres = asm_opt->purge_simi_rate_l3;
|
||||
else asm_opt->purge_simi_thres = asm_opt->purge_simi_rate_l2;
|
||||
|
||||
|
||||
if (argc == opt.ind)
|
||||
{
|
||||
@@ -693,5 +621,6 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
get_queries(argc, argv, &opt, asm_opt);
|
||||
|
||||
|
||||
|
||||
return check_option(asm_opt);
|
||||
}
|
||||
|
||||
+6
-22
@@ -2,9 +2,8 @@
|
||||
#define __COMMAND_LINE_PARSER__
|
||||
|
||||
#include <pthread.h>
|
||||
#include <stdint.h>
|
||||
|
||||
#define HA_VERSION "0.15.4-r343"
|
||||
#define HA_VERSION "0.14-r309"
|
||||
|
||||
#define VERBOSE 0
|
||||
|
||||
@@ -19,7 +18,6 @@
|
||||
#define HA_F_BAN_POST_JOIN 0x100
|
||||
#define HA_F_BAN_ASSEMBLY 0x200
|
||||
#define HA_F_HIGH_HET 0x400
|
||||
#define HA_F_PARTITION 0x800
|
||||
|
||||
#define HA_MIN_OV_DIFF 0.02 // min sequence divergence in an overlap
|
||||
|
||||
@@ -51,9 +49,7 @@ typedef struct {
|
||||
double max_ov_diff_final;
|
||||
int hom_cov;
|
||||
int het_cov;
|
||||
int b_low_cov;
|
||||
int b_high_cov;
|
||||
double m_rate;
|
||||
int break_cov;
|
||||
int max_n_chain; // fall-back max number of chains to consider
|
||||
int min_hist_kmer_cnt;
|
||||
int load_index_from_disk;
|
||||
@@ -72,22 +68,18 @@ typedef struct {
|
||||
int purge_level_primary;
|
||||
int purge_level_trio;
|
||||
int purge_overlap_len;
|
||||
///int purge_overlap_len_hic;
|
||||
int purge_overlap_len_hic;
|
||||
int recover_atg_cov_min;
|
||||
int recover_atg_cov_max;
|
||||
int hom_global_coverage;
|
||||
int hom_global_coverage_set;
|
||||
int bed_inconsist_rate;
|
||||
int hic_inconsist_rate;
|
||||
|
||||
float max_hang_rate;
|
||||
float min_drop_rate;
|
||||
float max_drop_rate;
|
||||
float purge_simi_rate_l2;
|
||||
float purge_simi_rate_l3;
|
||||
float purge_simi_thres;
|
||||
|
||||
///float purge_simi_rate_hic;
|
||||
float purge_simi_rate;
|
||||
float purge_simi_rate_hic;
|
||||
|
||||
long long small_pop_bubble_size;
|
||||
long long large_pop_bubble_size;
|
||||
@@ -96,15 +88,7 @@ typedef struct {
|
||||
long long num_recorrected_bases;
|
||||
long long mem_buf;
|
||||
long long coverage;
|
||||
int hap_occ;
|
||||
int polyploidy;
|
||||
int trio_flag_occ_thres;
|
||||
uint64_t seed;
|
||||
int32_t n_perturb;
|
||||
double f_perturb;
|
||||
int32_t n_weight;
|
||||
uint32_t is_alt;
|
||||
uint64_t misjoin_len;
|
||||
|
||||
} hifiasm_opt_t;
|
||||
|
||||
extern hifiasm_opt_t asm_opt;
|
||||
|
||||
+1
-164
@@ -193,168 +193,6 @@ int append_inexact_overlap_region_alloc(overlap_region_alloc* list, overlap_regi
|
||||
|
||||
|
||||
|
||||
resize_fake_cigar(&(list->list[list->length].f_cigar), (tmp->f_cigar.length + 2));
|
||||
if(add_beg_end == 1)
|
||||
{
|
||||
add_fake_cigar(&(list->list[list->length].f_cigar), list->list[list->length].x_pos_s, 0);
|
||||
}
|
||||
|
||||
long long distance_self_pos = tmp->x_pos_e - tmp->x_pos_s;
|
||||
long long distance_pos = tmp->y_pos_e - tmp->y_pos_s;
|
||||
long long init_distance_gap = distance_pos - distance_self_pos;
|
||||
/****************************may have bugs********************************/
|
||||
///long long pre_distance_gap = init_distance_gap;
|
||||
long long pre_distance_gap = 0xfffffffffffffff;
|
||||
/****************************may have bugs********************************/
|
||||
long long distance_gap;
|
||||
long long i = 0;
|
||||
for (i = tmp->f_cigar.length - 1; i >= 0; i--)
|
||||
{
|
||||
distance_gap = get_fake_gap_shift(&(tmp->f_cigar), i);
|
||||
if(distance_gap != pre_distance_gap)
|
||||
{
|
||||
pre_distance_gap = distance_gap;
|
||||
|
||||
add_fake_cigar(&(list->list[list->length].f_cigar),
|
||||
get_fake_gap_pos(&(tmp->f_cigar), i), init_distance_gap - pre_distance_gap);
|
||||
}
|
||||
}
|
||||
|
||||
if(add_beg_end == 1 && get_fake_gap_pos(&(list->list[list->length].f_cigar),
|
||||
list->list[list->length].f_cigar.length - 1) != (long long)list->list[list->length].x_pos_e)
|
||||
{
|
||||
add_fake_cigar(&(list->list[list->length].f_cigar),
|
||||
list->list[list->length].x_pos_e,
|
||||
get_fake_gap_shift(&(list->list[list->length].f_cigar),
|
||||
list->list[list->length].f_cigar.length - 1));
|
||||
}
|
||||
}
|
||||
|
||||
list->list[list->length].shared_seed = tmp->shared_seed;
|
||||
list->list[list->length].align_length = 0;
|
||||
list->list[list->length].is_match = 0;
|
||||
list->list[list->length].non_homopolymer_errors = 0;
|
||||
list->list[list->length].strong = 0;
|
||||
|
||||
list->length++;
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
int append_utg_inexact_overlap_region_alloc(overlap_region_alloc* list, overlap_region* tmp,
|
||||
ma_utg_v *ua, int add_beg_end)
|
||||
{
|
||||
|
||||
if (list->length + 1 > list->size)
|
||||
{
|
||||
list->size = list->size * 2;
|
||||
list->list = (overlap_region*)realloc(list->list, sizeof(overlap_region)*list->size);
|
||||
/// need to set new space to be 0
|
||||
memset(list->list + (list->size/2), 0, sizeof(overlap_region)*(list->size/2));
|
||||
}
|
||||
|
||||
if (list->length!=0 && list->list[list->length - 1].y_id==tmp->y_id)
|
||||
{
|
||||
///if(list->list[list->length - 1].shared_seed >= tmp->shared_seed)
|
||||
if((list->list[list->length - 1].shared_seed > tmp->shared_seed)
|
||||
||
|
||||
((list->list[list->length - 1].shared_seed == tmp->shared_seed) &&
|
||||
(list->list[list->length - 1].overlapLen <= tmp->overlapLen)))
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
list->length--;
|
||||
}
|
||||
}
|
||||
|
||||
if(tmp->x_pos_s <= tmp->y_pos_s)
|
||||
{
|
||||
tmp->y_pos_s = tmp->y_pos_s - tmp->x_pos_s;
|
||||
tmp->x_pos_s = 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
tmp->x_pos_s = tmp->x_pos_s - tmp->y_pos_s;
|
||||
tmp->y_pos_s = 0;
|
||||
}
|
||||
|
||||
|
||||
long long x_right_length = ua->a[tmp->x_id].len - tmp->x_pos_e - 1;
|
||||
long long y_right_length = ua->a[tmp->y_id].len - tmp->y_pos_e - 1;
|
||||
|
||||
if(x_right_length <= y_right_length)
|
||||
{
|
||||
tmp->x_pos_e = ua->a[tmp->x_id].len - 1;
|
||||
tmp->y_pos_e = tmp->y_pos_e + x_right_length;
|
||||
}
|
||||
else
|
||||
{
|
||||
tmp->x_pos_e = tmp->x_pos_e + y_right_length;
|
||||
tmp->y_pos_e = ua->a[tmp->y_id].len - 1;
|
||||
}
|
||||
|
||||
if (tmp->x_pos_strand == 1)
|
||||
{
|
||||
list->list[list->length].x_id = tmp->x_id;
|
||||
list->list[list->length].x_pos_e = ua->a[tmp->x_id].len - tmp->x_pos_s - 1;
|
||||
list->list[list->length].x_pos_s = ua->a[tmp->x_id].len - tmp->x_pos_e - 1;
|
||||
list->list[list->length].x_pos_strand = 0;
|
||||
|
||||
list->list[list->length].y_id = tmp->y_id;
|
||||
list->list[list->length].y_pos_e = ua->a[tmp->y_id].len - tmp->y_pos_s - 1;
|
||||
list->list[list->length].y_pos_s = ua->a[tmp->y_id].len - tmp->y_pos_e - 1;
|
||||
list->list[list->length].y_pos_strand = 1;
|
||||
|
||||
resize_fake_cigar(&(list->list[list->length].f_cigar), (tmp->f_cigar.length + 2));
|
||||
if(add_beg_end == 1)
|
||||
{
|
||||
add_fake_cigar(&(list->list[list->length].f_cigar), list->list[list->length].x_pos_s, 0);
|
||||
}
|
||||
|
||||
long long distance_gap;
|
||||
/****************************may have bugs********************************/
|
||||
///long long pre_distance_gap = 0;
|
||||
long long pre_distance_gap = 0xfffffffffffffff;
|
||||
/****************************may have bugs********************************/
|
||||
long long i = 0;
|
||||
for (i = 0; i < (long long)tmp->f_cigar.length; i++)
|
||||
{
|
||||
distance_gap = get_fake_gap_shift(&(tmp->f_cigar), i);
|
||||
if(distance_gap != pre_distance_gap)
|
||||
{
|
||||
pre_distance_gap = distance_gap;
|
||||
add_fake_cigar(&(list->list[list->length].f_cigar),
|
||||
ua->a[tmp->x_id].len - get_fake_gap_pos(&(tmp->f_cigar), i) - 1,
|
||||
pre_distance_gap);
|
||||
}
|
||||
}
|
||||
|
||||
if(add_beg_end == 1 && get_fake_gap_pos(&(list->list[list->length].f_cigar),
|
||||
list->list[list->length].f_cigar.length - 1) != (long long)list->list[list->length].x_pos_e)
|
||||
{
|
||||
add_fake_cigar(&(list->list[list->length].f_cigar),
|
||||
list->list[list->length].x_pos_e,
|
||||
get_fake_gap_shift(&(list->list[list->length].f_cigar),
|
||||
list->list[list->length].f_cigar.length - 1));
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
list->list[list->length].x_id = tmp->x_id;
|
||||
list->list[list->length].x_pos_e = tmp->x_pos_e;
|
||||
list->list[list->length].x_pos_s = tmp->x_pos_s;
|
||||
list->list[list->length].x_pos_strand = tmp->x_pos_strand;
|
||||
|
||||
list->list[list->length].y_id = tmp->y_id;
|
||||
list->list[list->length].y_pos_e = tmp->y_pos_e;
|
||||
list->list[list->length].y_pos_s = tmp->y_pos_s;
|
||||
list->list[list->length].y_pos_strand = tmp->y_pos_strand;
|
||||
|
||||
|
||||
|
||||
resize_fake_cigar(&(list->list[list->length].f_cigar), (tmp->f_cigar.length + 2));
|
||||
if(add_beg_end == 1)
|
||||
{
|
||||
@@ -584,7 +422,7 @@ int32_t ha_chain_check(k_mer_hit *a, int32_t n_a, Chain_Data *dp, int32_t min_sc
|
||||
}
|
||||
|
||||
///double band_width_threshold = 0.05;
|
||||
long long chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result,
|
||||
void chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result,
|
||||
double band_width_threshold, int max_skip, int x_readLen, int y_readLen)
|
||||
{
|
||||
long long i, j;
|
||||
@@ -775,7 +613,6 @@ skip_dp:
|
||||
i = dp->pre[i];
|
||||
}
|
||||
}
|
||||
return chainLen;
|
||||
}
|
||||
|
||||
void calculate_overlap_region_by_chaining_back(Candidates_list* candidates, overlap_region_alloc* overlap_list,
|
||||
|
||||
+2
-3
@@ -187,7 +187,6 @@ void init_window_list_alloc(window_list_alloc* x);
|
||||
void clear_window_list_alloc(window_list_alloc* x);
|
||||
void destory_window_list_alloc(window_list_alloc* x);
|
||||
void resize_window_list_alloc(window_list_alloc* x, long long size);
|
||||
long long chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result, double band_width_threshold, int max_skip, int x_readLen, int y_readLen);
|
||||
int append_utg_inexact_overlap_region_alloc(overlap_region_alloc* list, overlap_region* tmp,
|
||||
ma_utg_v *ua, int add_beg_end);
|
||||
void chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result, double band_width_threshold, int max_skip, int x_readLen, int y_readLen);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -6,8 +6,7 @@ CPPFLAGS=
|
||||
INCLUDES=
|
||||
OBJS= CommandLines.o Process_Read.o Assembly.o Hash_Table.o \
|
||||
POA.o Correct.o Levenshtein_distance.o Overlaps.o Trio.o kthread.o Purge_Dups.o \
|
||||
htab.o hist.o sketch.o anchor.o extract.o sys.o ksw2_extz2_sse.o hic.o rcut.o horder.o \
|
||||
tovlp.o
|
||||
htab.o hist.o sketch.o anchor.o extract.o sys.o ksw2_extz2_sse.o hic.o
|
||||
EXE= hifiasm
|
||||
LIBS= -lz -lpthread -lm
|
||||
|
||||
@@ -44,7 +43,7 @@ Assembly.o: kthread.h
|
||||
CommandLines.o: CommandLines.h ketopt.h
|
||||
Correct.o: Correct.h Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h
|
||||
Correct.o: kdq.h CommandLines.h Levenshtein_distance.h POA.h Assembly.h
|
||||
Correct.o: ksw2.h ksort.h
|
||||
Correct.o: ksw2.h
|
||||
Hash_Table.o: Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||
Hash_Table.o: CommandLines.h ksort.h
|
||||
Levenshtein_distance.o: Levenshtein_distance.h
|
||||
@@ -73,6 +72,3 @@ main.o: Levenshtein_distance.h htab.h
|
||||
sketch.o: kvec.h htab.h Process_Read.h Overlaps.h kdq.h CommandLines.h
|
||||
sys.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||
hic.o: hic.h
|
||||
rcut.o: rcut.h
|
||||
horder.o: horder.h
|
||||
tovlp.o: tovlp.h
|
||||
|
||||
+2288
-7168
File diff suppressed because it is too large
Load Diff
+372
-261
@@ -4,7 +4,6 @@
|
||||
#include <stdint.h>
|
||||
#include "kvec.h"
|
||||
#include "kdq.h"
|
||||
#include "ksort.h"
|
||||
|
||||
///#define MIN_OVERLAP_LEN 2000
|
||||
///#define MIN_OVERLAP_LEN 500
|
||||
@@ -27,7 +26,6 @@
|
||||
#define DOUBLE_CHECK_THRES 0.1
|
||||
#define FINAL_DOUBLE_CHECK_THRES 0.2
|
||||
#define CHIMERIC_TRIM_THRES 4
|
||||
#define GAP_LEN 100
|
||||
// #define PRIMARY_LABLE 1
|
||||
// #define ALTER_LABLE 2
|
||||
// #define HAP_LABLE 4
|
||||
@@ -54,8 +52,6 @@
|
||||
#define CUT_DIF_HAP 12
|
||||
|
||||
|
||||
|
||||
|
||||
///query is the read itself
|
||||
typedef struct {
|
||||
uint64_t qns;
|
||||
@@ -102,36 +98,6 @@ int max_hang, int min_ovlp);
|
||||
long long get_specific_overlap(ma_hit_t_alloc* x, uint32_t qn, uint32_t tn);
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint32_t qSpre, qEpre, qScur, qEcur, qn;///[qSp, qEp) && [qSn, qEn]
|
||||
uint32_t tSpre, tEpre, tScur, tEcur, tn;
|
||||
} u_trans_hit_t;
|
||||
|
||||
typedef struct {
|
||||
size_t n, m;
|
||||
u_trans_hit_t* a;
|
||||
} kv_u_trans_hit_t;
|
||||
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint32_t qs, qe, qn;
|
||||
uint32_t ts, te, tn;
|
||||
uint32_t occ;
|
||||
double nw;
|
||||
uint8_t f:6, rev:1, del:1;
|
||||
///uint8_t qo:4, to:4;
|
||||
} u_trans_t;
|
||||
|
||||
typedef struct {
|
||||
size_t n, m;
|
||||
u_trans_t* a;
|
||||
kvec_t(uint64_t) idx;
|
||||
} kv_u_trans_t;
|
||||
|
||||
#define u_trans_a(x, id) ((x).a + ((x).idx.a[(id)]>>32))
|
||||
#define u_trans_n(x, id) ((uint32_t)((x).idx.a[(id)]))
|
||||
|
||||
typedef struct {
|
||||
uint64_t ul;
|
||||
uint32_t v;
|
||||
@@ -402,6 +368,7 @@ typedef struct {
|
||||
kvec_t(uint32_t) e; // visited edges/arcs
|
||||
} buf_t;
|
||||
|
||||
|
||||
typedef struct {
|
||||
kvec_t(uint64_t) Nodes;
|
||||
kvec_t(uint64_t) Edges;
|
||||
@@ -503,7 +470,6 @@ uint64_t* source_index, long long listLen);
|
||||
typedef struct {
|
||||
uint64_t len;
|
||||
uint32_t* index;
|
||||
uint8_t* is_het;
|
||||
} R_to_U;
|
||||
|
||||
void init_R_to_U(R_to_U* x, uint64_t len);
|
||||
@@ -511,7 +477,8 @@ void destory_R_to_U(R_to_U* x);
|
||||
void set_R_to_U(R_to_U* x, uint32_t rID, uint32_t uID, uint32_t is_Unitig, uint8_t* flag);
|
||||
void get_R_to_U(R_to_U* x, uint32_t rID, uint32_t* uID, uint32_t* is_Unitig);
|
||||
void transfor_R_to_U(R_to_U* x);
|
||||
void debug_utg_graph(ma_ug_t *ug, asg_t* read_g, kvec_asg_arc_t_warp* edge, int require_equal_nv, int test_tangle);
|
||||
void debug_utg_graph(ma_ug_t *ug, asg_t* read_g, int require_equal_nv, int test_tangle);
|
||||
int asg_pop_bubble_primary(asg_t *g, int max_dist);
|
||||
long long asg_arc_del_simple_circle_untig(ma_hit_t_alloc* sources, ma_sub_t* coverage_cut, asg_t *g, long long circleLen, int is_drop);
|
||||
|
||||
typedef struct {
|
||||
@@ -526,21 +493,9 @@ typedef struct {
|
||||
uint32_t new_edges_i;
|
||||
} Edge_iter;
|
||||
|
||||
typedef struct {
|
||||
asg_arc_t x;
|
||||
uint64_t Off;
|
||||
uint64_t weight;
|
||||
}asg_arc_t_offset;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(asg_arc_t_offset) a;
|
||||
uint64_t i;
|
||||
}kvec_asg_arc_t_offset;
|
||||
|
||||
|
||||
|
||||
void init_Edge_iter(asg_t* g, uint32_t v, asg_arc_t* new_edges, uint32_t new_edges_n, Edge_iter* x);
|
||||
int get_arc_t(Edge_iter* x, asg_arc_t* get);
|
||||
int asg_pop_bubble_primary_trio(ma_ug_t *ug, int max_dist, uint32_t positive_flag, uint32_t negative_flag);
|
||||
|
||||
|
||||
inline int get_real_length(asg_t *g, uint32_t v, uint32_t* v_s)
|
||||
@@ -585,6 +540,55 @@ inline uint32_t check_tip(asg_t *sg, uint32_t begNode, uint32_t* endNode, buf_t*
|
||||
}
|
||||
}
|
||||
|
||||
inline uint32_t get_unitig_back(asg_t *sg, ma_ug_t *ug, uint32_t begNode, uint32_t* endNode,
|
||||
long long* nodeLen, long long* baseLen, buf_t* b)
|
||||
{
|
||||
ma_utg_v* u = NULL;
|
||||
uint32_t v = begNode, w, k;
|
||||
uint32_t kv;
|
||||
(*nodeLen) = (*baseLen) = 0;
|
||||
(*endNode) = (uint32_t)-1;
|
||||
if(ug!=NULL) u = &(ug->u);
|
||||
|
||||
while (1)
|
||||
{
|
||||
kv = get_real_length(sg, v, NULL);
|
||||
(*endNode) = v;
|
||||
if(u == NULL)
|
||||
{
|
||||
(*nodeLen)++;
|
||||
}
|
||||
else
|
||||
{
|
||||
(*nodeLen) += EvaluateLen((*u), v>>1);
|
||||
}
|
||||
if(b) kv_push(uint32_t, b->b, v);
|
||||
///means reach the end of a unitig
|
||||
if(kv!=1) (*baseLen) += sg->seq[v>>1].len;
|
||||
if(kv==0) return END_TIPS;
|
||||
if(kv>1) return MUL_OUTPUT;
|
||||
///kv must be 1 here
|
||||
kv = get_real_length(sg, v, &w);
|
||||
///means reach the end of a unitig
|
||||
if(get_real_length(sg, w^1, NULL)!=1)
|
||||
{
|
||||
(*baseLen) += sg->seq[v>>1].len;
|
||||
return MUL_INPUT;
|
||||
}
|
||||
|
||||
for (k = 0; k < asg_arc_n(sg, v); k++)
|
||||
{
|
||||
if(asg_arc_a(sg, v)[k].del) continue;
|
||||
///here is just one undeleted edge
|
||||
(*baseLen) += asg_arc_len(asg_arc_a(sg, v)[k]);
|
||||
break;
|
||||
}
|
||||
|
||||
v = w;
|
||||
if(v == begNode) return LOOP;
|
||||
}
|
||||
}
|
||||
|
||||
inline uint32_t get_unitig(asg_t *sg, ma_ug_t *ug, uint32_t begNode, uint32_t* endNode,
|
||||
long long* nodeLen, long long* baseLen, long long* max_stop_nodeLen, long long* max_stop_baseLen,
|
||||
uint32_t stops_threshold, buf_t* b)
|
||||
@@ -710,11 +714,302 @@ uint32_t stops_threshold, buf_t* b)
|
||||
#define UNAVAILABLE (uint32_t)-1
|
||||
#define PLOID 0
|
||||
#define NON_PLOID 1
|
||||
// #define DIFF_HAP_RATE 0.75
|
||||
#define DIFF_HAP_RATE 0.75
|
||||
#define TRIO_DROP_THRES 0.9
|
||||
#define TRIO_DROP_LENGTH_THRES 0.8
|
||||
#define MAX_STOP_RATE 0.6
|
||||
#define TANGLE_MISSED_THRES 0.6
|
||||
///if ug == NULL, nsg should be equal to read_sg
|
||||
inline uint32_t check_different_haps(asg_t *nsg, ma_ug_t *ug, asg_t *read_sg,
|
||||
uint32_t v_0, uint32_t v_1, ma_hit_t_alloc* reverse_sources, buf_t* b_0, buf_t* b_1,
|
||||
R_to_U* ruIndex, uint32_t min_edge_length, uint32_t stops_threshold)
|
||||
{
|
||||
uint32_t vEnd, qn, tn, j, is_Unitig, uId;
|
||||
long long ELen_0, ELen_1, tmp, max_stop_nodeLen, max_stop_baseLen;
|
||||
|
||||
b_0->b.n = b_1->b.n = 0;
|
||||
if(get_unitig(nsg, ug, v_0, &vEnd, &ELen_0, &tmp, &max_stop_nodeLen, &max_stop_baseLen,
|
||||
stops_threshold, b_0) == LOOP)
|
||||
{
|
||||
return UNAVAILABLE;
|
||||
}
|
||||
if(get_unitig(nsg, ug, v_1, &vEnd, &ELen_1, &tmp, &max_stop_nodeLen, &max_stop_baseLen,
|
||||
stops_threshold, b_1) == LOOP)
|
||||
{
|
||||
return UNAVAILABLE;
|
||||
}
|
||||
if(ELen_0<=min_edge_length || ELen_1<=min_edge_length) return UNAVAILABLE;
|
||||
|
||||
rIdContig b_max, b_min;
|
||||
b_max.b_0 = b_min.b_0 = NULL;
|
||||
b_max.offset = b_max.readI = b_max.untigI = 0;
|
||||
b_min.offset = b_min.readI = b_min.untigI = 0;
|
||||
|
||||
if(ELen_0<=ELen_1)
|
||||
{
|
||||
b_min.b_0 = b_0;
|
||||
b_max.b_0 = b_1;
|
||||
}
|
||||
else
|
||||
{
|
||||
b_min.b_0 = b_1;
|
||||
b_max.b_0 = b_0;
|
||||
}
|
||||
|
||||
uint32_t max_count = 0, min_count = 0;
|
||||
ma_utg_t *node_min = NULL, *node_max = NULL;
|
||||
if(ug != NULL)
|
||||
{
|
||||
/*****************************label all unitigs****************************************/
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
node_max = &(ug->u.a[b_max.b_0->b.a[b_max.untigI]>>1]);
|
||||
///each read
|
||||
for (b_max.readI = 0; b_max.readI < node_max->n; b_max.readI++)
|
||||
{
|
||||
qn = (node_max->a[b_max.readI]>>33);
|
||||
set_R_to_U(ruIndex, qn, (b_max.b_0->b.a[b_max.untigI]>>1), 1, &(read_sg->seq[qn].c));
|
||||
}
|
||||
}
|
||||
/*****************************label all unitigs****************************************/
|
||||
|
||||
///each unitig
|
||||
for (b_min.untigI = 0; b_min.untigI < b_min.b_0->b.n; b_min.untigI++)
|
||||
{
|
||||
|
||||
node_min = &(ug->u.a[(b_min.b_0->b.a[b_min.untigI]>>1)]);
|
||||
|
||||
///each read
|
||||
for (b_min.readI = 0; b_min.readI < node_min->n; b_min.readI++)
|
||||
{
|
||||
qn = node_min->a[b_min.readI]>>33;
|
||||
|
||||
/************************BUG: don't forget****************************/
|
||||
if(reverse_sources[qn].length > 0) min_count++;
|
||||
///if(reverse_sources[qn].length >= 0) min_count++;
|
||||
/************************BUG: don't forget****************************/
|
||||
for (j = 0; j < (long long)reverse_sources[qn].length; j++)
|
||||
{
|
||||
tn = Get_tn(reverse_sources[qn].buffer[j]);
|
||||
if(read_sg->seq[tn].del == 1)
|
||||
{
|
||||
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
|
||||
if(tn == (uint32_t)-1 || is_Unitig == 1 || read_sg->seq[tn].del == 1) continue;
|
||||
}
|
||||
|
||||
get_R_to_U(ruIndex, tn, &uId, &is_Unitig);
|
||||
if(uId!=(uint32_t)-1 && is_Unitig == 1)
|
||||
{
|
||||
max_count++;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
/*****************************label all unitigs****************************************/
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
node_max = &(ug->u.a[b_max.b_0->b.a[b_max.untigI]>>1]);
|
||||
///each read
|
||||
for (b_max.readI = 0; b_max.readI < node_max->n; b_max.readI++)
|
||||
{
|
||||
qn = (node_max->a[b_max.readI]>>33);
|
||||
ruIndex->index[qn] = (uint32_t)-1;
|
||||
}
|
||||
}
|
||||
/*****************************label all unitigs****************************************/
|
||||
}
|
||||
else
|
||||
{
|
||||
/*****************************label all reads****************************************/
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
qn = (b_max.b_0->b.a[b_max.untigI]>>1);
|
||||
set_R_to_U(ruIndex, qn, 1, 1, &(read_sg->seq[qn].c));
|
||||
}
|
||||
/*****************************label all reads****************************************/
|
||||
|
||||
///each read
|
||||
for (b_min.untigI = 0; b_min.untigI < b_min.b_0->b.n; b_min.untigI++)
|
||||
{
|
||||
qn = (b_min.b_0->b.a[b_min.untigI]>>1);
|
||||
|
||||
/************************BUG: don't forget****************************/
|
||||
if(reverse_sources[qn].length > 0) min_count++;
|
||||
///if(reverse_sources[qn].length >= 0) min_count++;
|
||||
/************************BUG: don't forget****************************/
|
||||
|
||||
for (j = 0; j < (long long)reverse_sources[qn].length; j++)
|
||||
{
|
||||
tn = Get_tn(reverse_sources[qn].buffer[j]);
|
||||
if(nsg->seq[tn].del == 1)
|
||||
{
|
||||
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
|
||||
if(tn == (uint32_t)-1 || is_Unitig == 1 || nsg->seq[tn].del == 1) continue;
|
||||
}
|
||||
|
||||
|
||||
get_R_to_U(ruIndex, tn, &uId, &is_Unitig);
|
||||
if(uId!=(uint32_t)-1 && is_Unitig == 1)
|
||||
{
|
||||
max_count++;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*****************************label all reads****************************************/
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
qn = (b_max.b_0->b.a[b_max.untigI]>>1);
|
||||
ruIndex->index[qn] = (uint32_t)-1;
|
||||
}
|
||||
/*****************************label all reads****************************************/
|
||||
}
|
||||
|
||||
// if(((v_0==7707) && (v_1==26867))||((v_1==7707) && (v_0==26867)))
|
||||
// {
|
||||
// fprintf(stderr, "******\nv_0>>1: %u, v_0&1: %u, ELen_0: %u\n", v_0>>1, v_0&1, (uint32_t)ELen_0);
|
||||
// fprintf(stderr, "v_1>>1: %u, v_1&1: %u, ELen_1: %u\n", v_1>>1, v_1&1, (uint32_t)ELen_1);
|
||||
// fprintf(stderr, "min_count: %u, max_count: %u, DIFF_HAP_RATE: %f\n\n",
|
||||
// min_count, max_count, DIFF_HAP_RATE);
|
||||
// }
|
||||
|
||||
if(min_count == 0) return UNAVAILABLE;
|
||||
if(max_count > min_count*DIFF_HAP_RATE) return PLOID;
|
||||
return NON_PLOID;
|
||||
}
|
||||
|
||||
inline uint32_t check_different_haps_naive(asg_t *nsg, ma_ug_t *ug, asg_t *read_sg,
|
||||
uint32_t v_0, uint32_t v_1, ma_hit_t_alloc* reverse_sources, buf_t* b_0, buf_t* b_1,
|
||||
R_to_U* ruIndex, uint32_t min_edge_length, uint32_t stops_threshold)
|
||||
{
|
||||
uint32_t vEnd, qn, tn, j, is_Unitig;
|
||||
long long ELen_0, ELen_1, tmp, max_stop_nodeLen, max_stop_baseLen;
|
||||
|
||||
b_0->b.n = b_1->b.n = 0;
|
||||
if(get_unitig(nsg, ug, v_0, &vEnd, &ELen_0, &tmp, &max_stop_nodeLen, &max_stop_baseLen,
|
||||
stops_threshold, b_0) == LOOP)
|
||||
{
|
||||
return UNAVAILABLE;
|
||||
}
|
||||
if(get_unitig(nsg, ug, v_1, &vEnd, &ELen_1, &tmp, &max_stop_nodeLen, &max_stop_baseLen,
|
||||
stops_threshold, b_1) == LOOP)
|
||||
{
|
||||
return UNAVAILABLE;
|
||||
}
|
||||
|
||||
if(ELen_0<=min_edge_length || ELen_1<=min_edge_length) return UNAVAILABLE;
|
||||
|
||||
rIdContig b_max, b_min;
|
||||
b_max.b_0 = b_min.b_0 = NULL;
|
||||
b_max.offset = b_max.readI = b_max.untigI = 0;
|
||||
b_min.offset = b_min.readI = b_min.untigI = 0;
|
||||
|
||||
if(ELen_0<=ELen_1)
|
||||
{
|
||||
b_min.b_0 = b_0;
|
||||
b_max.b_0 = b_1;
|
||||
}
|
||||
else
|
||||
{
|
||||
b_min.b_0 = b_1;
|
||||
b_max.b_0 = b_0;
|
||||
}
|
||||
|
||||
uint32_t max_count = 0, min_count = 0;
|
||||
ma_utg_t *node_min = NULL, *node_max = NULL;
|
||||
|
||||
if(ug != NULL)
|
||||
{
|
||||
///each unitig
|
||||
for (b_min.untigI = 0; b_min.untigI < b_min.b_0->b.n; b_min.untigI++)
|
||||
{
|
||||
|
||||
node_min = &(ug->u.a[(b_min.b_0->b.a[b_min.untigI]>>1)]);
|
||||
|
||||
///each read
|
||||
for (b_min.readI = 0; b_min.readI < node_min->n; b_min.readI++)
|
||||
{
|
||||
qn = node_min->a[b_min.readI]>>33;
|
||||
|
||||
if(reverse_sources[qn].length > 0) min_count++;
|
||||
for (j = 0; j < (long long)reverse_sources[qn].length; j++)
|
||||
{
|
||||
tn = Get_tn(reverse_sources[qn].buffer[j]);
|
||||
if(read_sg->seq[tn].del == 1)
|
||||
{
|
||||
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
|
||||
if(tn == (uint32_t)-1 || is_Unitig == 1 || read_sg->seq[tn].del == 1) continue;
|
||||
}
|
||||
|
||||
///each unitig
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
node_max = &(ug->u.a[b_max.b_0->b.a[b_max.untigI]>>1]);
|
||||
///each read
|
||||
for (b_max.readI = 0; b_max.readI < node_max->n; b_max.readI++)
|
||||
{
|
||||
if(tn == (node_max->a[b_max.readI]>>33))
|
||||
{
|
||||
max_count++;
|
||||
goto end_check_different_haps_ug;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
end_check_different_haps_ug:;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
///each read
|
||||
for (b_min.untigI = 0; b_min.untigI < b_min.b_0->b.n; b_min.untigI++)
|
||||
{
|
||||
qn = (b_min.b_0->b.a[b_min.untigI]>>1);
|
||||
|
||||
if(reverse_sources[qn].length > 0) min_count++;
|
||||
|
||||
for (j = 0; j < (long long)reverse_sources[qn].length; j++)
|
||||
{
|
||||
tn = Get_tn(reverse_sources[qn].buffer[j]);
|
||||
if(nsg->seq[tn].del == 1)
|
||||
{
|
||||
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
|
||||
if(tn == (uint32_t)-1 || is_Unitig == 1 || nsg->seq[tn].del == 1) continue;
|
||||
}
|
||||
|
||||
///each read
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
if((b_max.b_0->b.a[b_max.untigI]>>1) == tn)
|
||||
{
|
||||
max_count++;
|
||||
goto end_check_different_haps_non_ug;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
end_check_different_haps_non_ug:;
|
||||
}
|
||||
}
|
||||
|
||||
// if(((v_0==7707) && (v_1==26867))||((v_1==7707) && (v_0==26867)))
|
||||
// {
|
||||
// fprintf(stderr, "******\nv_0>>1: %u, v_0&1: %u, ELen_0: %u\n", v_0>>1, v_0&1, (uint32_t)ELen_0);
|
||||
// fprintf(stderr, "v_1>>1: %u, v_1&1: %u, ELen_1: %u\n", v_1>>1, v_1&1, (uint32_t)ELen_1);
|
||||
// fprintf(stderr, "min_count: %u, max_count: %u, DIFF_HAP_RATE: %f\n\n",
|
||||
// min_count, max_count, DIFF_HAP_RATE);
|
||||
// }
|
||||
|
||||
if(min_count == 0) return UNAVAILABLE;
|
||||
if(max_count > min_count*DIFF_HAP_RATE) return PLOID;
|
||||
return NON_PLOID;
|
||||
}
|
||||
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint32_t father_occ;
|
||||
@@ -724,55 +1019,36 @@ typedef struct {
|
||||
uint32_t total;
|
||||
} Trio_counter;
|
||||
|
||||
typedef struct {
|
||||
uint32_t p; // the optimal parent vertex
|
||||
uint32_t d; // the shortest distance from the initial vertex
|
||||
uint32_t r:31, s:1; // r: the number of remaining incoming arc; s: state
|
||||
} binfo_s_t;
|
||||
|
||||
typedef struct {
|
||||
///all information for each node
|
||||
binfo_s_t *a;
|
||||
kvec_t(uint32_t) S; // set of vertices without parents, nodes with all incoming edges visited
|
||||
kvec_t(uint32_t) b; // visited vertices
|
||||
kvec_t(uint32_t) e; // visited edges/arcs
|
||||
} buf_s_t;
|
||||
|
||||
typedef struct{
|
||||
buf_s_t *b;
|
||||
uint32_t n_thres, n_reads;
|
||||
asg_t *g;
|
||||
uint32_t check_cross;
|
||||
uint64_t bub_dist;
|
||||
} bub_label_t;
|
||||
|
||||
void resolve_tangles(ma_ug_t *src, asg_t *read_g, ma_hit_t_alloc* reverse_sources, long long minLongUntig,
|
||||
long long maxShortUntig, float l_untig_rate, float max_node_threshold, R_to_U* ruIndex, uint32_t trio_flag,
|
||||
float drop_ratio);
|
||||
void adjust_utg_advance(asg_t *sg, ma_ug_t *ug, ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, bub_label_t* b_mask_t);
|
||||
void adjust_utg_advance(asg_t *sg, ma_ug_t *ug, ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex);
|
||||
void rescue_contained_reads_aggressive(ma_ug_t *i_ug, asg_t *r_g, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, uint32_t chainLenThres, uint32_t is_bubble_check,
|
||||
uint32_t is_primary_check, kvec_asg_arc_t_warp* new_rtg_edges, kvec_t_u32_warp* new_rtg_nodes, bub_label_t* b_mask_t);
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, long long bubble_dist, uint32_t chainLenThres, uint32_t is_bubble_check,
|
||||
uint32_t is_primary_check, kvec_asg_arc_t_warp* new_rtg_edges, kvec_t_u32_warp* new_rtg_nodes);
|
||||
void rescue_missing_overlaps_aggressive(ma_ug_t *i_ug, asg_t *r_g, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, uint32_t is_bubble_check, uint32_t is_primary_check, kvec_asg_arc_t_warp* new_rtg_edges, bub_label_t* b_mask_t);
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, long long bubble_dist, uint32_t is_bubble_check,
|
||||
uint32_t is_primary_check, kvec_asg_arc_t_warp* new_rtg_edges);
|
||||
void all_to_all_deduplicate(ma_ug_t* ug, asg_t* read_g, ma_sub_t* coverage_cut,
|
||||
ma_hit_t_alloc* sources, uint8_t postive_flag, float drop_rate, ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, float double_check_rate, int non_tig_occ);
|
||||
ma_hit_t_alloc* sources, uint8_t postive_flag, float drop_rate, ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, float double_check_rate);
|
||||
void drop_semi_circle(ma_ug_t *ug, asg_t* nsg, asg_t* read_g, ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex);
|
||||
void rescue_wrong_overlaps_to_unitigs(ma_ug_t *i_ug, asg_t *r_g, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
|
||||
ma_sub_t *coverage_cut, R_to_U* ruIndex, int max_hang, int min_ovlp, long long bubble_dist, kvec_asg_arc_t_warp* keep_edges, bub_label_t* b_mask_t);
|
||||
ma_sub_t *coverage_cut, R_to_U* ruIndex, int max_hang, int min_ovlp, long long bubble_dist, kvec_asg_arc_t_warp* keep_edges);
|
||||
void get_unitig_trio_flag(ma_utg_t* nsu, uint32_t flag, uint32_t* require, uint32_t* non_require, uint32_t* ambigious);
|
||||
void rescue_missing_overlaps_backward(ma_ug_t *i_ug, asg_t *r_g, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, uint32_t backward_steps, uint32_t is_bubble_check, uint32_t is_primary_check, bub_label_t* b_mask_t);
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, long long bubble_dist, uint32_t backward_steps,
|
||||
uint32_t is_bubble_check, uint32_t is_primary_check);
|
||||
uint32_t get_edge_from_source(ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, uint32_t query, uint32_t target, asg_arc_t* t);
|
||||
uint64_t asg_bub_pop1_primary_trio(asg_t *g, ma_ug_t *utg, uint32_t v0, int max_dist, buf_t *b,
|
||||
uint32_t positive_flag, uint32_t negative_flag, uint32_t is_pop, uint64_t* path_base_len, uint64_t* path_nodes);
|
||||
int unitig_arc_del_short_diploid_by_length(asg_t *g, float drop_ratio);
|
||||
void asg_bub_backtrack_primary(asg_t *g, uint32_t v0, buf_t *b);
|
||||
|
||||
|
||||
typedef struct{
|
||||
double weight;
|
||||
uint32_t uID;
|
||||
uint32_t uID:31, del:1;
|
||||
uint64_t dis;
|
||||
uint8_t is_cc:7, del:1;
|
||||
uint64_t occ;
|
||||
///uint64_t occ:63, scaff:1;
|
||||
///uint32_t enzyme;
|
||||
@@ -795,62 +1071,10 @@ typedef struct{
|
||||
typedef struct{
|
||||
kvec_t(hc_linkeage) a;
|
||||
kvec_t(uint64_t) enzymes;
|
||||
} hc_links;
|
||||
|
||||
#define N_HET 0
|
||||
#define C_HET 1
|
||||
#define P_HET 2
|
||||
#define S_HET 4
|
||||
|
||||
typedef struct {
|
||||
uint32_t p_x_p, p_y_p, p_x, p_y;
|
||||
uint32_t c_x_p, c_y_p;
|
||||
uint8_t c_rev;
|
||||
} ca_buf_t;
|
||||
|
||||
typedef struct {
|
||||
size_t n, m;
|
||||
ca_buf_t* a;
|
||||
} kv_ca_buf_t;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(uint32_t) uIDs;
|
||||
kvec_t(uint32_t) iDXs;
|
||||
uint32_t chain_num;
|
||||
} sub_tran_t;
|
||||
|
||||
typedef struct{
|
||||
uint32_t* rUidx;
|
||||
uint64_t* rUpos;
|
||||
uint8_t* is_r_het;
|
||||
uint32_t r_num, u_num;
|
||||
kvec_t(bed_in) bed;
|
||||
kvec_t(uint32_t) topo_buf;
|
||||
kvec_t(uint32_t) topo_res;
|
||||
buf_t b_buf_0, b_buf_1;
|
||||
///uint32_t* uLen;
|
||||
kv_u_trans_t k_trans;
|
||||
kv_u_trans_hit_t k_t_b;
|
||||
kv_ca_buf_t c_buf;
|
||||
sub_tran_t st;
|
||||
}trans_chain;
|
||||
|
||||
typedef struct {
|
||||
uint32_t n;
|
||||
uint32_t* cov;
|
||||
uint64_t* pos_idx;
|
||||
ma_hit_t_alloc* reverse_sources;
|
||||
ma_sub_t *coverage_cut;
|
||||
R_to_U* ruIndex;
|
||||
asg_t *read_g;
|
||||
int max_hang;
|
||||
int min_ovlp;
|
||||
kvec_asg_arc_t_offset u_buffer;
|
||||
kvec_t_i32_warp tailIndex;
|
||||
kvec_t_i32_warp prevIndex;
|
||||
///hc_links* link;
|
||||
trans_chain* t_ch;
|
||||
}hap_cov_t;
|
||||
uint32_t* u_idx;
|
||||
uint64_t r_num;
|
||||
} hc_links;
|
||||
|
||||
typedef struct{
|
||||
///kvec_t(hc_edge) a;
|
||||
@@ -858,55 +1082,24 @@ typedef struct{
|
||||
hc_edge *a;
|
||||
}hc_edge_warp;
|
||||
|
||||
typedef struct {
|
||||
uint32_t qs, qe, qn, qus, que;
|
||||
uint32_t ts, te, tn, tus, tue;
|
||||
} utg_thit_t;
|
||||
|
||||
typedef struct {
|
||||
size_t n, m;
|
||||
utg_thit_t* a;
|
||||
} kv_utg_thit_t_t;
|
||||
|
||||
typedef struct {
|
||||
ma_hit_t_alloc* reverse_sources;
|
||||
ma_sub_t *coverage_cut;
|
||||
R_to_U* ruIndex;
|
||||
asg_t *read_g;
|
||||
kvec_asg_arc_t_offset u_buffer;
|
||||
kvec_t_i32_warp tailIndex;
|
||||
kvec_t_i32_warp prevIndex;
|
||||
kv_utg_thit_t_t k_t_b;
|
||||
kv_ca_buf_t c_buf;
|
||||
kv_u_trans_t k_trans;
|
||||
uint64_t *pos_idx, rn;
|
||||
kvec_t(uint32_t) topo_res;
|
||||
ma_ug_t *cug;
|
||||
int max_hang;
|
||||
int min_ovlp;
|
||||
|
||||
ma_utg_v u;
|
||||
kv_u_trans_t t;
|
||||
buf_t b0, b1;
|
||||
} utg_trans_t;
|
||||
|
||||
void init_hc_links(hc_links* link, uint64_t ug_num, trans_chain* t_ch);
|
||||
void init_hc_links(hc_links* link, uint64_t ug_num, uint64_t r_num);
|
||||
void destory_hc_links(hc_links* link);
|
||||
uint64_t get_bub_pop_max_dist(asg_t *g, buf_t *b);
|
||||
uint64_t get_bub_pop_max_dist_advance(asg_t *g, buf_t *b);
|
||||
uint64_t asg_bub_pop1_primary_trio(asg_t *g, ma_ug_t *utg, uint32_t v0, uint64_t max_dist, buf_t *b, uint32_t positive_flag,
|
||||
uint32_t negative_flag, uint32_t is_pop, uint64_t* path_base_len, uint64_t* path_nodes, hap_cov_t *cov, uint32_t is_update_chain, uint32_t keep_d, utg_trans_t *o);
|
||||
|
||||
void clean_primary_untig_graph(ma_ug_t *ug, asg_t *read_g, ma_hit_t_alloc* reverse_sources,
|
||||
long long bubble_dist, long long tipsLen, float tip_drop_ratio, long long stops_threshold,
|
||||
R_to_U* ruIndex, buf_t* b_0, uint8_t* visit, float density, uint32_t miniHapLen,
|
||||
uint32_t miniBiGraph, float chimeric_rate, int is_final_clean, int just_bubble_pop,
|
||||
float drop_ratio, hc_links* link);
|
||||
void adjust_utg_by_primary(ma_ug_t **ug, asg_t* read_g, float drop_rate,
|
||||
ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, ma_sub_t* coverage_cut,
|
||||
long long tipsLen, float tip_drop_ratio, long long stops_threshold,
|
||||
long long bubble_dist, long long tipsLen, float tip_drop_ratio, long long stops_threshold,
|
||||
R_to_U* ruIndex, float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp,
|
||||
kvec_asg_arc_t_warp* new_rtg_edges, hap_cov_t **i_cov, bub_label_t* b_mask_t, uint32_t collect_p_trans, uint32_t collect_p_trans_f);
|
||||
kvec_asg_arc_t_warp* new_rtg_edges, hc_links* link);
|
||||
void collect_reverse_unitigs(buf_t* b_0, buf_t* b_1, hc_links* link, ma_ug_t *ug, asg_t *read_sg);
|
||||
ma_ug_t* copy_untig_graph(ma_ug_t *src);
|
||||
ma_ug_t* output_trio_unitig_graph(asg_t *sg, ma_sub_t* coverage_cut, char* output_file_name,
|
||||
uint8_t flag, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
|
||||
uint8_t flag, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, long long bubble_dist,
|
||||
long long tipsLen, float tip_drop_ratio, long long stops_threshold, R_to_U* ruIndex,
|
||||
float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp, int is_bench, bub_label_t* b_mask_t);
|
||||
float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp, int is_bench);
|
||||
asg_t* copy_read_graph(asg_t *src);
|
||||
ma_ug_t *ma_ug_gen(asg_t *g);
|
||||
void ma_ug_destroy(ma_ug_t *ug);
|
||||
@@ -919,88 +1112,6 @@ inline int inter_interval(int a_s, int a_e, int b_s, int b_e, int* i_s, int* i_e
|
||||
return 1;
|
||||
}
|
||||
|
||||
inline uint32_t get_origin_uid(uint32_t v, trans_chain* t_ch, uint32_t *off, uint32_t *idx)
|
||||
{
|
||||
if(off) (*off) = (t_ch->rUpos[v>>1]>>32);
|
||||
if(idx) (*idx) = (uint32_t)(t_ch->rUpos[v>>1]);
|
||||
if(t_ch->rUpos[v>>1] == (uint64_t)-1) return (uint32_t)-1;
|
||||
return (uint32_t)(((t_ch->rUidx[v>>1]>>1)<<1) + ((t_ch->rUidx[v>>1]^v)&1));
|
||||
}
|
||||
void chain_origin_trans_uid_by_distance(hap_cov_t *cov, asg_t *read_sg,
|
||||
uint32_t *pri_a, uint32_t pri_n, uint32_t pri_beg, uint64_t *i_pri_len,
|
||||
uint32_t *aux_a, uint32_t aux_n, uint32_t aux_beg, uint64_t *i_aux_len,
|
||||
ma_ug_t *ug, uint32_t flag, double overall_score, const char* cmd);
|
||||
int asg_arc_del_trans(asg_t *g, int fuzz);
|
||||
void kt_u_trans_t_idx(kv_u_trans_t *ta, uint32_t n);
|
||||
void kt_u_trans_t_simple_symm(kv_u_trans_t *ta, uint32_t un, uint32_t symm_add);
|
||||
uint32_t get_u_trans_spec(kv_u_trans_t *ta, uint32_t qn, uint32_t tn, u_trans_t **r_a, uint32_t *occ);
|
||||
int ma_ug_seq(ma_ug_t *g, asg_t *read_g, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources,
|
||||
kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp, kvec_asg_arc_t_warp *E, uint32_t is_polish);
|
||||
|
||||
|
||||
typedef struct{
|
||||
ma_sub_t* coverage_cut;
|
||||
ma_hit_t_alloc* sources;
|
||||
ma_hit_t_alloc* reverse_sources;
|
||||
long long tipsLen;
|
||||
float tip_drop_ratio;
|
||||
long long stops_threshold;
|
||||
R_to_U* ruIndex;
|
||||
float chimeric_rate;
|
||||
float drop_ratio;
|
||||
int max_hang;
|
||||
int min_ovlp;
|
||||
int is_bench;
|
||||
long long gap_fuzz;
|
||||
bub_label_t* b_mask_t;
|
||||
}ug_opt_t;
|
||||
|
||||
void adjust_utg_by_trio(ma_ug_t **ug, asg_t* read_g, uint8_t flag, float drop_rate,
|
||||
ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources, ma_sub_t* coverage_cut,
|
||||
long long tipsLen, float tip_drop_ratio, long long stops_threshold,
|
||||
R_to_U* ruIndex, float chimeric_rate, float drop_ratio, int max_hang, int min_ovlp,
|
||||
kvec_asg_arc_t_warp* new_rtg_edges, bub_label_t* b_mask_t);
|
||||
uint32_t cmp_untig_graph(ma_ug_t *src, ma_ug_t *dest);
|
||||
void reduce_hamming_error(asg_t *sg, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
int max_hang, int min_ovlp, long long gap_fuzz);
|
||||
int ma_ug_seq_scaffold(ma_ug_t *g, asg_t *read_g, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources,
|
||||
kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp, kvec_asg_arc_t_warp *E, uint32_t is_polish);
|
||||
void ma_ug_print(const ma_ug_t *ug, asg_t* read_g, const ma_sub_t *coverage_cut,
|
||||
ma_hit_t_alloc* sources, R_to_U* ruIndex, const char* prefix, FILE *fp);
|
||||
void ma_ug_print_simple(const ma_ug_t *ug, asg_t* read_g, const ma_sub_t *coverage_cut,
|
||||
ma_hit_t_alloc* sources, R_to_U* ruIndex, const char* prefix, FILE *fp);
|
||||
trans_chain* init_trans_chain(ma_ug_t *ug, uint64_t r_num);
|
||||
void destory_trans_chain(trans_chain **x);
|
||||
|
||||
typedef struct {///[cBeg, cEnd)
|
||||
uint32_t u_i, r_i, len, s_pos_cur, s_pre_v, s_pre_w, p_v, p_idx, p_uId, cBeg, cEnd;
|
||||
///buf_t* x;
|
||||
uint32_t *a, an;
|
||||
ma_ug_t *ug;
|
||||
asg_t *read_sg;
|
||||
trans_chain* t_ch;
|
||||
} u_trans_hit_idx;
|
||||
void reset_u_trans_hit_idx(u_trans_hit_idx *t, uint32_t* i_x_a, uint32_t i_x_n, ma_ug_t *i_ug,
|
||||
asg_t *i_read_sg, trans_chain* i_t_ch, uint32_t i_cBeg, uint32_t i_cEnd);
|
||||
uint32_t get_u_trans_hit(u_trans_hit_idx *t, u_trans_hit_t *hit);
|
||||
inline uint32_t get_offset_adjust(uint32_t offset, uint32_t offsetLen, uint32_t targetLen)
|
||||
{
|
||||
return ((double)(offset)/(double)(offsetLen))*targetLen;
|
||||
}
|
||||
|
||||
uint32_t set_utg_offset(uint32_t *a, uint32_t a_n, ma_ug_t *ug, asg_t *read_sg, uint64_t* pos_idx, uint32_t is_clear,
|
||||
uint32_t only_len);
|
||||
uint64_t get_utg_cov(ma_ug_t *ug, uint32_t uID, asg_t* read_g,
|
||||
const ma_sub_t* coverage_cut, ma_hit_t_alloc* sources, R_to_U* ruIndex, uint8_t* r_flag);
|
||||
trans_chain* load_hc_trans(const char *fn);
|
||||
char *get_outfile_name(char* output_file_name);
|
||||
void reset_u_trans_hit_idx(u_trans_hit_idx *t, uint32_t* i_x_a, uint32_t i_x_n, ma_ug_t *i_ug,
|
||||
asg_t *i_read_sg, trans_chain* i_t_ch, uint32_t i_cBeg, uint32_t i_cEnd);
|
||||
void extract_sub_overlaps(uint32_t i_tScur, uint32_t i_tEcur, uint32_t i_tSpre, uint32_t i_tEpre,
|
||||
uint32_t tn, kv_u_trans_hit_t* ktb, uint32_t bn);
|
||||
void clean_u_trans_t_idx(kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g);
|
||||
|
||||
|
||||
#define JUNK_COV 5
|
||||
#define DISCARD_RATE 0.8
|
||||
|
||||
|
||||
@@ -101,9 +101,6 @@ typedef struct
|
||||
#define MIX_TRIO 3
|
||||
#define NON_TRIO 4
|
||||
#define DROP 5
|
||||
#define SET_TRIO 8
|
||||
#define CHAIN_MATCH 1
|
||||
#define CHAIN_UNMATCH 0.334
|
||||
|
||||
typedef struct
|
||||
{
|
||||
|
||||
+1168
-2162
File diff suppressed because it is too large
Load Diff
+2
-71
@@ -11,83 +11,14 @@
|
||||
#define HET_PEAK_RATE (HOM_PEAK_RATE*2)
|
||||
#define ALTER_COV_THRES 0.9
|
||||
#define REAL_ALTER_THRES 0.1
|
||||
#define CHAIN_FILTER_RATE 0.7
|
||||
|
||||
#define SELF_EXIST 0
|
||||
#define REVE_EXIST 1
|
||||
#define DELETE 2
|
||||
#define MIXED 3
|
||||
#define FLIP 4
|
||||
|
||||
#define X2Y 0
|
||||
#define Y2X 1
|
||||
#define XCY 2
|
||||
#define YCX 3
|
||||
|
||||
#define Cal_Off(OFF) ((long long)((uint32_t)((OFF)>>32)) - (long long)((uint32_t)((OFF))))
|
||||
#define Get_xOff(OFF) ((long long)((uint32_t)((OFF)>>32)))
|
||||
#define Get_yOff(OFF) ((long long)((uint32_t)((OFF))))
|
||||
#define Get_match(x) ((x).weight)
|
||||
#define Get_total(x) ((x).index_beg)
|
||||
#define Get_type(x) ((x).index_end)
|
||||
#define Get_x_beg(x) ((x).x_beg_pos)
|
||||
#define Get_x_end(x) ((x).x_end_pos)
|
||||
#define Get_y_beg(x) ((x).y_beg_pos)
|
||||
#define Get_y_end(x) ((x).y_end_pos)
|
||||
#define Get_rev(x) ((x).rev)
|
||||
|
||||
typedef struct {
|
||||
uint8_t rev;
|
||||
uint8_t type;
|
||||
uint8_t status;
|
||||
uint32_t x_beg_pos;
|
||||
uint32_t x_end_pos;
|
||||
uint32_t y_beg_pos;
|
||||
uint32_t y_end_pos;
|
||||
uint32_t x_beg_id;
|
||||
uint32_t x_end_id;
|
||||
uint32_t y_beg_id;
|
||||
uint32_t y_end_id;
|
||||
uint32_t xUid;
|
||||
uint32_t yUid;
|
||||
uint32_t weight;
|
||||
long long score;
|
||||
float s;
|
||||
}hap_overlaps;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(hap_overlaps) a;
|
||||
}kvec_hap_overlaps;
|
||||
|
||||
typedef struct {
|
||||
kvec_hap_overlaps* x;
|
||||
uint32_t num;
|
||||
}hap_overlaps_list;
|
||||
|
||||
void purge_dups(ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut, ma_hit_t_alloc* sources,
|
||||
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, kvec_asg_arc_t_warp* edge, float density,
|
||||
uint32_t purege_minLen, int max_hang, int min_ovlp, float drop_ratio, uint32_t just_contain,
|
||||
uint32_t just_coverage, hap_cov_t *cov, uint32_t collect_p_trans, uint32_t collect_p_trans_f);
|
||||
uint32_t purege_minLen, int max_hang, int min_ovlp, long long bubble_dist, float drop_ratio,
|
||||
uint32_t just_contain, uint32_t just_coverage, hc_links* link);
|
||||
void fill_unitig(uint64_t* buffer, uint32_t bufferLen, asg_t* read_g, kvec_asg_arc_t_warp* edge,
|
||||
uint32_t is_circle, uint64_t* rLen);
|
||||
void get_contig_length(ma_ug_t *ug, asg_t *g, uint64_t* primaryLen, uint64_t* alterLen);
|
||||
void enable_debug_mode(uint32_t mode);
|
||||
hap_cov_t* init_hap_cov_t(ma_ug_t *ug, asg_t* read_g, ma_hit_t_alloc* sources, R_to_U* ruIndex, ma_hit_t_alloc* reverse_sources,
|
||||
ma_sub_t *coverage_cut, int max_hang, int min_ovlp, uint32_t is_collect_trans);
|
||||
void destory_hap_cov_t(hap_cov_t **x);
|
||||
void chain_trans_ovlp(hap_cov_t *cov, utg_trans_t *o, ma_ug_t *ug, asg_t *read_sg, buf_t* xReads, uint32_t targetBaseLen, uint32_t* xEnd);
|
||||
int get_specific_hap_overlap(kvec_hap_overlaps* x, uint32_t qn, uint32_t tn);
|
||||
void set_reverse_hap_overlap(hap_overlaps* dest, hap_overlaps* source, uint32_t* types);
|
||||
void print_hap_paf(ma_ug_t *ug, hap_overlaps* ovlp);
|
||||
uint64_t get_xy_pos_by_pos(asg_t *read_g, asg_arc_t* t, uint32_t v_in_unitig, uint32_t w_in_unitig,
|
||||
uint32_t v_in_pos, uint32_t w_in_pos, uint32_t xUnitigLen, uint32_t yUnitigLen, uint8_t* rev);
|
||||
void quick_LIS(asg_arc_t_offset* x, uint32_t n, kvec_t_i32_warp* tailIndex, kvec_t_i32_warp* prevIndex);
|
||||
uint32_t classify_hap_overlap(long long xBeg, long long xEnd, long long xLen,
|
||||
long long yBeg, long long yEnd, long long yLen, long long* r_xBeg, long long* r_xEnd,
|
||||
long long* r_yBeg, long long* r_yEnd);
|
||||
int cmp_hap_alignment_chaining(const void * a, const void * b);
|
||||
uint32_t classify_hap_overlap(long long xBeg, long long xEnd, long long xLen,
|
||||
long long yBeg, long long yEnd, long long yLen, long long* r_xBeg, long long* r_xEnd,
|
||||
long long* r_yBeg, long long* r_yEnd);
|
||||
|
||||
#endif
|
||||
@@ -1,4 +1,4 @@
|
||||
## <a name="started"></a>Getting Started
|
||||
## Getting Started
|
||||
|
||||
```sh
|
||||
# Install hifiasm (requiring g++ and zlib)
|
||||
@@ -12,47 +12,30 @@ awk '/^S/{print ">"$2;print $3}' test.p_ctg.gfa > test.p_ctg.fa # get primary c
|
||||
|
||||
# Assemble inbred/homozygous genomes (-l0 disables duplication purging)
|
||||
hifiasm -o CHM13.asm -t32 -l0 CHM13-HiFi.fa.gz 2> CHM13.asm.log
|
||||
# Assemble heterozygous genomes with built-in duplication purging
|
||||
# Assemble heterozygous with built-in duplication purging
|
||||
hifiasm -o HG002.asm -t32 HG002-file1.fq.gz HG002-file2.fq.gz
|
||||
|
||||
# Hi-C phasing with paired-end short reads in two FASTQ files
|
||||
hifiasm -o HG002.asm --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
||||
|
||||
# Trio binning assembly (requiring https://github.com/lh3/yak)
|
||||
yak count -b37 -t16 -o pat.yak <(cat pat_1.fq.gz pat_2.fq.gz) <(cat pat_1.fq.gz pat_2.fq.gz)
|
||||
yak count -b37 -t16 -o mat.yak <(cat mat_1.fq.gz mat_2.fq.gz) <(cat mat_1.fq.gz mat_2.fq.gz)
|
||||
hifiasm -o HG002.asm -t32 -1 pat.yak -2 mat.yak HG002-HiFi.fa.gz
|
||||
```
|
||||
|
||||
## Table of Contents
|
||||
## Introduction
|
||||
|
||||
- [Getting Started](#started)
|
||||
- [Introduction](#intro)
|
||||
- [Why Hifiasm?](#why)
|
||||
- [Usage](#use)
|
||||
- [Assembling HiFi reads without additional data types](#hifionly)
|
||||
- [Hi-C integration](#hic)
|
||||
- [Trio binning](#trio)
|
||||
- [Output files](#output)
|
||||
- [Results](#results)
|
||||
- [Getting Help](#help)
|
||||
- [Limitations](#limit)
|
||||
- [Citing Hifiasm](#cite)
|
||||
Hifiasm is a fast haplotype-resolved de novo assembler for PacBio Hifi reads.
|
||||
It can assemble a human genome in several hours and works with the California
|
||||
redwood genome, one of the most complex genomes sequenced so far. Hifiasm can
|
||||
produce primary/alternate assemblies of quality competitive with the best
|
||||
assemblers. It also introduces a new graph binning algorithm and achieves
|
||||
the best haplotype-resolved assembly given trio data.
|
||||
|
||||
## <a name="intro"></a>Introduction
|
||||
|
||||
Hifiasm is a fast haplotype-resolved de novo assembler for PacBio HiFi reads.
|
||||
It can assemble a human genome in several hours and assemble a ~30Gb California
|
||||
redwood genome in a few days. Hifiasm emits partially phased assemblies of
|
||||
quality competitive with the best assemblers. Given parental short reads or
|
||||
Hi-C data, it produces arguably the best haplotype-resolved assemblies so far.
|
||||
|
||||
## <a name="why"></a>Why Hifiasm?
|
||||
## Why Hifiasm?
|
||||
|
||||
* Hifiasm delivers high-quality assemblies. It tends to generate longer contigs
|
||||
and resolve more segmental duplications than other assemblers.
|
||||
|
||||
* Given Hi-C reads or short reads from the parents, hifiasm can produce overall the best
|
||||
* Given sequence reads from the parents, hifiasm can produce overall the best
|
||||
haplotype-resolved assembly so far. It is the assembler of choice by the
|
||||
[Human Pangenome Project][hpp] for the first batch of samples.
|
||||
|
||||
@@ -64,15 +47,13 @@ Hi-C data, it produces arguably the best haplotype-resolved assemblies so far.
|
||||
* Hifiasm is fast. It can assemble a human genome in half a day and assemble a
|
||||
~30Gb redwood genome in three days. No genome is too large for hifiasm.
|
||||
|
||||
* Hifiasm is trivial to install and easy to use. It does not required Python,
|
||||
R or C++11 compilers, and can be compiled into a single executable. The
|
||||
* Hifiasm is trivial to install and easy to use. It does not required python,
|
||||
R or C++11 compilers and can be compiled into a single executable. The
|
||||
default setting works well with a variety of genomes.
|
||||
|
||||
[hpp]: https://humanpangenome.org
|
||||
|
||||
## <a name="use"></a>Usage
|
||||
|
||||
### <a name="hifionly"></a>Assembling HiFi reads without additional data types
|
||||
## Usage
|
||||
|
||||
A typical hifiasm command line looks like:
|
||||
```sh
|
||||
@@ -80,21 +61,11 @@ hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
||||
```
|
||||
where `NA12878.fq.gz` provides the input reads, `-t` sets the number of CPUs in
|
||||
use and `-o` specifies the prefix of output files. For this example, the
|
||||
primary contigs are written to `NA12878.asm.bp.p_ctg.gfa` and alternate contigs to
|
||||
`NA12878.asm.bp.a_ctg.gfa`. Since v0.15, hifiasm also produces two sets of
|
||||
partially phased contigs at `NA12878.asm.bp.hap?.p_ctg.gfa`. This pair of files
|
||||
can be thought to represent the two haplotypes in a diploid genome, though with
|
||||
occasional switch errors. The frequency of switches is determined by the
|
||||
heterozygosity of the input sample.
|
||||
|
||||
At the first run, hifiasm saves corrected reads and
|
||||
primary contigs are written to `NA12878.asm.p_ctg.gfa` and alternate contigs to
|
||||
`NA12878.asm.a_ctg.gfa`. At the first run, hifiasm saves corrected reads and
|
||||
overlaps to disk as `NA12878.asm.*.bin`. It reuses the saved results to avoid
|
||||
the time-consuming all-vs-all overlap calculation next time. You may specify
|
||||
`-i` to ignore precomputed overlaps and redo overlapping from raw reads.
|
||||
You can also dump error corrected reads in FASTA and read overlaps in PAF with
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t 32 --write-paf --write-ec /dev/null
|
||||
```
|
||||
|
||||
Hifiasm purges haplotig duplications by default. For inbred or homozygous
|
||||
genomes, you may disable purging with option `-l0`. Old HiFi reads may contain
|
||||
@@ -104,27 +75,7 @@ bloom filter which takes 16GB memory at the beginning. For genomes much larger
|
||||
than human, applying `-f38` or even `-f39` is preferred to save memory on k-mer
|
||||
counting.
|
||||
|
||||
### <a name="hic"></a>Hi-C integration
|
||||
|
||||
Hifiasm can generate a pair of haplotype-resolved assemblies with paired-end
|
||||
Hi-C reads:
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t32 --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
|
||||
```
|
||||
In this mode, each contig is supposed to be a haplotig, which by definition
|
||||
comes from one parental haplotype only. Hifiasm often puts all contigs from the
|
||||
same parental chromosome in one assembly. It has cleanly separated chrX and
|
||||
chrY for a human male dataset. Nonetheless, phasing across centromeres is
|
||||
challenging. Users should not expect hifiasm to phase entire chromosomes at the
|
||||
moment. Also, contigs from different parental chromosomes are randomly mixed as
|
||||
it is just not possible to phase across chromosomes with Hi-C.
|
||||
|
||||
Hifiasm does not perform scaffolding for now. You need to run a standalone
|
||||
scaffolder such as SALSA or 3D-DNA to scaffold phased haplotigs.
|
||||
|
||||
### <a name="trio"></a>Trio binning
|
||||
|
||||
When parental short reads are available, hifiasm can also generate a pair of
|
||||
When parental short reads are available, hifiasm can generate a pair of
|
||||
haplotype-resolved assemblies with trio binning. To perform such assembly, you
|
||||
need to count k-mers first with [yak][yak] first and then do assembly:
|
||||
```sh
|
||||
@@ -134,15 +85,19 @@ hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak NA12878.fq.gz
|
||||
```
|
||||
Here `NA12878.asm.hap1.p_ctg.gfa` and `NA12878.asm.hap2.p_ctg.gfa` give the two
|
||||
haplotype assemblies. In the binning mode, hifiasm does not purge haplotig
|
||||
duplicates by default. Because hifiasm reuses saved overlaps, you can
|
||||
duplications by default. Because hifiasm reuses saved overlaps, you can
|
||||
generate both primary/alternate assemblies and trio binning assemblies with
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz 2> NA12878.asm.pri.log
|
||||
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak /dev/null 2> NA12878.asm.trio.log
|
||||
```
|
||||
The second command line will run much faster than the first.
|
||||
The second command line will run much faster than the first. You can also dump
|
||||
error corrected in FASTA and/or overlaps in PAF with
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t 32 --write-paf --write-ec /dev/null
|
||||
```
|
||||
|
||||
### <a name="output"></a>Output files
|
||||
## Output files
|
||||
|
||||
For non-trio assembly, hifiasm generates the following files:
|
||||
|
||||
@@ -171,9 +126,9 @@ For trio assembly, hifiasm generates the following files:
|
||||
Hifiasm writes error corrected reads to the *prefix*.ec.bin binary file and
|
||||
writes overlaps to *prefix*.ovlp.source.bin and *prefix*.ovlp.reverse.bin.
|
||||
|
||||
## <a name="results"></a>Results
|
||||
## Results
|
||||
|
||||
The following table shows the statistics of several hifiasm primary assemblies assembled with v0.12:
|
||||
The following table shows the statistics of several hifiasm primary assemblies:
|
||||
|
||||
|<sub>Dataset<sub>|<sub>Size<sub>|<sub>Cov.<sub>|<sub>Asm options<sub>|<sub>CPU time<sub>|<sub>Wall time<sub>|<sub>RAM<sub>|<sub> N50<sub>|
|
||||
|:---------------|-----:|-----:|:---------------------|-------:|--------:|----:|----------------:|
|
||||
@@ -200,10 +155,7 @@ redwood genome in a few days on a single machine. For trio binning assembly:
|
||||
|:---------------|-----:|-------:|--------:|----:|----------------:|
|
||||
|<sub>[HG00733][HG00733-data], [\[father\]][HG00731-data], [\[mother\]][HG00732-data]</sub>|<sub>×33</sub>|<sub>269.1h</sub>|<sub>6.9h</sub>|<sub>135G</sub>|<sub>35.1Mb (paternal), 34.9Mb (maternal)</sub>|
|
||||
|<sub>[HG002][NA24385-data], [\[father\]][NA24149-data], [\[mother\]][NA24143-data]</sup>|<sub>×36</sub>|<sub>305.4h</sub>|<sub>7.7h</sub>|<sub>137G</sub>|<sub>41.0Mb (paternal), 40.8Mb (maternal)</sub>|
|
||||
|
||||
<!--
|
||||
|<sub>[NA12878][NA12878-data], [\[father\]][NA12891-data], [\[mother\]][NA12892-data]</sub>|<sub>×30</sub>|<sub>180.8h</sub>|<sub>4.9h</sub>|<sub>123G</sub>|<sub>27.7Mb (paternal), 27.0Mb (maternal)</sub>|
|
||||
-->
|
||||
|
||||
[HG00733-data]: https://www.ebi.ac.uk/ena/data/view/ERX3831682
|
||||
[HG00731-data]: https://www.ebi.ac.uk/ena/data/view/ERR3241754
|
||||
@@ -215,32 +167,29 @@ redwood genome in a few days on a single machine. For trio binning assembly:
|
||||
[NA12891-data]: https://www.ebi.ac.uk/ena/data/view/ERR194160
|
||||
[NA12892-data]: https://www.ebi.ac.uk/ena/data/view/ERR194161
|
||||
|
||||
Human assemblies above can be acquired [from Zenodo][zenodo-human] and
|
||||
non-human ones are available [here][zenodo-nonh].
|
||||
Except NA12878, the assemblies above were produced by hifiasm v0.12 and can be
|
||||
downloaded at
|
||||
```txt
|
||||
ftp://ftp.dfci.harvard.edu/pub/hli/hifiasm/submission/hifiasm-0.12/
|
||||
```
|
||||
NA12878 was assembled with an older version of hifiasm and is available at
|
||||
```txt
|
||||
ftp://ftp.dfci.harvard.edu/pub/hli/hifiasm/NA12878-r253/
|
||||
```
|
||||
|
||||
|
||||
[zenodo-human]: https://zenodo.org/record/4393631
|
||||
[zenodo-nonh]: https://zenodo.org/record/4393750
|
||||
[unitig]: http://wgs-assembler.sourceforge.net/wiki/index.php/Celera_Assembler_Terminology
|
||||
[gfa]: https://github.com/pmelsted/GFA-spec/blob/master/GFA-spec.md
|
||||
[paf]: https://github.com/lh3/miniasm/blob/master/PAF.md
|
||||
[yak]: https://github.com/lh3/yak
|
||||
|
||||
## <a name="help"></a>Getting Help
|
||||
## Getting Help
|
||||
|
||||
For detailed description of options, please see `man ./hifiasm.1`. The `-h`
|
||||
option of hifiasm also provides brief description of options. If you have
|
||||
further questions, please raise an issue at the [issue
|
||||
page](https://github.com/chhylp123/hifiasm/issues).
|
||||
|
||||
## <a name="limit"></a>Limitations
|
||||
## Limitations
|
||||
|
||||
1. Purging haplotig duplications may introduce misassemblies.
|
||||
|
||||
## <a name="cite"></a>Citating Hifiasm
|
||||
|
||||
If you use hifiasm in your work, please cite:
|
||||
|
||||
> Cheng, H., Concepcion, G.T., Feng, X., Zhang, H., Li H. (2021)
|
||||
> Haplotype-resolved de novo assembly using phased assembly graphs with
|
||||
> hifiasm. *Nat Methods*, **18**:170-175.
|
||||
> https://doi.org/10.1038/s41592-020-01056-5
|
||||
1. Purging haplotig duplications may introduce misassemblies.
|
||||
-181
@@ -178,187 +178,6 @@ kvec_t_u64_warp* chain_idx, void *ha_flt_tab, ha_pt_t *ha_idx, overlap_region* f
|
||||
}
|
||||
|
||||
|
||||
void calculate_ug_chaining(Candidates_list* candidates, overlap_region_alloc* overlap_list, kvec_t_u64_warp* chain_idx,
|
||||
uint64_t readID, ma_utg_v *ua, double band_width_threshold, int add_beg_end, overlap_region* f_cigar, long long mz_occ, double mz_rate)
|
||||
{
|
||||
long long i = 0;
|
||||
uint64_t current_ID;
|
||||
uint64_t current_stand;
|
||||
|
||||
if (candidates->length == 0)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
long long sub_region_beg;
|
||||
long long sub_region_end;
|
||||
long long chain_len;
|
||||
|
||||
clear_fake_cigar(&((*f_cigar).f_cigar));
|
||||
|
||||
i = 0;
|
||||
while (i < candidates->length)
|
||||
{
|
||||
chain_idx->a.n = 0;
|
||||
current_ID = candidates->list[i].readID;
|
||||
current_stand = candidates->list[i].strand;
|
||||
|
||||
///reference read
|
||||
(*f_cigar).x_id = readID;
|
||||
(*f_cigar).x_pos_strand = current_stand;
|
||||
///query read
|
||||
(*f_cigar).y_id = current_ID;
|
||||
///here the strand of query is always 0
|
||||
(*f_cigar).y_pos_strand = 0;
|
||||
|
||||
sub_region_beg = i;
|
||||
sub_region_end = i;
|
||||
i++;
|
||||
|
||||
while (i < candidates->length
|
||||
&&
|
||||
current_ID == candidates->list[i].readID
|
||||
&&
|
||||
current_stand == candidates->list[i].strand)
|
||||
{
|
||||
sub_region_end = i;
|
||||
i++;
|
||||
}
|
||||
|
||||
if ((*f_cigar).x_id == (*f_cigar).y_id)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
chain_len = chain_DP(candidates->list + sub_region_beg,
|
||||
sub_region_end - sub_region_beg + 1, &(candidates->chainDP), f_cigar, band_width_threshold,
|
||||
50, ua->a[(*f_cigar).x_id].len, ua->a[(*f_cigar).y_id].len);
|
||||
|
||||
|
||||
// if ((*f_cigar).x_id != (*f_cigar).y_id)
|
||||
if ((*f_cigar).x_id != (*f_cigar).y_id && chain_len > mz_occ*mz_rate)
|
||||
{
|
||||
append_utg_inexact_overlap_region_alloc(overlap_list, f_cigar, ua, add_beg_end);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void ha_get_ug_candidates(ha_abuf_t *ab, int64_t rid, ma_utg_t *u, ma_utg_v *ua, overlap_region_alloc *overlap_list, Candidates_list *cl, double bw_thres, int max_n_chain, int keep_whole_chain, kvec_t_u8_warp* k_flag,
|
||||
kvec_t_u64_warp* chain_idx, void *ha_flt_tab, ha_pt_t *ha_idx, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct, double chain_match_rate)
|
||||
{
|
||||
uint32_t i;
|
||||
uint64_t k, l;
|
||||
|
||||
// prepare
|
||||
clear_Candidates_list(cl);
|
||||
clear_overlap_region_alloc(overlap_list);
|
||||
ab->mz.n = 0, ab->n_a = 0;
|
||||
|
||||
// get the list of anchors
|
||||
ha_sketch_query(u->s, u->len, asm_opt.mz_win, asm_opt.k_mer_length, 0, !(asm_opt.flag & HA_F_NO_HPC), &ab->mz, ha_flt_tab, k_flag, dbg_ct);
|
||||
// minimizer of queried read
|
||||
if (ab->mz.m > ab->old_mz_m) {
|
||||
ab->old_mz_m = ab->mz.m;
|
||||
REALLOC(ab->seed, ab->old_mz_m);
|
||||
}
|
||||
for (i = 0, ab->n_a = 0; i < ab->mz.n; ++i) {
|
||||
int n;
|
||||
ab->seed[i].a = ha_pt_get(ha_idx, ab->mz.a[i].x, &n);
|
||||
ab->seed[i].n = n;
|
||||
ab->seed[i].good = 0;
|
||||
ab->n_a += n;
|
||||
}
|
||||
if (ab->n_a > ab->m_a) {
|
||||
ab->m_a = ab->n_a;
|
||||
kroundup64(ab->m_a);
|
||||
REALLOC(ab->a, ab->m_a);
|
||||
}
|
||||
for (i = 0, k = 0; i < ab->mz.n; ++i) {
|
||||
int j;
|
||||
///z is one of the minimizer
|
||||
ha_mz1_t *z = &ab->mz.a[i];
|
||||
seed1_t *s = &ab->seed[i];
|
||||
for (j = 0; j < s->n; ++j) {
|
||||
const ha_idxpos_t *y = &s->a[j];
|
||||
anchor1_t *an = &ab->a[k++];
|
||||
uint8_t rev = z->rev == y->rev? 0 : 1;
|
||||
an->other_off = y->pos;
|
||||
an->self_off = rev? u->len - 1 - (z->pos + 1 - z->span) : z->pos;
|
||||
an->good = s->good;
|
||||
an->srt = (uint64_t)y->rid<<33 | (uint64_t)rev<<32 | an->other_off;
|
||||
}
|
||||
}
|
||||
|
||||
// sort anchors
|
||||
radix_sort_ha_an1(ab->a, ab->a + ab->n_a);
|
||||
for (k = 1, l = 0; k <= ab->n_a; ++k) {
|
||||
if (k == ab->n_a || ab->a[k].srt != ab->a[l].srt) {
|
||||
if (k - l > 1)
|
||||
radix_sort_ha_an2(ab->a + l, ab->a + k);
|
||||
l = k;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// copy over to _cl_
|
||||
if (ab->m_a >= (uint64_t)cl->size) {
|
||||
cl->size = ab->m_a;
|
||||
REALLOC(cl->list, cl->size);
|
||||
}
|
||||
for (k = 0; k < ab->n_a; ++k) {
|
||||
k_mer_hit *p = &cl->list[k];
|
||||
p->readID = ab->a[k].srt >> 33;
|
||||
p->strand = ab->a[k].srt >> 32 & 1;
|
||||
p->offset = ab->a[k].other_off;
|
||||
p->self_offset = ab->a[k].self_off;
|
||||
p->good = ab->a[k].good;
|
||||
}
|
||||
cl->length = ab->n_a;
|
||||
|
||||
calculate_ug_chaining(cl, overlap_list, chain_idx, rid, ua, bw_thres, keep_whole_chain, f_cigar, ab->mz.n, chain_match_rate);
|
||||
|
||||
#if 0
|
||||
if (overlap_list->length > 0) {
|
||||
fprintf(stderr, "B\t%ld\t%ld\t%d\n", (long)rid, (long)overlap_list->length, rlen);
|
||||
for (int i = 0; i < (int)overlap_list->length; ++i) {
|
||||
overlap_region *r = &overlap_list->list[i];
|
||||
fprintf(stderr, "C\t%d\t%d\t%d\t%c\t%d\t%ld\t%d\t%d\t%c\t%d\t%d\n", (int)r->x_id, (int)r->x_pos_s, (int)r->x_pos_e, "+-"[r->x_pos_strand],
|
||||
(int)r->y_id, (long)Get_READ_LENGTH(R_INF, r->y_id), (int)r->y_pos_s, (int)r->y_pos_e, "+-"[r->y_pos_strand], (int)r->shared_seed, ha_ov_type(r, rlen));
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
if ((int)overlap_list->length > max_n_chain) {
|
||||
int32_t w, n[4], s[4];
|
||||
n[0] = n[1] = n[2] = n[3] = 0, s[0] = s[1] = s[2] = s[3] = 0;
|
||||
ks_introsort_or_ss(overlap_list->length, overlap_list->list);
|
||||
for (i = 0; i < (uint32_t)overlap_list->length; ++i) {
|
||||
const overlap_region *r = &overlap_list->list[i];
|
||||
w = ha_ov_type(r, u->len);
|
||||
++n[w];
|
||||
if ((int)n[w] == max_n_chain) s[w] = r->shared_seed;
|
||||
}
|
||||
if (s[0] > 0 || s[1] > 0 || s[2] > 0 || s[3] > 0) {
|
||||
for (i = 0, k = 0; i < (uint32_t)overlap_list->length; ++i) {
|
||||
overlap_region *r = &overlap_list->list[i];
|
||||
w = ha_ov_type(r, u->len);
|
||||
if (r->shared_seed >= s[w]) {
|
||||
if ((uint32_t)k != i) {
|
||||
overlap_region t;
|
||||
t = overlap_list->list[k];
|
||||
overlap_list->list[k] = overlap_list->list[i];
|
||||
overlap_list->list[i] = t;
|
||||
}
|
||||
++k;
|
||||
}
|
||||
}
|
||||
overlap_list->length = k;
|
||||
}
|
||||
}
|
||||
|
||||
///ks_introsort_or_xs(overlap_list->length, overlap_list->list);
|
||||
}
|
||||
|
||||
void lable_matched_ovlp(overlap_region_alloc* overlap_list, ma_hit_t_alloc* paf)
|
||||
{
|
||||
uint64_t j = 0, inner_j = 0;
|
||||
|
||||
@@ -1,30 +0,0 @@
|
||||
## Contributor Code of Conduct
|
||||
|
||||
As contributors and maintainers of this project, we pledge to respect all
|
||||
people who contribute through reporting issues, posting feature requests,
|
||||
updating documentation, submitting pull requests or patches, and other
|
||||
activities.
|
||||
|
||||
We are committed to making participation in this project a harassment-free
|
||||
experience for everyone, regardless of level of experience, gender, gender
|
||||
identity and expression, sexual orientation, disability, personal appearance,
|
||||
body size, race, age, or religion.
|
||||
|
||||
Examples of unacceptable behavior by participants include the use of sexual
|
||||
language or imagery, derogatory comments or personal attacks, trolling, public
|
||||
or private harassment, insults, or other unprofessional conduct.
|
||||
|
||||
Project maintainers have the right and responsibility to remove, edit, or
|
||||
reject comments, commits, code, wiki edits, issues, and other contributions
|
||||
that are not aligned to this Code of Conduct. Project maintainers or
|
||||
contributors who do not follow the Code of Conduct may be removed from the
|
||||
project team.
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported by opening an issue or contacting the maintainer via email.
|
||||
|
||||
This Code of Conduct is adapted from the [Contributor Covenant][cc], [version
|
||||
1.0.0][v1].
|
||||
|
||||
[cc]: http://contributor-covenant.org/
|
||||
[v1]: http://contributor-covenant.org/version/1/0/0/
|
||||
@@ -10,8 +10,8 @@
|
||||
#define RC_2 2
|
||||
|
||||
hc_edge* get_hc_edge(hc_links* link, uint64_t src, uint64_t dest, uint64_t dir);
|
||||
hc_edge* push_hc_edge(hc_linkeage* x, uint64_t uID, double weight, int dir, uint64_t* d);
|
||||
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, uint32_t is_poy);
|
||||
void push_hc_edge(hc_linkeage* x, uint64_t uID, double weight, int dir, uint64_t* d);
|
||||
void hic_analysis(ma_ug_t *ug, asg_t* read_g, hc_links* link);
|
||||
void hic_benchmark(ma_ug_t *ug, asg_t* read_g);
|
||||
|
||||
typedef struct {
|
||||
@@ -49,33 +49,6 @@ typedef struct {
|
||||
kvec_t(chain_w_type) chain_weight;
|
||||
chain_hic_warp c_w;
|
||||
} bubble_type;
|
||||
|
||||
typedef struct {
|
||||
int8_t *s;
|
||||
uint64_t xs;
|
||||
uint64_t n;
|
||||
} ps_t;
|
||||
|
||||
typedef struct {
|
||||
uint64_t s, e, id, len;
|
||||
} pe_hit;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(pe_hit) a;
|
||||
kvec_t(uint64_t) idx;
|
||||
kvec_t(uint64_t) occ;
|
||||
uint64_t uID_bits;
|
||||
uint64_t pos_mode;
|
||||
} kvec_pe_hit;
|
||||
|
||||
typedef struct{
|
||||
kvec_t(uint8_t) vis;
|
||||
kvec_t(uint64_t) x;
|
||||
kvec_t(uint64_t) dis;
|
||||
uint64_t uID_mode, uID_shift, tmp_v, tmp_d;
|
||||
}pdq;
|
||||
|
||||
|
||||
#define P_het(B) ((B).num.n)
|
||||
#define M_het(B) ((B).num.n + 1)
|
||||
// #define IF_BUB(ID, B) ((B).index[(ID)] < (B).num.n)
|
||||
@@ -89,22 +62,11 @@ void get_bubbles(bubble_type* bub, uint64_t id, uint32_t* beg, uint32_t* sink, u
|
||||
int load_hc_links(hc_links* link, const char *fn);
|
||||
void write_hc_links(hc_links* link, const char *fn);
|
||||
void destory_bubbles(bubble_type* bub);
|
||||
void identify_bubbles(ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, kv_u_trans_t *ref);
|
||||
void identify_bubbles(ma_ug_t* ug, bubble_type* bub, hc_links* link);
|
||||
void resolve_bubble_chain_tangle(ma_ug_t* ug, bubble_type* bub);
|
||||
uint32_t connect_bub_occ(bubble_type* bub, uint32_t root_id, uint32_t check_het);
|
||||
void get_bub_id(bubble_type* bub, uint32_t root, uint64_t* id0, uint64_t* id1, uint32_t check_het);
|
||||
void update_bubble_chain(ma_ug_t* ug, bubble_type* bub, uint32_t is_middle, uint32_t is_end);
|
||||
void set_b_utg_weight_flag(bubble_type* bub, buf_t* b, uint32_t v, uint8_t* vis_flag, uint32_t flag, uint32_t* occ);
|
||||
void debug_gfa_space(ma_ug_t* ug, hap_cov_t *cov);
|
||||
void init_ug_idx(ma_ug_t *ug, uint64_t k, uint64_t up_bound, uint64_t low_bound, uint64_t build_idx);
|
||||
void des_ug_idx();
|
||||
uint64_t count_unique_k_mers(char *r, uint64_t len, uint64_t query, uint64_t target, uint64_t *all, uint64_t *found);
|
||||
void init_pdq(pdq* q, uint64_t utg_num);
|
||||
void destory_pdq(pdq* q);
|
||||
uint32_t check_trans_relation_by_path(uint32_t v, uint32_t w, pdq* pqv, uint32_t* path_v, buf_t *resv,
|
||||
pdq* pqw, uint32_t* path_w, buf_t *resw, asg_t *sg, uint8_t *dest, uint8_t df, uint32_t df_occ, double rate,
|
||||
long long *dis);
|
||||
void set_utg_by_dis(uint32_t v, pdq* pq, asg_t *g, kvec_t_u32_warp *res, uint32_t dis);
|
||||
void dedup_hits(kvec_pe_hit* hits, uint64_t is_dup);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
.TH hifiasm 1 "16 April 2021" "hifiasm-0.15 (r327)" "Bioinformatics tools"
|
||||
.TH hifiasm 1 "19 July 2020" "hifiasm-0.9 (r289)" "Bioinformatics tools"
|
||||
|
||||
.SH NAME
|
||||
.PP
|
||||
@@ -212,38 +212,6 @@ with suffix
|
||||
.B lowQ.bed
|
||||
[70]. Set 0 to disable.
|
||||
|
||||
|
||||
.TP
|
||||
.BI --b-cov \ INT
|
||||
Break contigs at potential misassemblies with <INT-fold coverage [0].
|
||||
Work with
|
||||
.B --m-rate.
|
||||
Set 0 to disable.
|
||||
|
||||
.TP
|
||||
.BI --h-cov \ INT
|
||||
Break contigs at potential misassemblies with >INT-fold coverage [-1].
|
||||
Work with
|
||||
.B --m-rate.
|
||||
Set -1 to disable.
|
||||
|
||||
.TP
|
||||
.BI --m-rate \ FLOAT
|
||||
Break contigs with <=FLOAT*coverage exact overlaps [0.75].
|
||||
Only work with
|
||||
.B --b-cov
|
||||
and
|
||||
.B --h-cov.
|
||||
|
||||
.TP
|
||||
.BI --primary
|
||||
Output a primary assembly and an alternate assembly.
|
||||
Hifiasm outputs two balanced assemblies and a primary
|
||||
assembly in default. Enable this option or
|
||||
.B -l0
|
||||
outputs a primary assembly and an alternate assembly.
|
||||
|
||||
|
||||
.SS Trio-partition options
|
||||
|
||||
.TP 10
|
||||
@@ -292,14 +260,12 @@ times in the other sample.
|
||||
.TP 10
|
||||
.BI -l \ INT
|
||||
Level of purge-dup. 0 to disable purge-dup, 1 to only purge contained haplotigs,
|
||||
2 to purge all types of haplotigs, 3 to purge all types of haplotigs in most aggressive way
|
||||
for high heterozygosity sample.
|
||||
In default, [3] for non-trio assembly, [0] for trio assembly.
|
||||
2 to purge all types of haplotigs. In default, [2] for non-trio assembly, [0] for trio assembly.
|
||||
For trio assembly, only level 0 and level 1 are allowed.
|
||||
|
||||
.TP
|
||||
.BI -s \ FLOAT
|
||||
Similarity threshold for duplicate haplotigs that should be purged [0.75 for -l1/-l2, 0.55 for -l3].
|
||||
Similarity threshold for duplicate haplotigs that should be purged [0.75].
|
||||
|
||||
.TP
|
||||
.BI -O \ FLOAT
|
||||
@@ -311,8 +277,9 @@ Coverage upper bound of Purge-dups, which is inferred automatically in default.
|
||||
If the coverage of a contig is higher than this bound, don't apply Purge-dups.
|
||||
|
||||
.TP
|
||||
.BI --n-hap \ INT
|
||||
Assumption of haplotype number.
|
||||
.BI --high-het \ INT
|
||||
Enable this mode for high heterozygosity sample, which will increase running time.
|
||||
For ordinary samples, no need to enable this mode [experimental, not stable].
|
||||
|
||||
|
||||
.SS Debugging options
|
||||
@@ -322,39 +289,14 @@ Assumption of haplotype number.
|
||||
Write additional files to speed up the debugging of graph cleaning.
|
||||
|
||||
|
||||
.SS Hi-C-partition options [experimental, not stable]
|
||||
|
||||
.TP
|
||||
.BI --h1 \ FILEs
|
||||
File names of input Hi-C R1 [r1_1.fq,r1_2.fq,...].
|
||||
|
||||
.TP
|
||||
.BI --h2 \ FILEs
|
||||
File names of input Hi-C R2 [r2_1.fq,r2_2.fq,...].
|
||||
|
||||
.TP
|
||||
.BI --n-weight \ INT
|
||||
Rounds of reweighting Hi-C links [3]. Increasing this may improves
|
||||
phasing results but takes longer time.
|
||||
|
||||
.TP
|
||||
.BI --n-perturb \ INT
|
||||
Rounds of perturbation [10000]. Increasing this may improves
|
||||
phasing results but takes longer time.
|
||||
|
||||
.TP
|
||||
.BI --f-perturb \ FLOAT
|
||||
Fraction to flip for perturbation [0.1]. Increasing this may improves
|
||||
phasing results but takes longer time.
|
||||
|
||||
.TP
|
||||
.BI --seed \ INT
|
||||
RNG seed [11].
|
||||
|
||||
.SH OUTPUTS
|
||||
|
||||
.PP
|
||||
In general, hifiasm generates the following assembly graphs in the GFA format:
|
||||
Without trio partition options
|
||||
.B -1
|
||||
and
|
||||
.BR -2 ,
|
||||
hifiasm generates the following assembly graphs in the GFA format:
|
||||
|
||||
.RS 2
|
||||
.TP 2
|
||||
@@ -381,111 +323,30 @@ assembly graph of primary contigs. This graph collapses different haplotypes.
|
||||
assembly graph of alternate contigs. This graph consists of all assemblies that
|
||||
are discarded in primary contig graph.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .hap*.p_ctg.gfa:
|
||||
phased contig graph. This graph keeps the phased assembly.
|
||||
|
||||
.RE
|
||||
|
||||
.PP
|
||||
Hifiasm outputs
|
||||
.B *.r_utg.gfa
|
||||
and
|
||||
.B *.p_utg.gfa
|
||||
in any cases.
|
||||
Specifically, hifiasm outputs the following assembly graphs
|
||||
with trio-binning options:
|
||||
With trio partition, hifiasm outputs the following assembly graphs:
|
||||
|
||||
.RS 2
|
||||
.TP 2
|
||||
*
|
||||
.IR prefix .dip.hap1.p_ctg.gfa:
|
||||
phased paternal/haplotype1 contig graph keeping the phased
|
||||
.IR prefix .dip.r_utg.gfa:
|
||||
haplotype-resolved raw unitig graph. This graph keeps all haplotype information.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .hap1.p_ctg.gfa:
|
||||
phased paternal/haplotype1 contig graph. This graph keeps the phased
|
||||
paternal/haplotype1 assembly.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .dip.hap2.p_ctg.gfa:
|
||||
phased maternal/haplotype2 contig graph keeping the phased
|
||||
.IR prefix .hap2.p_ctg.gfa:
|
||||
phased maternal/haplotype2 contig graph. This graph keeps the phased
|
||||
maternal/haplotype2 assembly.
|
||||
.RE
|
||||
|
||||
.PP
|
||||
With Hi-C partition options, hifiasm outputs:
|
||||
|
||||
.RS 2
|
||||
.TP 2
|
||||
*
|
||||
.IR prefix .hic.p_ctg.gfa:
|
||||
assembly graph of primary contigs. This graph collapses different haplotypes.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .hic.hap1.p_ctg.gfa:
|
||||
phased contig graph where each contig is fully phased.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .hic.hap2.p_ctg.gfa:
|
||||
phased contig graph where each contig is fully phased.
|
||||
.RE
|
||||
|
||||
|
||||
.PP
|
||||
Hifiasm keeps Hi-C alignment results and Hi-C index in two bin
|
||||
files:
|
||||
.B *hic.lk.bin
|
||||
and
|
||||
.B *hic.tlb.bin.
|
||||
Rerunning hifiasm with different Hi-C reads needs to delete these bin files
|
||||
or enable
|
||||
.BR -i .
|
||||
.RE
|
||||
|
||||
.PP
|
||||
Hifiasm generates the following assembly graphs only with HiFi reads:
|
||||
|
||||
.RS 2
|
||||
.TP 2
|
||||
*
|
||||
.IR prefix .p_ctg.gfa:
|
||||
assembly graph of primary contigs. This graph collapses different haplotypes.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .bp.hap1.p_ctg.gfa:
|
||||
balanced contig graph where each contig is partially phased.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .bp.hap2.p_ctg.gfa:
|
||||
balanced contig graph where each contig is partially phased.
|
||||
.RE
|
||||
|
||||
.PP
|
||||
If the option
|
||||
.BR -p
|
||||
or
|
||||
.BR --primary
|
||||
is specified, hifiasm outputs:
|
||||
|
||||
.RS 2
|
||||
.TP 2
|
||||
*
|
||||
.IR prefix .p_ctg.gfa:
|
||||
assembly graph of primary contigs. This graph collapses different haplotypes.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .a_ctg.gfa:
|
||||
assembly graph of alternate contigs. This graph consists of all assemblies that
|
||||
are discarded in primary contig graph.
|
||||
.RE
|
||||
|
||||
|
||||
|
||||
|
||||
.PP
|
||||
For each graph, hifiasm also outputs a simplified version without sequences for
|
||||
the ease of visualization. Hifiasm keeps corrected reads and overlaps in three
|
||||
|
||||
@@ -12,37 +12,6 @@ static void ha_hist_line(int c, int x, int exceed, int64_t cnt)
|
||||
fprintf(stderr, " %lld\n", (long long)cnt);
|
||||
}
|
||||
|
||||
void print_hist_lines(int n_cnt, int start_cnt, const int64_t *cnt)
|
||||
{
|
||||
const int hist_max = 100;
|
||||
int i, start, low_i, max_i, max;
|
||||
// determine the start point
|
||||
assert(n_cnt > start_cnt);
|
||||
start = cnt[1] > 0? 1 : 2;
|
||||
|
||||
// find the low point from the left
|
||||
low_i = start > start_cnt? start : start_cnt;
|
||||
for (i = low_i; i < n_cnt; ++i)
|
||||
if (cnt[i] > cnt[i-1]) break;
|
||||
low_i = i - 1;
|
||||
fprintf(stderr, "[M::%s] lowest: count[%d] = %ld\n", __func__, low_i, (long)cnt[low_i]);
|
||||
|
||||
// find the highest peak
|
||||
max_i = start > start_cnt? start : start_cnt, max = cnt[max_i];
|
||||
for (i = max_i; i < n_cnt; ++i)
|
||||
if (cnt[i] > max)
|
||||
max = cnt[i], max_i = i;
|
||||
fprintf(stderr, "[M::%s] highest: count[%d] = %ld\n", __func__, max_i, (long)cnt[max_i]);
|
||||
|
||||
for (i = start; i < n_cnt; ++i) {
|
||||
int x, exceed = 0;
|
||||
x = (int)((double)hist_max * cnt[i] / cnt[max_i] + .499);
|
||||
if (x > hist_max) exceed = 1, x = hist_max; // may happen if cnt[2] is higher
|
||||
if (i > max_i && x == 0) break;
|
||||
ha_hist_line(i, x, exceed, cnt[i]);
|
||||
}
|
||||
}
|
||||
|
||||
int ha_analyze_count(int n_cnt, int start_cnt, const int64_t *cnt, int *peak_het)
|
||||
{
|
||||
const int hist_max = 100;
|
||||
|
||||
-3868
File diff suppressed because it is too large
Load Diff
@@ -1,64 +0,0 @@
|
||||
#ifndef __HORDER__
|
||||
#define __HORDER__
|
||||
#include <stdint.h>
|
||||
#include "hic.h"
|
||||
|
||||
#define get_hit_srev(x, k) ((x).a.a[(k)].s>>63)
|
||||
#define get_hit_slen(x, k) ((x).a.a[(k)].len>>32)
|
||||
#define get_hit_suid(x, k) (((x).a.a[(k)].s<<1)>>(64 - (x).uID_bits))
|
||||
#define get_hit_spos(x, k) ((x).a.a[(k)].s & (x).pos_mode)
|
||||
#define get_hit_spos_e(x, k) (get_hit_srev((x),(k))?\
|
||||
((get_hit_spos((x),(k))+1>=get_hit_slen((x),(k)))?\
|
||||
(get_hit_spos((x),(k))+1-get_hit_slen((x),(k))):0)\
|
||||
:(get_hit_spos((x),(k))+get_hit_slen((x),(k))-1))
|
||||
|
||||
#define get_hit_erev(x, k) ((x).a.a[(k)].e>>63)
|
||||
#define get_hit_elen(x, k) ((uint32_t)((x).a.a[(k)].len))
|
||||
#define get_hit_euid(x, k) (((x).a.a[(k)].e<<1)>>(64 - (x).uID_bits))
|
||||
#define get_hit_epos(x, k) ((x).a.a[(k)].e & (x).pos_mode)
|
||||
#define get_hit_epos_e(x, k) (get_hit_erev((x),(k))?\
|
||||
((get_hit_epos((x),(k))+1>=get_hit_elen((x),(k)))?\
|
||||
(get_hit_epos((x),(k))+1-get_hit_elen((x),(k))):0)\
|
||||
:(get_hit_epos((x),(k))+get_hit_elen((x),(k))-1))
|
||||
|
||||
typedef struct {
|
||||
uint32_t v;
|
||||
uint32_t u;
|
||||
uint32_t occ:31, del:1;
|
||||
double w, nw;
|
||||
} osg_arc_t;
|
||||
|
||||
typedef struct {
|
||||
double mw[2], ez[2];
|
||||
uint8_t del;
|
||||
} osg_seq_t;
|
||||
|
||||
typedef struct {
|
||||
uint32_t m_arc, n_arc:31, is_srt:1;
|
||||
osg_arc_t *arc;
|
||||
|
||||
uint32_t m_seq, n_seq:31, is_symm:1;
|
||||
osg_seq_t *seq;
|
||||
|
||||
uint64_t *idx;
|
||||
} osg_t;
|
||||
|
||||
typedef struct {
|
||||
osg_t *g;
|
||||
}scg_t;
|
||||
typedef struct {
|
||||
kvec_t(uint64_t) avoid;
|
||||
kvec_pe_hit r_hits, u_hits;
|
||||
ma_ug_t *ug;
|
||||
asg_t *r_g;
|
||||
scg_t sg;
|
||||
}horder_t;
|
||||
|
||||
horder_t *init_horder_t(kvec_pe_hit *i_hits, uint64_t i_hits_uid_bits, uint64_t i_hits_pos_mode,
|
||||
asg_t *i_rg, ma_ug_t* i_ug, bubble_type* bub, kv_u_trans_t *ref, ug_opt_t *opt, uint32_t round);
|
||||
void destory_horder_t(horder_t **h);
|
||||
void horder_clean_sg_by_utg(asg_t *sg, ma_ug_t *ug);
|
||||
kvec_pe_hit *get_r_hits_for_trio(kvec_pe_hit *u_hits, asg_t* r_g, ma_ug_t* ug, bubble_type* bub, uint64_t uID_bits, uint64_t pos_mode);
|
||||
void update_switch_unitig(ma_ug_t *ug, asg_t *rg, kvec_pe_hit *hits, kv_u_trans_t *k_trans, uint64_t cutoff_s, uint64_t cutoff_e,
|
||||
uint64_t min_ulen, double boundaryRate);
|
||||
#endif
|
||||
@@ -193,8 +193,10 @@ static int ha_ct_insert_list(ha_ct_t *h, int create_new, int n, const uint64_t *
|
||||
khint_t k;
|
||||
if ((a[j]&mask) != (a[0]&mask)) continue;
|
||||
if (create_new) {
|
||||
///for 0-th counting, g->b = NULL
|
||||
if (g->b)
|
||||
ins = (yak_bf_insert(g->b, x) == h->n_hash);
|
||||
///for 0-th counting, g->b = NULL
|
||||
///x = the high 52 bits of a[j] + low 12 bits 0
|
||||
///the low 12 bits are used for counting
|
||||
if (ins) {
|
||||
@@ -513,7 +515,6 @@ KSEQ_INIT(gzFile, gzread)
|
||||
#define HAF_RS_READ 0x10
|
||||
#define HAF_CREATE_NEW 0x20
|
||||
#define HAF_SKIP_READ 0x40
|
||||
#define HAF_UG_READ 0x80
|
||||
|
||||
typedef struct { // global data structure for kt_pipeline()
|
||||
const yak_copt_t *opt;
|
||||
@@ -526,7 +527,6 @@ typedef struct { // global data structure for kt_pipeline()
|
||||
ha_pt_t *pt;
|
||||
const All_reads *rs_in;
|
||||
All_reads *rs_out;
|
||||
const ma_utg_v *us_in;
|
||||
} pl_data_t;
|
||||
|
||||
typedef struct { // data structure for each step in kt_pipeline()
|
||||
@@ -597,24 +597,6 @@ static void *worker_count(void *data, int step, void *in) // callback for kt_pip
|
||||
if (s->sum_len >= p->opt->chunk_size)
|
||||
break;
|
||||
}
|
||||
} else if(p->us_in) {
|
||||
ma_utg_t *u;
|
||||
while (p->n_seq < p->us_in->n) {
|
||||
u = &(p->us_in->a[p->n_seq]);
|
||||
if (s->n_seq == s->m_seq) {
|
||||
s->m_seq = s->m_seq < 16? 16 : s->m_seq + (s->m_seq>>1);
|
||||
REALLOC(s->len, s->m_seq);
|
||||
REALLOC(s->seq, s->m_seq);
|
||||
}
|
||||
MALLOC(s->seq[s->n_seq], u->len);
|
||||
memcpy(s->seq[s->n_seq], u->s, u->len);
|
||||
s->len[s->n_seq++] = u->len;
|
||||
++p->n_seq;
|
||||
s->sum_len += u->len;
|
||||
s->nk += u->len >= p->opt->k? u->len - p->opt->k + 1 : 0;
|
||||
if (s->sum_len >= p->opt->chunk_size)
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
while ((ret = kseq_read(p->ks)) >= 0) {
|
||||
int l = (int)(p->ks->seq.l) - (int)(p->opt->adaLen) - (int)(p->opt->adaLen);
|
||||
@@ -783,18 +765,15 @@ void debug_adapter(const hifiasm_opt_t *asm_opt, All_reads *rs)
|
||||
exit(1);
|
||||
}
|
||||
|
||||
static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt_t *p0, ha_ct_t *c0, const void *flt_tab, All_reads *rs, ma_utg_v *us, int64_t *n_seq)
|
||||
static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt_t *p0, ha_ct_t *c0, const void *flt_tab, All_reads *rs, int64_t *n_seq)
|
||||
{
|
||||
///for 0-th counting, flag = HAF_COUNT_ALL|HAF_RS_WRITE_LEN|HAF_CREATE_NEW
|
||||
int read_rs = (rs && (flag & HAF_RS_READ));
|
||||
int ug_rs = (us && (flag & HAF_UG_READ));
|
||||
pl_data_t pl;
|
||||
gzFile fp = 0;
|
||||
memset(&pl, 0, sizeof(pl_data_t));
|
||||
pl.n_seq = *n_seq;
|
||||
if(ug_rs) {
|
||||
pl.us_in = us;
|
||||
} else if (read_rs) {
|
||||
if (read_rs) {
|
||||
pl.rs_in = rs;
|
||||
init_UC_Read(&pl.ucr);
|
||||
} else {///for 0-th counting, go into here
|
||||
@@ -825,7 +804,7 @@ static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt
|
||||
kt_pipeline(3, worker_count, &pl, 3);
|
||||
if (read_rs) {
|
||||
destory_UC_Read(&pl.ucr);
|
||||
} else if(!read_rs && !ug_rs) {
|
||||
} else {
|
||||
kseq_destroy(pl.ks);
|
||||
gzclose(fp);
|
||||
}
|
||||
@@ -833,7 +812,7 @@ static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt
|
||||
return pl.ct;
|
||||
}
|
||||
|
||||
ha_ct_t *ha_count(const hifiasm_opt_t *asm_opt, int flag, ha_pt_t *p0, const void *flt_tab, All_reads *rs, ma_utg_v *us, int keep_adapter)
|
||||
ha_ct_t *ha_count(const hifiasm_opt_t *asm_opt, int flag, ha_pt_t *p0, const void *flt_tab, All_reads *rs)
|
||||
{
|
||||
int i;
|
||||
int64_t n_seq = 0;
|
||||
@@ -857,10 +836,10 @@ ha_ct_t *ha_count(const hifiasm_opt_t *asm_opt, int flag, ha_pt_t *p0, const voi
|
||||
///for ha_pt_gen, shoud be 0
|
||||
opt.bf_shift = flag & HAF_COUNT_EXACT? 0 : asm_opt->bf_shift;
|
||||
opt.n_thread = asm_opt->thread_num;
|
||||
opt.adaLen = (keep_adapter? asm_opt->adapterLen : 0);
|
||||
opt.adaLen = asm_opt->adapterLen;
|
||||
///asm_opt->num_reads is the number of fastq files
|
||||
for (i = 0; i < asm_opt->num_reads; ++i)
|
||||
h = yak_count(&opt, asm_opt->read_file_names[i], flag|HAF_CREATE_NEW, p0, h, flt_tab, rs, us, &n_seq);
|
||||
h = yak_count(&opt, asm_opt->read_file_names[i], flag|HAF_CREATE_NEW, p0, h, flt_tab, rs, &n_seq);
|
||||
if (h && opt.bf_shift > 0)
|
||||
ha_ct_destroy_bf(h);
|
||||
return h;
|
||||
@@ -935,59 +914,6 @@ void debug_ct_index(void* q_ct_idx, void* r_ct_idx)
|
||||
* High-level interfaces *
|
||||
*************************/
|
||||
|
||||
void *ha_ft_ug_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int hap_n)
|
||||
{
|
||||
yak_ft_t *flt_tab;
|
||||
int64_t cnt[YAK_N_COUNTS];
|
||||
int cutoff = hap_n + 1;
|
||||
ha_ct_t *h;
|
||||
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_UG_READ|HAF_COUNT_EXACT, NULL, NULL, NULL, us, 0);
|
||||
|
||||
ha_ct_hist(h, cnt, asm_opt->thread_num);
|
||||
print_hist_lines(YAK_N_COUNTS, 1, cnt);
|
||||
|
||||
ha_ct_shrink(h, cutoff, YAK_MAX_COUNT, asm_opt->thread_num);
|
||||
flt_tab = gen_hh(h);
|
||||
ha_ct_destroy(h);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f@%.3fGB] ==> filtered out %ld k-mers occurring %d or more times\n", __func__,
|
||||
yak_realtime(), yak_cpu_usage(), yak_peakrss_in_gb(), (long)kh_size(flt_tab), cutoff);
|
||||
return (void*)flt_tab;
|
||||
}
|
||||
|
||||
|
||||
ha_pt_t *ha_pt_ug_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, ma_utg_v *us, int hap_n)
|
||||
{
|
||||
int64_t cnt[YAK_N_COUNTS], tot_cnt;
|
||||
int i;
|
||||
ha_ct_t *ct;
|
||||
ha_pt_t *pt;
|
||||
///HAF_COUNT_EXACT: no bf
|
||||
ct = ha_count(asm_opt, HAF_COUNT_EXACT|HAF_UG_READ, NULL, flt_tab, NULL, us, 0);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> counted %ld distinct minimizer k-mers\n", __func__,
|
||||
yak_realtime(), yak_cpu_usage(), (long)ct->tot);
|
||||
|
||||
ha_ct_hist(ct, cnt, asm_opt->thread_num);
|
||||
print_hist_lines(YAK_N_COUNTS, 1, cnt);
|
||||
|
||||
///here ha_ct_shrink is mostly used to remove k-mer appearing only 1 time
|
||||
if (flt_tab == 0) {
|
||||
ha_ct_shrink(ct, 2, hap_n, asm_opt->thread_num);
|
||||
for (i = 2, tot_cnt = 0; i <= hap_n; ++i) tot_cnt += cnt[i] * i;
|
||||
} else {
|
||||
///Note: here is just to remove minimizer appearing YAK_MAX_COUNT times
|
||||
///minimizer with YAK_MAX_COUNT occ may apper > YAK_MAX_COUNT times, so it may lead to overflow at ha_pt_gen
|
||||
ha_ct_shrink(ct, 2, YAK_MAX_COUNT - 1, asm_opt->thread_num);
|
||||
for (i = 2, tot_cnt = 0; i <= YAK_MAX_COUNT - 1; ++i) tot_cnt += cnt[i] * i;
|
||||
}
|
||||
pt = ha_pt_gen(ct, asm_opt->thread_num);
|
||||
ha_count(asm_opt, HAF_COUNT_EXACT|HAF_UG_READ, pt, flt_tab, NULL, us, 0);
|
||||
assert((uint64_t)tot_cnt == pt->tot_pos);
|
||||
//ha_pt_sort(pt, asm_opt->thread_num);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> indexed %ld positions\n", __func__,
|
||||
yak_realtime(), yak_cpu_usage(), (long)pt->tot_pos);
|
||||
return pt;
|
||||
}
|
||||
|
||||
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode)
|
||||
{
|
||||
yak_ft_t *flt_tab;
|
||||
@@ -995,7 +921,7 @@ void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int i
|
||||
int peak_hom, peak_het, cutoff = YAK_MAX_COUNT - 1, ex_flag = 0;
|
||||
if(is_hp_mode) ex_flag = HAF_RS_READ|HAF_SKIP_READ;
|
||||
ha_ct_t *h;
|
||||
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_RS_WRITE_LEN|ex_flag, NULL, NULL, rs, NULL, 1);
|
||||
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_RS_WRITE_LEN|ex_flag, NULL, NULL, rs);
|
||||
if((asm_opt->flag & HA_F_VERBOSE_GFA))
|
||||
{
|
||||
write_ct_index((void*)h, asm_opt->output_file_name);
|
||||
@@ -1040,7 +966,7 @@ ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_f
|
||||
}
|
||||
if(is_hp_mode) extra_flag1 |= HAF_SKIP_READ, extra_flag2 |= HAF_SKIP_READ;
|
||||
|
||||
ct = ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag1, NULL, flt_tab, rs, NULL, 1);
|
||||
ct = ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag1, NULL, flt_tab, rs);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> counted %ld distinct minimizer k-mers\n", __func__,
|
||||
yak_realtime(), yak_cpu_usage(), (long)ct->tot);
|
||||
ha_ct_hist(ct, cnt, asm_opt->thread_num);
|
||||
@@ -1063,7 +989,7 @@ ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_f
|
||||
for (i = 2, tot_cnt = 0; i <= YAK_MAX_COUNT - 1; ++i) tot_cnt += cnt[i] * i;
|
||||
}
|
||||
pt = ha_pt_gen(ct, asm_opt->thread_num);
|
||||
ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag2, pt, flt_tab, rs, NULL, 1);
|
||||
ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag2, pt, flt_tab, rs);
|
||||
assert((uint64_t)tot_cnt == pt->tot_pos);
|
||||
//ha_pt_sort(pt, asm_opt->thread_num);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> indexed %ld positions\n", __func__,
|
||||
|
||||
@@ -18,41 +18,6 @@ typedef struct {
|
||||
|
||||
typedef struct { uint32_t n, m; ha_mz1_t *a; } ha_mz1_v;
|
||||
|
||||
typedef struct {
|
||||
uint64_t x; ///x is the hash key
|
||||
///rid is the read id, pos is the end pos of this minimizer, rev is the direction
|
||||
///span is the length of this k-mer. For non-HPC k-mer, span may not be equal to k
|
||||
uint64_t rid:30, pos:34;
|
||||
uint16_t rev:1, span:15;
|
||||
} ha_mzl_t;
|
||||
|
||||
typedef struct {
|
||||
uint64_t rid:30, pos:34;
|
||||
uint16_t rev:1, span:15;
|
||||
} ha_mzl_idxpos_t;
|
||||
|
||||
typedef struct { uint32_t n, m; ha_mzl_t *a; } ha_mzl_v;
|
||||
|
||||
typedef struct { // a simplified version of kdq
|
||||
int front, count;
|
||||
int a[64];
|
||||
} tiny_queue_t;
|
||||
|
||||
static inline void tq_push(tiny_queue_t *q, int x)
|
||||
{
|
||||
q->a[((q->count++) + q->front) & 0x3f] = x;
|
||||
}
|
||||
|
||||
static inline int tq_shift(tiny_queue_t *q)
|
||||
{
|
||||
int x;
|
||||
if (q->count == 0) return -1;
|
||||
x = q->a[q->front++];
|
||||
q->front &= 0x3f;
|
||||
--q->count;
|
||||
return x;
|
||||
}
|
||||
|
||||
struct ha_pt_s;
|
||||
typedef struct ha_pt_s ha_pt_t;
|
||||
|
||||
@@ -66,12 +31,11 @@ extern void *ha_flt_tab_hp;
|
||||
extern ha_pt_t *ha_idx_hp;
|
||||
extern void *ha_ct_table;
|
||||
|
||||
void *ha_ft_ug_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int hap_n);
|
||||
|
||||
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode);
|
||||
int ha_ft_isflt(const void *hh, uint64_t y);
|
||||
void ha_ft_destroy(void *h);
|
||||
|
||||
ha_pt_t *ha_pt_ug_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, ma_utg_v *us, int hap_n);
|
||||
ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_from_store, int is_hp_mode, All_reads *rs, int *hom_cov, int *het_cov);
|
||||
void ha_pt_destroy(ha_pt_t *h);
|
||||
const ha_idxpos_t *ha_pt_get(const ha_pt_t *h, uint64_t hash, int *n);
|
||||
@@ -98,7 +62,6 @@ void ha_triobin(const hifiasm_opt_t *opt);
|
||||
void ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mz1_v *p, const void *hf);
|
||||
void ha_sketch_query(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mz1_v *p, const void *hf, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct);
|
||||
int ha_analyze_count(int n_cnt, int start_cnt, const int64_t *cnt, int *peak_het);
|
||||
void print_hist_lines(int n_cnt, int start_cnt, const int64_t *cnt);
|
||||
void debug_adapter(const hifiasm_opt_t *asm_opt, All_reads *rs);
|
||||
|
||||
static inline uint64_t yak_hash64(uint64_t key, uint64_t mask) // invertible integer hash function
|
||||
|
||||
@@ -11,7 +11,7 @@ int main(int argc, char *argv[])
|
||||
int i, ret;
|
||||
yak_reset_realtime();
|
||||
init_opt(&asm_opt);
|
||||
if (!CommandLine_process(argc, argv, &asm_opt)) return 0;
|
||||
if (!CommandLine_process(argc, argv, &asm_opt)) return 1;
|
||||
ret = ha_assemble();
|
||||
destory_opt(&asm_opt);
|
||||
fprintf(stderr, "[M::%s] Version: %s\n", __func__, HA_VERSION);
|
||||
|
||||
@@ -1,128 +0,0 @@
|
||||
#ifndef __RCUT__
|
||||
#define __RCUT__
|
||||
#include <stdio.h>
|
||||
#include <stdint.h>
|
||||
#include "kvec.h"
|
||||
#include "Overlaps.h"
|
||||
#include "Purge_Dups.h"
|
||||
#include "hic.h"
|
||||
|
||||
typedef struct {
|
||||
uint32_t bS, bE;
|
||||
uint32_t nS, nE;
|
||||
uint32_t uID;
|
||||
uint8_t hs;
|
||||
}mc_interval_t;
|
||||
|
||||
#define mc_node_t int8_t
|
||||
#define mcg_node_t uint32_t
|
||||
// #define w_t int64_t
|
||||
// #define t_w_t int64_t
|
||||
// #define w_cast(x) ((t_w_t)((x) < 0 ? (x) - 0.5 : (x) + 0.5))
|
||||
|
||||
#define w_t double
|
||||
#define t_w_t double
|
||||
#define w_cast(x) ((t_w_t)((x)))
|
||||
#define MC_NAME "debug_mc.bin"
|
||||
|
||||
typedef struct {
|
||||
uint64_t x; ///(uint64_t)nid1 << 32 | nid2;
|
||||
w_t w; ///might be negative or positive
|
||||
} mc_edge_t;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(uint64_t) idx;
|
||||
kvec_t(mc_edge_t) ma;
|
||||
uint64_t* cc;
|
||||
uint32_t n_seq;
|
||||
} mc_match_t;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(mc_node_t) s;
|
||||
ma_ug_t *ug;
|
||||
asg_t *rg;
|
||||
mc_match_t* e;
|
||||
}mc_g_t;
|
||||
|
||||
typedef struct {
|
||||
uint32_t a[2], occ[2];
|
||||
mc_node_t s[2];
|
||||
t_w_t z[4];
|
||||
}mb_node_t;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(uint32_t) bid;
|
||||
kvec_t(uint32_t) idx;
|
||||
kvec_t(mb_node_t) u;
|
||||
}mb_nodes_t;
|
||||
|
||||
typedef struct {
|
||||
uint64_t x; ///(uint64_t)nid1 << 32 | nid2;
|
||||
t_w_t w[4]; ///might be negative or positive
|
||||
} mb_edge_t;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(uint64_t) idx;
|
||||
kvec_t(mb_edge_t) ma;
|
||||
uint64_t* cc;
|
||||
uint32_t n_seq;
|
||||
} mb_match_t;
|
||||
|
||||
typedef struct {
|
||||
mb_nodes_t* u;
|
||||
mb_match_t* e;
|
||||
}mb_g_t;
|
||||
|
||||
typedef struct {
|
||||
mcg_node_t s;
|
||||
uint16_t h[2], hc;
|
||||
double hw[2];
|
||||
}mc_gg_status;
|
||||
|
||||
typedef struct {
|
||||
mc_gg_status *a;
|
||||
size_t n, m;
|
||||
}kv_gg_status;
|
||||
|
||||
typedef struct {
|
||||
mcg_node_t *a;
|
||||
size_t n, m;
|
||||
}mcb_t;
|
||||
|
||||
|
||||
typedef struct {
|
||||
kv_gg_status *s;
|
||||
// ma_ug_t *ug;
|
||||
// asg_t *rg;
|
||||
uint32_t un;
|
||||
mc_match_t* e;
|
||||
kvec_t(mcb_t) m;
|
||||
mcg_node_t mask;
|
||||
uint16_t hN;
|
||||
}mc_gg_t;
|
||||
|
||||
|
||||
static inline uint64_t kr_splitmix64(uint64_t x)
|
||||
{
|
||||
uint64_t z = (x += 0x9E3779B97F4A7C15ULL);
|
||||
z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9ULL;
|
||||
z = (z ^ (z >> 27)) * 0x94D049BB133111EBULL;
|
||||
return z ^ (z >> 31);
|
||||
}
|
||||
|
||||
static inline double kr_drand_r(uint64_t *x)
|
||||
{
|
||||
union { uint64_t i; double d; } u;
|
||||
*x = kr_splitmix64(*x);
|
||||
u.i = 0x3FFULL << 52 | (*x) >> 12;
|
||||
return u.d - 1.0;
|
||||
}
|
||||
|
||||
void mc_solve(hap_overlaps_list* ovlp, trans_chain* t_ch, kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g, double f_rate, uint8_t* trio_flag, uint32_t renew_s, int8_t *s, uint32_t is_sys, bubble_type* bub, kv_u_trans_t *ref);
|
||||
void debug_mc_g_t(const char* name);
|
||||
void mc_solve_general(kv_u_trans_t *ta, uint32_t un, kv_gg_status *s, uint16_t hapN, uint16_t update_ta, uint16_t write_dump);
|
||||
kv_gg_status *init_mc_gg_status(ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut,
|
||||
ma_hit_t_alloc* sources, R_to_U* ruIndex, uint64_t t_cov, uint16_t hapN);
|
||||
void destory_mc_gg_t(mc_gg_t **p);
|
||||
void debug_mc_gg_t(const char* fn, uint32_t update_ta, uint32_t convert_mc_g_t);
|
||||
#endif
|
||||
+21
@@ -5,6 +5,26 @@
|
||||
#include "kvec.h"
|
||||
#include "htab.h"
|
||||
|
||||
typedef struct { // a simplified version of kdq
|
||||
int front, count;
|
||||
int a[64];
|
||||
} tiny_queue_t;
|
||||
|
||||
static inline void tq_push(tiny_queue_t *q, int x)
|
||||
{
|
||||
q->a[((q->count++) + q->front) & 0x3f] = x;
|
||||
}
|
||||
|
||||
static inline int tq_shift(tiny_queue_t *q)
|
||||
{
|
||||
int x;
|
||||
if (q->count == 0) return -1;
|
||||
x = q->a[q->front++];
|
||||
q->front &= 0x3f;
|
||||
--q->count;
|
||||
return x;
|
||||
}
|
||||
|
||||
/**
|
||||
* Find symmetric (w,k)-minimizers on a DNA sequence
|
||||
*
|
||||
@@ -140,6 +160,7 @@ kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct)
|
||||
memset(k_flag->a.a, 0, k_flag->a.n);
|
||||
}
|
||||
|
||||
|
||||
assert(len > 0 && len < 1<<27 && rid < 1<<28 && (w > 0 && w < 256) && (k > 0 && k <= 63));
|
||||
///sizeof(ha_mz1_t) = 16
|
||||
memset(buf, 0xff, w * 16);
|
||||
|
||||
@@ -1,26 +0,0 @@
|
||||
#ifndef __TOVLP__
|
||||
#define __TOVLP__
|
||||
#include <stdint.h>
|
||||
#include "Overlaps.h"
|
||||
|
||||
typedef struct {///[cBeg, cEnd)
|
||||
uint32_t ui, len, cBeg, cEnd;
|
||||
uint32_t *a, an;
|
||||
ma_ug_t *ug;
|
||||
utg_trans_t *o;
|
||||
} utg_trans_hit_idx;
|
||||
|
||||
utg_trans_t *init_utg_trans_t(ma_ug_t *ug, ma_hit_t_alloc* reverse_sources, ma_sub_t *coverage_cut, R_to_U* ruIndex, asg_t *read_g, int max_hang, int min_ovlp);
|
||||
void destroy_utg_trans_t(utg_trans_t **o);
|
||||
void asg_bub_collect_ovlp(ma_ug_t *ug, uint32_t v0, buf_t *b, utg_trans_t *o);
|
||||
void collect_trans_ovlp(const char* cmd, buf_t* pri, uint64_t pri_offset, buf_t* aux, uint64_t aux_offset,
|
||||
ma_ug_t *ug, utg_trans_t *o);
|
||||
int asg_arc_decompress(asg_t *g, ma_ug_t *ug, asg_t *read_sg, ma_hit_t_alloc* reverse_sources,
|
||||
R_to_U* ruIndex, utg_trans_t *o);
|
||||
int asg_arc_decompress_mul(asg_t *g, ma_ug_t *ug, asg_t *read_sg, uint32_t positive_flag, uint32_t negative_flag,
|
||||
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, utg_trans_t *o);
|
||||
kv_u_trans_t *pt_pdist(ma_ug_t *ug, asg_t *read_g, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources,
|
||||
kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp, uint32_t min_chain_cnt);
|
||||
void reset_utg_trans_hit_idx(utg_trans_hit_idx *t, uint32_t* i_x_a, uint32_t i_x_n, ma_ug_t *i_ug,
|
||||
utg_trans_t *i_o, uint32_t i_cBeg, uint32_t i_cEnd);
|
||||
#endif
|
||||
Reference in New Issue
Block a user