mirror of
https://github.com/chhylp123/hifiasm.git
synced 2026-10-12 06:00:56 +08:00
r852 -> hybrid correction
This commit is contained in:
+2217
-2190
File diff suppressed because it is too large
Load Diff
+56
-56
@@ -1,56 +1,56 @@
|
|||||||
#ifndef __ASSEMBLY__
|
#ifndef __ASSEMBLY__
|
||||||
#define __ASSEMBLY__
|
#define __ASSEMBLY__
|
||||||
#include "CommandLines.h"
|
#include "CommandLines.h"
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
#include "Process_Read.h"
|
#include "Process_Read.h"
|
||||||
#include "Hash_Table.h"
|
#include "Hash_Table.h"
|
||||||
#include "Correct.h"
|
#include "Correct.h"
|
||||||
|
|
||||||
#define FORWARD 0
|
#define FORWARD 0
|
||||||
#define REVERSE_COMPLEMENT (0x8000000000000000)
|
#define REVERSE_COMPLEMENT (0x8000000000000000)
|
||||||
|
|
||||||
#define Get_Cigar_Type(RECORD) (RECORD&3)
|
#define Get_Cigar_Type(RECORD) (RECORD&3)
|
||||||
#define Get_Cigar_Length(RECORD) (RECORD>>2)
|
#define Get_Cigar_Length(RECORD) (RECORD>>2)
|
||||||
|
|
||||||
#define RESEED_DP 4
|
#define RESEED_DP 4
|
||||||
#define RESEED_PEAK_RATE 0.15
|
#define RESEED_PEAK_RATE 0.15
|
||||||
#define RESEED_LEN 2000
|
#define RESEED_LEN 2000
|
||||||
#define RESEED_HP_RATE 0.9
|
#define RESEED_HP_RATE 0.9
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int is_final, save_ov;
|
int is_final, save_ov;
|
||||||
// chaining and overlapping related buffers
|
// chaining and overlapping related buffers
|
||||||
UC_Read self_read, ovlp_read;
|
UC_Read self_read, ovlp_read;
|
||||||
Candidates_list clist;
|
Candidates_list clist;
|
||||||
overlap_region_alloc olist;
|
overlap_region_alloc olist;
|
||||||
overlap_region_alloc olist_hp;
|
overlap_region_alloc olist_hp;
|
||||||
ha_abuf_t *ab;
|
ha_abuf_t *ab;
|
||||||
ha_abufl_t *abl;
|
ha_abufl_t *abl;
|
||||||
// error correction related buffers
|
// error correction related buffers
|
||||||
int64_t num_read_base, num_correct_base, num_recorrect_base;
|
int64_t num_read_base, num_correct_base, num_recorrect_base;
|
||||||
Cigar_record cigar1;
|
Cigar_record cigar1;
|
||||||
Graph POA_Graph;
|
Graph POA_Graph;
|
||||||
Graph DAGCon;
|
Graph DAGCon;
|
||||||
Correct_dumy correct;
|
Correct_dumy correct;
|
||||||
haplotype_evdience_alloc hap;
|
haplotype_evdience_alloc hap;
|
||||||
Round2_alignment round2;
|
Round2_alignment round2;
|
||||||
kvec_t_u32_warp b_buf;
|
kvec_t_u32_warp b_buf;
|
||||||
kvec_t_u64_warp r_buf;
|
kvec_t_u64_warp r_buf;
|
||||||
kvec_t_u8_warp k_flag;
|
kvec_t_u8_warp k_flag;
|
||||||
overlap_region tmp_region;
|
overlap_region tmp_region;
|
||||||
ma_utg_v *ua;
|
ma_utg_v *ua;
|
||||||
st_mt_t sp;
|
st_mt_t sp;
|
||||||
bit_extz_t exz;
|
bit_extz_t exz;
|
||||||
} ha_ovec_buf_t;
|
} ha_ovec_buf_t;
|
||||||
|
|
||||||
int ha_assemble(void);
|
int ha_assemble(void);
|
||||||
int ha_assemble_pair(void);
|
int ha_assemble_pair(void);
|
||||||
void ug_idx_build(ma_ug_t *ug, int hap_n);
|
void ug_idx_build(ma_ug_t *ug, int hap_n);
|
||||||
ha_ovec_buf_t *ha_ovec_init(int is_final, int save_ov, int is_ug);
|
ha_ovec_buf_t *ha_ovec_init(int is_final, int save_ov, int is_ug);
|
||||||
ha_ovec_buf_t *ha_ovec_buf_init(void *km, int is_final, int save_ov, int is_ug);
|
ha_ovec_buf_t *ha_ovec_buf_init(void *km, int is_final, int save_ov, int is_ug);
|
||||||
void ha_ovec_destroy(ha_ovec_buf_t *b);
|
void ha_ovec_destroy(ha_ovec_buf_t *b);
|
||||||
int64_t ha_ovec_mem(const ha_ovec_buf_t *b, int64_t *mem_a);
|
int64_t ha_ovec_mem(const ha_ovec_buf_t *b, int64_t *mem_a);
|
||||||
int ha_assemble_ovec(void);
|
int ha_assemble_ovec(void);
|
||||||
int ha_ec_dbg(void);
|
int ha_ec_dbg(void);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
+1121
-1048
File diff suppressed because it is too large
Load Diff
+217
-195
@@ -1,195 +1,217 @@
|
|||||||
#ifndef __COMMAND_LINE_PARSER__
|
#ifndef __COMMAND_LINE_PARSER__
|
||||||
#define __COMMAND_LINE_PARSER__
|
#define __COMMAND_LINE_PARSER__
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <pthread.h>
|
#include <pthread.h>
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
|
|
||||||
#define HA_VERSION "0.25.0-r726"
|
#define HA_VERSION "0.25.0-r852"
|
||||||
|
|
||||||
#define VERBOSE 0
|
#define VERBOSE 0
|
||||||
|
|
||||||
#define HA_F_NO_HPC 0x1
|
#define HA_F_NO_HPC 0x1
|
||||||
#define HA_F_NO_KMER_FLT 0x2
|
#define HA_F_NO_KMER_FLT 0x2
|
||||||
#define HA_F_VERBOSE_GFA 0x4
|
#define HA_F_VERBOSE_GFA 0x4
|
||||||
#define HA_F_WRITE_EC 0x8
|
#define HA_F_WRITE_EC 0x8
|
||||||
#define HA_F_WRITE_PAF 0x10
|
#define HA_F_WRITE_PAF 0x10
|
||||||
#define HA_F_SKIP_TRIOBIN 0x20
|
#define HA_F_SKIP_TRIOBIN 0x20
|
||||||
#define HA_F_PURGE_CONTAIN 0x40
|
#define HA_F_PURGE_CONTAIN 0x40
|
||||||
#define HA_F_PURGE_JOIN 0x80
|
#define HA_F_PURGE_JOIN 0x80
|
||||||
#define HA_F_BAN_POST_JOIN 0x100
|
#define HA_F_BAN_POST_JOIN 0x100
|
||||||
#define HA_F_BAN_ASSEMBLY 0x200
|
#define HA_F_BAN_ASSEMBLY 0x200
|
||||||
#define HA_F_HIGH_HET 0x400
|
#define HA_F_HIGH_HET 0x400
|
||||||
#define HA_F_PARTITION 0x800
|
#define HA_F_PARTITION 0x800
|
||||||
#define HA_F_FAST 0x1000
|
#define HA_F_FAST 0x1000
|
||||||
#define HA_F_USKEW 0x2000
|
#define HA_F_USKEW 0x2000
|
||||||
|
|
||||||
#define HA_MIN_OV_DIFF 0.02 // min sequence divergence in an overlap
|
#define HA_MIN_OV_DIFF 0.02 // min sequence divergence in an overlap
|
||||||
#define MIN_N_CHAIN 100
|
#define MIN_N_CHAIN 100
|
||||||
|
|
||||||
typedef struct{
|
typedef struct{
|
||||||
int *l, n;
|
int *l, n;
|
||||||
char **a;
|
char **a;
|
||||||
}enzyme;
|
}enzyme;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int flag;
|
int flag;
|
||||||
int num_reads;
|
int num_reads;
|
||||||
char** read_file_names;
|
char** read_file_names;
|
||||||
char* output_file_name;
|
char* output_file_name;
|
||||||
char* required_read_name;
|
char* required_read_name;
|
||||||
char *fn_bin_yak[2];
|
char *fn_bin_yak[2];
|
||||||
char *fn_bin_list[2];
|
char *fn_bin_list[2];
|
||||||
char *fn_bin_poy;
|
char *fn_bin_poy;
|
||||||
char *extract_list;
|
char *fn_chr_bin;
|
||||||
enzyme *hic_reads[2];
|
char *extract_list;
|
||||||
enzyme *hic_enzymes;
|
enzyme *hic_reads[2];
|
||||||
enzyme *ar;
|
enzyme *hic_enzymes;
|
||||||
enzyme *sec_in;
|
enzyme *ar;
|
||||||
int extract_iter;
|
enzyme *hf;
|
||||||
int thread_num;
|
enzyme *sec_in;
|
||||||
int k_mer_length;
|
int extract_iter;
|
||||||
int hic_mer_length;
|
int thread_num;
|
||||||
int ul_mer_length;
|
int k_mer_length;
|
||||||
int trans_mer_length;
|
int hic_mer_length;
|
||||||
int bub_mer_length;
|
int ul_mer_length;
|
||||||
int mz_win;
|
int trans_mer_length;
|
||||||
int ul_mz_win;
|
int bub_mer_length;
|
||||||
int trans_win;
|
int mz_win;
|
||||||
int mz_rewin;
|
int ul_mz_win;
|
||||||
int ul_mz_rewin;
|
int trans_win;
|
||||||
int mz_sample_dist;
|
int mz_rewin;
|
||||||
int bf_shift;
|
int ul_mz_rewin;
|
||||||
int max_kmer_cnt;
|
int mz_sample_dist;
|
||||||
double high_factor; // coverage cutoff set to high_factor*hom_cov
|
int bf_shift;
|
||||||
double max_ov_diff_ec;
|
int max_kmer_cnt;
|
||||||
double max_ov_diff_final;
|
double high_factor; // coverage cutoff set to high_factor*hom_cov
|
||||||
int hom_cov;
|
double max_ov_diff_ec;
|
||||||
int het_cov;
|
double max_ov_diff_ec_sec;
|
||||||
int b_low_cov;
|
double max_ov_diff_final;
|
||||||
int b_high_cov;
|
int hom_cov;
|
||||||
double m_rate;
|
int het_cov;
|
||||||
int max_n_chain; // fall-back max number of chains to consider
|
int b_low_cov;
|
||||||
int min_hist_kmer_cnt;
|
int b_high_cov;
|
||||||
int load_index_from_disk;
|
double m_rate;
|
||||||
int write_index_to_disk;
|
int max_n_chain; // fall-back max number of chains to consider
|
||||||
int number_of_round;
|
int min_hist_kmer_cnt;
|
||||||
int number_of_pround;
|
int load_index_from_disk;
|
||||||
int adapterLen;
|
int write_index_to_disk;
|
||||||
int clean_round;
|
int number_of_round;
|
||||||
int roundID;
|
int number_of_pround;
|
||||||
int max_hang_Len;
|
int adapterLen;
|
||||||
int gap_fuzz;
|
int clean_round;
|
||||||
int min_overlap_Len;
|
int roundID;
|
||||||
int min_overlap_coverage;
|
int max_hang_Len;
|
||||||
int max_short_tip;
|
int gap_fuzz;
|
||||||
int max_short_ul_tip;
|
int min_overlap_Len;
|
||||||
int max_contig_tip;
|
int min_overlap_coverage;
|
||||||
int min_cnt;
|
int max_short_tip;
|
||||||
int mid_cnt;
|
int max_short_ul_tip;
|
||||||
int purge_level_primary;
|
int max_contig_tip;
|
||||||
int purge_level_trio;
|
int min_cnt;
|
||||||
int purge_overlap_len;
|
int mid_cnt;
|
||||||
///int purge_overlap_len_hic;
|
int purge_level_primary;
|
||||||
int recover_atg_cov_min;
|
int purge_level_trio;
|
||||||
int recover_atg_cov_max;
|
int purge_overlap_len;
|
||||||
int hom_global_coverage;
|
///int purge_overlap_len_hic;
|
||||||
int hom_global_coverage_set;
|
int recover_atg_cov_min;
|
||||||
int pur_global_coverage;
|
int recover_atg_cov_max;
|
||||||
int bed_inconsist_rate;
|
int hom_global_coverage;
|
||||||
int hic_inconsist_rate;
|
int hom_global_coverage_set;
|
||||||
|
int pur_global_coverage;
|
||||||
float max_hang_rate;
|
int bed_inconsist_rate;
|
||||||
float min_drop_rate;
|
int hic_inconsist_rate;
|
||||||
float max_drop_rate;
|
|
||||||
float purge_simi_rate_l2;
|
float max_hang_rate;
|
||||||
float purge_simi_rate_l3;
|
float min_drop_rate;
|
||||||
float purge_simi_thres;
|
float max_drop_rate;
|
||||||
float trans_base_rate;
|
float purge_simi_rate_l2;
|
||||||
float trans_base_rate_sec;
|
float purge_simi_rate_l3;
|
||||||
float min_path_drop_rate;
|
float purge_simi_thres;
|
||||||
float max_path_drop_rate;
|
float trans_base_rate;
|
||||||
// uint64_t path_clean_round;
|
float trans_base_rate_sec;
|
||||||
|
float min_path_drop_rate;
|
||||||
///float purge_simi_rate_hic;
|
float max_path_drop_rate;
|
||||||
|
// uint64_t path_clean_round;
|
||||||
long long small_pop_bubble_size;
|
|
||||||
long long large_pop_bubble_size;
|
///float purge_simi_rate_hic;
|
||||||
long long num_bases;
|
|
||||||
long long num_corrected_bases;
|
long long small_pop_bubble_size;
|
||||||
long long num_recorrected_bases;
|
long long large_pop_bubble_size;
|
||||||
long long mem_buf;
|
long long num_bases;
|
||||||
long long coverage;
|
long long num_corrected_bases;
|
||||||
int hap_occ;
|
long long num_recorrected_bases;
|
||||||
int polyploidy;
|
long long mem_buf;
|
||||||
int trio_flag_occ_thres;
|
long long coverage;
|
||||||
uint64_t seed;
|
int hap_occ;
|
||||||
int32_t n_perturb;
|
int polyploidy;
|
||||||
double f_perturb;
|
int trio_flag_occ_thres;
|
||||||
int32_t n_weight;
|
uint64_t seed;
|
||||||
uint32_t is_alt;
|
int32_t n_perturb;
|
||||||
uint64_t misjoin_len;
|
double f_perturb;
|
||||||
uint64_t scffold;
|
int32_t n_weight;
|
||||||
int32_t dp_min_len;
|
uint32_t is_alt;
|
||||||
float dp_e;
|
uint64_t misjoin_len;
|
||||||
int64_t hg_size;
|
uint64_t scffold;
|
||||||
float kpt_rate;
|
int32_t dp_min_len;
|
||||||
int64_t infor_cov, s_hap_cov, trio_cov_het_ovlp;
|
float dp_e;
|
||||||
double ul_error_rate, ul_error_rate_low, ul_error_rate_hpc;
|
int64_t hg_size;
|
||||||
int32_t ul_ec_round;
|
float kpt_rate;
|
||||||
int32_t ul_mod;
|
int64_t infor_cov, s_hap_cov, trio_cov_het_ovlp;
|
||||||
uint8_t is_dbg_het_cnt;
|
double ul_error_rate, ul_error_rate_low, ul_error_rate_hpc;
|
||||||
uint8_t is_low_het_ul;
|
int32_t ul_ec_round;
|
||||||
uint8_t is_base_trans;
|
int32_t ul_mod;
|
||||||
uint8_t is_read_trans;
|
uint8_t is_dbg_het_cnt;
|
||||||
uint8_t is_topo_trans;
|
uint8_t is_low_het_ul;
|
||||||
uint8_t is_bub_trans;
|
uint8_t is_base_trans;
|
||||||
uint8_t bin_only;
|
uint8_t is_read_trans;
|
||||||
int32_t ul_clean_round;
|
uint8_t is_topo_trans;
|
||||||
int32_t prt_dbg_gfa;
|
uint8_t is_bub_trans;
|
||||||
int32_t integer_correct_round;
|
uint8_t bin_only;
|
||||||
uint8_t dbg_ovec_cal;
|
int32_t ul_clean_round;
|
||||||
uint8_t hifi_pst_join, ul_pst_join;
|
int32_t prt_dbg_gfa;
|
||||||
uint32_t ul_min_base;
|
int32_t integer_correct_round;
|
||||||
uint8_t self_scaf;
|
uint8_t dbg_ovec_cal;
|
||||||
uint64_t self_scaf_min;
|
uint8_t hifi_pst_join, ul_pst_join;
|
||||||
uint64_t self_scaf_reliable_min;
|
uint32_t ul_min_base;
|
||||||
int64_t self_scaf_gap_max;
|
uint8_t self_scaf;
|
||||||
int64_t somatic_cov;
|
uint64_t self_scaf_min;
|
||||||
|
uint64_t self_scaf_reliable_min;
|
||||||
char *telo_motif;
|
int64_t self_scaf_gap_max;
|
||||||
int64_t telo_pen;
|
int64_t somatic_cov;
|
||||||
int64_t telo_drop;
|
|
||||||
int64_t telo_mic_sc;
|
char *telo_motif;
|
||||||
|
int64_t telo_pen;
|
||||||
uint64_t is_ont;
|
int64_t telo_drop;
|
||||||
uint64_t is_sc;
|
int64_t telo_mic_sc;
|
||||||
uint64_t chemical_cov;
|
|
||||||
uint64_t chemical_flank;
|
uint64_t is_ont;
|
||||||
|
uint64_t is_sc;
|
||||||
int64_t rl_cut;
|
uint64_t chemical_cov;
|
||||||
int64_t sc_cut;
|
uint64_t chemical_flank;
|
||||||
|
|
||||||
} hifiasm_opt_t;
|
int64_t rl_cut;
|
||||||
|
int64_t sc_cut;
|
||||||
extern hifiasm_opt_t asm_opt;
|
uint8_t gpath;
|
||||||
|
|
||||||
void init_opt(hifiasm_opt_t* asm_opt);
|
uint64_t hf_rate;///cannot be larger than 128?
|
||||||
void destory_opt(hifiasm_opt_t* asm_opt);
|
uint64_t hf_rate_max;///cannot be larger than 128?
|
||||||
void ha_opt_reset_to_round(hifiasm_opt_t* asm_opt, int round);
|
uint64_t ont_rate;///cannot be 0, should be 1 in anyway
|
||||||
void ha_opt_update_cov(hifiasm_opt_t *opt, int hom_cov);
|
|
||||||
void ha_opt_update_cov_min(hifiasm_opt_t *opt, int hom_cov, int min_chain);
|
int64_t het_cov_set;
|
||||||
int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt);
|
int64_t restart;
|
||||||
double Get_T(void);
|
|
||||||
|
int64_t hf_cutoff;
|
||||||
static inline int ha_opt_triobin(const hifiasm_opt_t *opt)
|
|
||||||
{
|
uint8_t write_pos_idx;
|
||||||
return ((opt->fn_bin_yak[0] && opt->fn_bin_yak[1]) || (opt->fn_bin_list[0] && opt->fn_bin_list[1]));
|
|
||||||
}
|
int hom_cov_0;
|
||||||
|
int het_cov_0;
|
||||||
static inline int ha_opt_hic(const hifiasm_opt_t *opt)
|
int max_n_chain_0; // fall-back max number of chains to consider
|
||||||
{
|
|
||||||
return ((opt->hic_reads[0] && opt->hic_reads[1]));
|
int64_t hmo_cov_ss;
|
||||||
}
|
int64_t het_cov_ss;
|
||||||
|
int64_t chn_occ;
|
||||||
#endif
|
} hifiasm_opt_t;
|
||||||
|
|
||||||
|
extern hifiasm_opt_t asm_opt;
|
||||||
|
|
||||||
|
void init_opt(hifiasm_opt_t* asm_opt);
|
||||||
|
void destory_opt(hifiasm_opt_t* asm_opt);
|
||||||
|
void ha_opt_reset_to_round(hifiasm_opt_t* asm_opt, int round);
|
||||||
|
void ha_opt_update_cov(hifiasm_opt_t *opt, int hom_cov);
|
||||||
|
void ha_opt_update_cov_min(hifiasm_opt_t *opt, int hom_cov, int min_chain);
|
||||||
|
int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt);
|
||||||
|
double Get_T(void);
|
||||||
|
|
||||||
|
static inline int ha_opt_triobin(const hifiasm_opt_t *opt)
|
||||||
|
{
|
||||||
|
return ((opt->fn_bin_yak[0] && opt->fn_bin_yak[1]) || (opt->fn_bin_list[0] && opt->fn_bin_list[1]));
|
||||||
|
}
|
||||||
|
|
||||||
|
static inline int ha_opt_hic(const hifiasm_opt_t *opt)
|
||||||
|
{
|
||||||
|
return ((opt->hic_reads[0] && opt->hic_reads[1]));
|
||||||
|
}
|
||||||
|
|
||||||
|
#endif
|
||||||
|
|||||||
+34307
-25961
File diff suppressed because it is too large
Load Diff
+2734
-2684
File diff suppressed because it is too large
Load Diff
+265
-260
@@ -1,260 +1,265 @@
|
|||||||
#ifndef __HASHTABLE__
|
#ifndef __HASHTABLE__
|
||||||
#define __HASHTABLE__
|
#define __HASHTABLE__
|
||||||
#include "htab.h"
|
#include "htab.h"
|
||||||
|
|
||||||
#define PREFIX_BITS 16
|
#define PREFIX_BITS 16
|
||||||
#define MAX_SUFFIX_BITS 64
|
#define MAX_SUFFIX_BITS 64
|
||||||
#define MODE_VALUE 101
|
#define MODE_VALUE 101
|
||||||
|
|
||||||
#define WINDOW 375
|
#define WINDOW 375
|
||||||
#define WINDOW_BOUNDARY 375
|
#define WINDOW_BOUNDARY 375
|
||||||
#define WINDOW_HC 775
|
#define WINDOW_HC 775
|
||||||
///ONT high error
|
///ONT high error
|
||||||
// #define WINDOW_OHC 475
|
// #define WINDOW_OHC 475
|
||||||
#define WINDOW_OHC 375
|
#define WINDOW_OHC 375
|
||||||
#define WINDOW_HC_FAST 512
|
#define WINDOW_HC_FAST 512
|
||||||
///for one side, the first or last WINDOW_UNCORRECT_SINGLE_SIDE_BOUNDARY bases should not be corrected
|
///for one side, the first or last WINDOW_UNCORRECT_SINGLE_SIDE_BOUNDARY bases should not be corrected
|
||||||
#define WINDOW_UNCORRECT_SINGLE_SIDE_BOUNDARY 25
|
#define WINDOW_UNCORRECT_SINGLE_SIDE_BOUNDARY 25
|
||||||
#define THRESHOLD 15
|
#define THRESHOLD 15
|
||||||
#define OVERLAP_THRESHOLD_HIFI_FILTER 0.9
|
#define OVERLAP_THRESHOLD_HIFI_FILTER 0.9
|
||||||
#define OVERLAP_THRESHOLD_NOSI_FILTER 0.7
|
#define OVERLAP_THRESHOLD_HIFI_FF_FILTER 0.6
|
||||||
#define OVERLAP_THRESHOLD_FILTER_HPC 0.75
|
#define OVERLAP_THRESHOLD_HIFI_FF_DE_FILTER 0.5
|
||||||
#define HIGH_HET_OVERLAP_THRESHOLD_FILTER 0.3
|
#define OVERLAP_THRESHOLD_NOSI_FILTER 0.7
|
||||||
#define HIGH_HET_ERROR_RATE 0.08
|
#define OVERLAP_THRESHOLD_FILTER_HPC 0.75
|
||||||
#define THRESHOLD_MAX_SIZE 31
|
#define HIGH_HET_OVERLAP_THRESHOLD_FILTER 0.3
|
||||||
#define THRESHOLD_UL_MAX 0.2
|
#define HIGH_HET_ERROR_RATE 0.08
|
||||||
#define WINDOW_UL 75
|
#define THRESHOLD_MAX_SIZE 31
|
||||||
#define WINDOW_UL_H 200
|
#define THRESHOLD_UL_MAX 0.2
|
||||||
// #define WINDOW_UL_H 150
|
#define WINDOW_UL 75
|
||||||
#define MIN_UL_ALIN_RATE 0.5
|
#define WINDOW_UL_H 200
|
||||||
#define MIN_UL_ALIN_LEN (WINDOW_UL*6)
|
// #define WINDOW_UL_H 150
|
||||||
#define WINDOW_UL_BOUND 48
|
#define MIN_UL_ALIN_RATE 0.5
|
||||||
#define WINDOW_UL_BOUND_RATE 0.55
|
#define MIN_UL_ALIN_LEN (WINDOW_UL*6)
|
||||||
|
#define WINDOW_UL_BOUND 48
|
||||||
#define GROUP_SIZE 4
|
#define WINDOW_UL_BOUND_RATE 0.55
|
||||||
///the max cigar likes 10M10D10M10D10M
|
|
||||||
///#define CIGAR_MAX_LENGTH THRESHOLD*2+2
|
#define GROUP_SIZE 4
|
||||||
#define CIGAR_MAX_LENGTH 31*2+4
|
///the max cigar likes 10M10D10M10D10M
|
||||||
|
///#define CIGAR_MAX_LENGTH THRESHOLD*2+2
|
||||||
typedef struct
|
#define CIGAR_MAX_LENGTH 31*2+4
|
||||||
{
|
|
||||||
uint32_t offset;
|
typedef struct
|
||||||
uint32_t readID:31, rev:1;
|
{
|
||||||
} k_mer_pos;
|
uint32_t offset;
|
||||||
|
uint32_t readID:31, rev:1;
|
||||||
typedef struct
|
} k_mer_pos;
|
||||||
{
|
|
||||||
k_mer_pos* list;
|
typedef struct
|
||||||
uint64_t length;
|
{
|
||||||
uint64_t size;
|
k_mer_pos* list;
|
||||||
uint8_t direction;
|
uint64_t length;
|
||||||
uint64_t end_pos;
|
uint64_t size;
|
||||||
} k_mer_pos_list;
|
uint8_t direction;
|
||||||
|
uint64_t end_pos;
|
||||||
typedef struct
|
} k_mer_pos_list;
|
||||||
{
|
|
||||||
///the begining and end of a window, instead of the whole overlap
|
typedef struct
|
||||||
int32_t x_start, x_end;
|
{
|
||||||
int32_t y_start, y_end;
|
///the begining and end of a window, instead of the whole overlap
|
||||||
int16_t extra_begin, extra_end;
|
int32_t x_start, x_end;
|
||||||
int16_t error, error_threshold;
|
int32_t y_start, y_end;
|
||||||
uint32_t cidx, clen;
|
int16_t extra_begin, extra_end;
|
||||||
} window_list;
|
int16_t error, error_threshold;
|
||||||
|
uint32_t cidx, clen;
|
||||||
typedef struct
|
} window_list;
|
||||||
{
|
|
||||||
size_t n, m;
|
typedef struct
|
||||||
window_list *a;
|
{
|
||||||
kvec_t(uint16_t) c;
|
size_t n, m;
|
||||||
} window_list_alloc;
|
window_list *a;
|
||||||
|
kvec_t(uint16_t) c;
|
||||||
typedef struct
|
} window_list_alloc;
|
||||||
{
|
|
||||||
uint64_t* buffer;
|
typedef struct
|
||||||
uint32_t length;
|
{
|
||||||
uint32_t size;
|
uint64_t* buffer;
|
||||||
} Fake_Cigar;
|
uint32_t length;
|
||||||
|
uint32_t size;
|
||||||
typedef struct
|
} Fake_Cigar;
|
||||||
{
|
|
||||||
uint32_t x_id;
|
typedef struct
|
||||||
///the begining and end of the whole overlap
|
{
|
||||||
uint32_t x_pos_s;
|
uint32_t x_id;
|
||||||
uint32_t x_pos_e;
|
///the begining and end of the whole overlap
|
||||||
uint32_t x_pos_strand;
|
uint32_t x_pos_s;
|
||||||
|
uint32_t x_pos_e;
|
||||||
uint32_t y_id;
|
uint32_t x_pos_strand;
|
||||||
uint32_t y_pos_s;
|
|
||||||
uint32_t y_pos_e;
|
uint32_t y_id;
|
||||||
uint32_t y_pos_strand;
|
uint32_t y_pos_s;
|
||||||
|
uint32_t y_pos_e;
|
||||||
uint32_t overlapLen;
|
uint32_t y_pos_strand;
|
||||||
int32_t shared_seed;
|
|
||||||
uint32_t align_length;
|
uint32_t overlapLen;
|
||||||
uint8_t is_match;
|
int32_t shared_seed;
|
||||||
uint8_t without_large_indel;
|
uint32_t align_length;
|
||||||
int8_t strong;
|
uint8_t is_match;
|
||||||
uint32_t non_homopolymer_errors;
|
uint8_t without_large_indel;
|
||||||
|
int8_t strong;
|
||||||
// window_list* w_list;
|
uint32_t non_homopolymer_errors;
|
||||||
// uint32_t w_list_size;
|
|
||||||
// uint32_t w_list_length;
|
// window_list* w_list;
|
||||||
Fake_Cigar f_cigar;
|
// uint32_t w_list_size;
|
||||||
|
// uint32_t w_list_length;
|
||||||
window_list_alloc w_list;
|
Fake_Cigar f_cigar;
|
||||||
window_list_alloc boundary_cigars;
|
|
||||||
} overlap_region;
|
window_list_alloc w_list;
|
||||||
|
window_list_alloc boundary_cigars;
|
||||||
typedef struct
|
} overlap_region;
|
||||||
{
|
|
||||||
overlap_region* list;
|
typedef struct
|
||||||
uint64_t size;
|
{
|
||||||
uint64_t length;
|
overlap_region* list;
|
||||||
int64_t mapped_overlaps_length;
|
uint64_t size;
|
||||||
} overlap_region_alloc;
|
uint64_t length;
|
||||||
|
int64_t mapped_overlaps_length;
|
||||||
typedef struct
|
} overlap_region_alloc;
|
||||||
{
|
|
||||||
uint32_t readID:31, strand:1;
|
typedef struct
|
||||||
uint32_t offset, self_offset, cnt;
|
{
|
||||||
} k_mer_hit;
|
uint32_t readID:31, strand:1;
|
||||||
|
uint32_t offset, self_offset, cnt;
|
||||||
typedef struct {
|
} k_mer_hit;
|
||||||
int32_t *score;
|
|
||||||
int64_t *pre;
|
typedef struct {
|
||||||
int32_t *indels;
|
int32_t *score;
|
||||||
int32_t *self_length;
|
int64_t *pre;
|
||||||
int32_t *occ;
|
int32_t *indels;
|
||||||
int64_t *tmp; // MUST BE 64-bit integer
|
int32_t *self_length;
|
||||||
int64_t length;
|
int32_t *occ;
|
||||||
int64_t size;
|
int64_t *tmp; // MUST BE 64-bit integer
|
||||||
} Chain_Data;
|
int64_t length;
|
||||||
|
int64_t size;
|
||||||
typedef struct
|
} Chain_Data;
|
||||||
{
|
|
||||||
k_mer_hit* list;
|
typedef struct
|
||||||
long long length;
|
{
|
||||||
long long size;
|
k_mer_hit* list;
|
||||||
Chain_Data chainDP;
|
long long length;
|
||||||
} Candidates_list;
|
long long size;
|
||||||
|
Chain_Data chainDP;
|
||||||
void init_Candidates_list(Candidates_list* l);
|
} Candidates_list;
|
||||||
void clear_Candidates_list(Candidates_list* l);
|
|
||||||
void destory_Candidates_list(Candidates_list* l);
|
void init_Candidates_list(Candidates_list* l);
|
||||||
void destory_Candidates_list_buf(void *km, Candidates_list* l, int is_z);
|
void clear_Candidates_list(Candidates_list* l);
|
||||||
|
void destory_Candidates_list(Candidates_list* l);
|
||||||
void init_overlap_region_alloc(overlap_region_alloc* list);
|
void destory_Candidates_list_buf(void *km, Candidates_list* l, int is_z);
|
||||||
void clear_overlap_region_alloc(overlap_region_alloc* list);
|
|
||||||
void destory_overlap_region_alloc(overlap_region_alloc* list);
|
void init_overlap_region_alloc(overlap_region_alloc* list);
|
||||||
void append_window_list(overlap_region* region, uint64_t x_start, uint64_t x_end, int y_start, int y_end, int error,
|
void clear_overlap_region_alloc(overlap_region_alloc* list);
|
||||||
int extra_begin, int extra_end, int error_threshold, int blockLen, void *km);
|
void destory_overlap_region_alloc(overlap_region_alloc* list);
|
||||||
|
void append_window_list(overlap_region* region, uint64_t x_start, uint64_t x_end, int y_start, int y_end, int error,
|
||||||
void overlap_region_sort_y_id(overlap_region *a, long long n);
|
int extra_begin, int extra_end, int error_threshold, int blockLen, void *km);
|
||||||
|
|
||||||
void calculate_overlap_region_by_chaining(Candidates_list* candidates, overlap_region_alloc* overlap_list, kvec_t_u64_warp* chain_idx,
|
void overlap_region_sort_y_id(overlap_region *a, long long n);
|
||||||
uint64_t readID, uint64_t readLength, All_reads* R_INF, const ul_idx_t *uref, double band_width_threshold, int add_beg_end, overlap_region* f_cigar, void *km);
|
|
||||||
|
void calculate_overlap_region_by_chaining(Candidates_list* candidates, overlap_region_alloc* overlap_list, kvec_t_u64_warp* chain_idx,
|
||||||
void init_fake_cigar(Fake_Cigar* x);
|
uint64_t readID, uint64_t readLength, All_reads* R_INF, const ul_idx_t *uref, double band_width_threshold, int add_beg_end, overlap_region* f_cigar, void *km);
|
||||||
void destory_fake_cigar(Fake_Cigar* x);
|
|
||||||
void clear_fake_cigar(Fake_Cigar* x);
|
void init_fake_cigar(Fake_Cigar* x);
|
||||||
void add_fake_cigar(Fake_Cigar* x, uint32_t gap_site, int32_t gap_shift, void *km);
|
void destory_fake_cigar(Fake_Cigar* x);
|
||||||
void resize_fake_cigar(Fake_Cigar* x, uint64_t size, void *km);
|
void clear_fake_cigar(Fake_Cigar* x);
|
||||||
int get_fake_gap_pos(Fake_Cigar* x, int index);
|
void add_fake_cigar(Fake_Cigar* x, uint32_t gap_site, int32_t gap_shift, void *km);
|
||||||
int get_fake_gap_shift(Fake_Cigar* x, int index);
|
void resize_fake_cigar(Fake_Cigar* x, uint64_t size, void *km);
|
||||||
|
int get_fake_gap_pos(Fake_Cigar* x, int index);
|
||||||
static inline long long y_start_offset(long long x_start, Fake_Cigar* o)
|
int get_fake_gap_shift(Fake_Cigar* x, int index);
|
||||||
{
|
|
||||||
if(x_start == get_fake_gap_pos(o, o->length - 1))
|
static inline long long y_start_offset(long long x_start, Fake_Cigar* o)
|
||||||
{
|
{
|
||||||
return get_fake_gap_shift(o, o->length - 1);
|
if(x_start == get_fake_gap_pos(o, o->length - 1))
|
||||||
}
|
{
|
||||||
|
return get_fake_gap_shift(o, o->length - 1);
|
||||||
long long i;
|
}
|
||||||
for (i = 0; i < (long long)o->length; i++)
|
|
||||||
{
|
long long i;
|
||||||
if(x_start < get_fake_gap_pos(o, i))
|
for (i = 0; i < (long long)o->length; i++)
|
||||||
{
|
{
|
||||||
break;
|
if(x_start < get_fake_gap_pos(o, i))
|
||||||
}
|
{
|
||||||
}
|
break;
|
||||||
|
}
|
||||||
if(i == 0 || i == (long long)o->length)
|
}
|
||||||
{
|
|
||||||
fprintf(stderr, "ERROR at %s:%d\n", __FILE__, __LINE__);
|
if(i == 0 || i == (long long)o->length)
|
||||||
exit(0);
|
{
|
||||||
}
|
fprintf(stderr, "ERROR at %s:%d\n", __FILE__, __LINE__);
|
||||||
|
exit(0);
|
||||||
///note here return i - 1
|
}
|
||||||
return get_fake_gap_shift(o, i - 1);
|
|
||||||
}
|
///note here return i - 1
|
||||||
|
return get_fake_gap_shift(o, i - 1);
|
||||||
void resize_Chain_Data(Chain_Data* x, long long size, void *km);
|
}
|
||||||
void init_window_list_alloc(window_list_alloc* x);
|
|
||||||
void clear_window_list_alloc(window_list_alloc* x);
|
void resize_Chain_Data(Chain_Data* x, long long size, void *km);
|
||||||
void destory_window_list_alloc(window_list_alloc* x);
|
void init_window_list_alloc(window_list_alloc* x);
|
||||||
void resize_window_list_alloc(window_list_alloc* x, uint64_t size);
|
void clear_window_list_alloc(window_list_alloc* x);
|
||||||
long long chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result, double band_width_threshold, int max_skip, int x_readLen, int y_readLen, void *km);
|
void destory_window_list_alloc(window_list_alloc* x);
|
||||||
uint64_t lchain_dp(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp, overlap_region* res,
|
void resize_window_list_alloc(window_list_alloc* x, uint64_t size);
|
||||||
int64_t max_skip, int64_t max_iter, int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
long long chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result, double band_width_threshold, int max_skip, int x_readLen, int y_readLen, void *km);
|
||||||
int64_t xl, int64_t yl, int64_t quick_check);
|
uint64_t lchain_dp(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp, overlap_region* res,
|
||||||
int ovlp_chain_gen(overlap_region_alloc* ol, overlap_region* t, int64_t xl, int64_t yl, int64_t apend_be, k_mer_hit* hit, int64_t n_hit);
|
int64_t max_skip, int64_t max_iter, int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
||||||
void gen_fake_cigar(Fake_Cigar* z, overlap_region *o, int64_t apend_be, k_mer_hit* hit, int64_t n_hit);
|
int64_t xl, int64_t yl, int64_t quick_check);
|
||||||
int append_utg_inexact_overlap_region_alloc(overlap_region_alloc* list, overlap_region* tmp,
|
int ovlp_chain_gen(overlap_region_alloc* ol, overlap_region* t, int64_t xl, int64_t yl, int64_t apend_be, k_mer_hit* hit, int64_t n_hit);
|
||||||
ma_utg_v *ua, int add_beg_end, void *km);
|
void gen_fake_cigar(Fake_Cigar* z, overlap_region *o, int64_t apend_be, k_mer_hit* hit, int64_t n_hit);
|
||||||
|
int append_utg_inexact_overlap_region_alloc(overlap_region_alloc* list, overlap_region* tmp,
|
||||||
#define kv_pushp_cl(type, v, p) do { \
|
ma_utg_v *ua, int add_beg_end, void *km);
|
||||||
if ((v).length == (v).size) { \
|
|
||||||
(v).size = (v).size? (v).size<<1 : 2; \
|
#define kv_pushp_cl(type, v, p) do { \
|
||||||
(v).list = (type*)realloc((v).list, sizeof(type) * (v).size); \
|
if ((v).length == (v).size) { \
|
||||||
} \
|
(v).size = (v).size? (v).size<<1 : 2; \
|
||||||
*(p) = &((v).list[(v).length++]); \
|
(v).list = (type*)realloc((v).list, sizeof(type) * (v).size); \
|
||||||
} while (0)
|
} \
|
||||||
|
*(p) = &((v).list[(v).length++]); \
|
||||||
#define kv_resize_cl(type, v, s) do { \
|
} while (0)
|
||||||
if ((v).size < (s)) { \
|
|
||||||
(v).size = (s); \
|
#define kv_resize_cl(type, v, s) do { \
|
||||||
kv_roundup32((v).size); \
|
if ((v).size < (s)) { \
|
||||||
(v).list = (type*)realloc((v).list, sizeof(type) * (v).size); \
|
(v).size = (s); \
|
||||||
} \
|
kv_roundup32((v).size); \
|
||||||
} while (0)
|
(v).list = (type*)realloc((v).list, sizeof(type) * (v).size); \
|
||||||
|
} \
|
||||||
#define is_alnw(a) (((a).readID) == ((uint32_t)(0x7fffffff)))
|
} while (0)
|
||||||
#define is_pri_aln(a) ((((a).readID) == ((uint32_t)(0x7fffffff)))||((a).cnt >= (a).readID))
|
|
||||||
|
#define is_alnw(a) (((a).readID) == ((uint32_t)(0x7fffffff)))
|
||||||
uint64_t lchain_dp_trace(k_mer_hit* a, int64_t a_n, int64_t max_lgap, double sgap_rate, int64_t sgap);
|
#define is_pri_aln(a) ((((a).readID) == ((uint32_t)(0x7fffffff)))||((a).cnt >= (a).readID))
|
||||||
uint64_t lchain_qdp(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp, overlap_region* res,
|
|
||||||
int64_t max_skip, int64_t max_iter, int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
uint64_t lchain_dp_trace(k_mer_hit* a, int64_t a_n, int64_t max_lgap, double sgap_rate, int64_t sgap);
|
||||||
int64_t xl, int64_t yl, int64_t quick_check);
|
uint64_t lchain_qdp(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp, overlap_region* res,
|
||||||
int ovlp_chain_qgen(overlap_region_alloc* ol, overlap_region* t, int64_t xl, int64_t yl, int64_t apend_be, k_mer_hit* hit, int64_t n_hit);
|
int64_t max_skip, int64_t max_iter, int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
||||||
uint64_t lchain_refine(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp,
|
int64_t xl, int64_t yl, int64_t quick_check);
|
||||||
int64_t max_skip, int64_t max_iter, int64_t max_dis, int64_t long_gap);
|
int ovlp_chain_qgen(overlap_region_alloc* ol, overlap_region* t, int64_t xl, int64_t yl, int64_t apend_be, k_mer_hit* hit, int64_t n_hit);
|
||||||
uint64_t lchain_qdp_fix(k_mer_hit* a, int64_t a_n, Chain_Data* dp, int64_t max_skip,
|
uint64_t lchain_refine(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp,
|
||||||
int64_t max_iter, int64_t max_dis, double chn_pen_gap, double chn_pen_skip,
|
int64_t max_skip, int64_t max_iter, int64_t max_dis, int64_t long_gap);
|
||||||
double bw_rate, int64_t xl, int64_t yl, int64_t quick_check,
|
uint64_t lchain_qdp_fix(k_mer_hit* a, int64_t a_n, Chain_Data* dp, int64_t max_skip,
|
||||||
int64_t left_fix, int64_t right_fix);
|
int64_t max_iter, int64_t max_dis, double chn_pen_gap, double chn_pen_skip,
|
||||||
uint64_t lchain_simple(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp,
|
double bw_rate, int64_t xl, int64_t yl, int64_t quick_check,
|
||||||
int64_t max_skip, int64_t max_iter);
|
int64_t left_fix, int64_t right_fix);
|
||||||
uint64_t lchain_qdp_mcopy(Candidates_list *cl, int64_t a_idx, int64_t a_n, int64_t des_idx,
|
uint64_t lchain_simple(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp,
|
||||||
Chain_Data* dp, overlap_region_alloc* res, int64_t max_skip, int64_t max_iter,
|
int64_t max_skip, int64_t max_iter);
|
||||||
int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
uint64_t lchain_simple0(k_mer_hit* a, int64_t a_n, k_mer_hit* des, Chain_Data* dp, int64_t max_skip, int64_t max_iter);
|
||||||
uint32_t xid, int64_t xl, int64_t yl, int64_t quick_check, uint32_t apend_be,
|
void push_ovlp_chain_qgen(overlap_region* o, uint32_t xid, int64_t xl, int64_t yl, int64_t sc, k_mer_hit *beg, k_mer_hit *end);
|
||||||
int64_t gen_cigar, int64_t enable_mcopy, double mcopy_rate, int64_t mcopy_khit_cutoff,
|
|
||||||
int64_t khit_n);
|
uint64_t lchain_qdp_mcopy(Candidates_list *cl, int64_t a_idx, int64_t a_n, int64_t des_idx,
|
||||||
|
Chain_Data* dp, overlap_region_alloc* res, int64_t max_skip, int64_t max_iter,
|
||||||
uint64_t lchain_qdp_mcopy_fast(Candidates_list *cl, int64_t a_idx, int64_t a_n, int64_t des_idx,
|
int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
||||||
Chain_Data* dp, overlap_region_alloc* res, int64_t max_skip, int64_t max_iter,
|
uint32_t xid, int64_t xl, int64_t yl, int64_t quick_check, uint32_t apend_be,
|
||||||
int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
int64_t gen_cigar, int64_t enable_mcopy, double mcopy_rate, int64_t mcopy_khit_cutoff,
|
||||||
uint32_t xid, int64_t xl, int64_t yl, int64_t quick_check, uint32_t apend_be,
|
int64_t khit_n);
|
||||||
int64_t gen_cigar, int64_t enable_mcopy, double mcopy_rate, int64_t mcopy_khit_cutoff,
|
|
||||||
int64_t khit_n);
|
uint64_t lchain_qdp_mcopy_fast(Candidates_list *cl, int64_t a_idx, int64_t a_n, int64_t des_idx,
|
||||||
|
Chain_Data* dp, overlap_region_alloc* res, int64_t max_skip, int64_t max_iter,
|
||||||
#define kv_pushp_ol(type, v, p) do { \
|
int64_t max_dis, double chn_pen_gap, double chn_pen_skip, double bw_rate,
|
||||||
if ((v).length == (v).size) { \
|
uint32_t xid, int64_t xl, int64_t yl, int64_t quick_check, uint32_t apend_be,
|
||||||
(v).list = (type*)realloc((v).list, sizeof(type)*((v).size?((v).size<<1):(2))); \
|
int64_t gen_cigar, int64_t enable_mcopy, double mcopy_rate, int64_t mcopy_khit_cutoff,
|
||||||
memset((v).list+(v).size, 0, sizeof(overlap_region)*(((v).size?((v).size<<1):2)-(v).size));\
|
int64_t khit_n);
|
||||||
(v).size = (v).size?((v).size<<1):(2); \
|
|
||||||
} \
|
#define kv_pushp_ol(type, v, p) do { \
|
||||||
*(p) = &((v).list[(v).length++]); \
|
if ((v).length == (v).size) { \
|
||||||
} while (0)
|
(v).list = (type*)realloc((v).list, sizeof(type)*((v).size?((v).size<<1):(2))); \
|
||||||
|
memset((v).list+(v).size, 0, sizeof(overlap_region)*(((v).size?((v).size<<1):2)-(v).size));\
|
||||||
#endif
|
(v).size = (v).size?((v).size<<1):(2); \
|
||||||
|
} \
|
||||||
|
*(p) = &((v).list[(v).length++]); \
|
||||||
|
} while (0)
|
||||||
|
|
||||||
|
#endif
|
||||||
|
|||||||
@@ -1,21 +1,21 @@
|
|||||||
MIT License
|
MIT License
|
||||||
|
|
||||||
Copyright (c) 2019
|
Copyright (c) 2019
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
of this software and associated documentation files (the "Software"), to deal
|
of this software and associated documentation files (the "Software"), to deal
|
||||||
in the Software without restriction, including without limitation the rights
|
in the Software without restriction, including without limitation the rights
|
||||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
copies of the Software, and to permit persons to whom the Software is
|
copies of the Software, and to permit persons to whom the Software is
|
||||||
furnished to do so, subject to the following conditions:
|
furnished to do so, subject to the following conditions:
|
||||||
|
|
||||||
The above copyright notice and this permission notice shall be included in all
|
The above copyright notice and this permission notice shall be included in all
|
||||||
copies or substantial portions of the Software.
|
copies or substantial portions of the Software.
|
||||||
|
|
||||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
SOFTWARE.
|
SOFTWARE.
|
||||||
|
|||||||
@@ -1 +1 @@
|
|||||||
#include "Levenshtein_distance.h"
|
#include "Levenshtein_distance.h"
|
||||||
|
|||||||
+5052
-5042
File diff suppressed because it is too large
Load Diff
@@ -1,83 +1,83 @@
|
|||||||
CXX= g++
|
CXX= g++
|
||||||
CC= gcc
|
CC= gcc
|
||||||
CXXFLAGS= -g -O3 -msse4.2 -mpopcnt -fomit-frame-pointer -Wall
|
CXXFLAGS= -g -O3 -msse4.2 -mpopcnt -fomit-frame-pointer -Wall
|
||||||
CFLAGS= $(CXXFLAGS)
|
CFLAGS= $(CXXFLAGS)
|
||||||
CPPFLAGS=
|
CPPFLAGS=
|
||||||
INCLUDES=
|
INCLUDES=
|
||||||
OBJS= CommandLines.o Process_Read.o Assembly.o Hash_Table.o \
|
OBJS= CommandLines.o Process_Read.o Assembly.o Hash_Table.o \
|
||||||
POA.o Correct.o Levenshtein_distance.o Overlaps.o Trio.o kthread.o Purge_Dups.o \
|
POA.o Correct.o Levenshtein_distance.o Overlaps.o Trio.o kthread.o Purge_Dups.o \
|
||||||
htab.o hist.o sketch.o anchor.o extract.o sys.o hic.o rcut.o horder.o ecovlp.o\
|
htab.o hist.o sketch.o anchor.o extract.o sys.o hic.o rcut.o horder.o ecovlp.o\
|
||||||
tovlp.o inter.o kalloc.o gfa_ut.o gchain_map.o
|
tovlp.o inter.o kalloc.o gfa_ut.o gchain_map.o
|
||||||
EXE= hifiasm
|
EXE= hifiasm
|
||||||
LIBS= -lz -lpthread -lm
|
LIBS= -lz -lpthread -lm
|
||||||
|
|
||||||
ifneq ($(asan),)
|
ifneq ($(asan),)
|
||||||
CXXFLAGS+=-fsanitize=address
|
CXXFLAGS+=-fsanitize=address
|
||||||
LIBS+=-fsanitize=address
|
LIBS+=-fsanitize=address
|
||||||
endif
|
endif
|
||||||
|
|
||||||
.SUFFIXES:.cpp .c .o
|
.SUFFIXES:.cpp .c .o
|
||||||
.PHONY:all clean depend
|
.PHONY:all clean depend
|
||||||
|
|
||||||
.cpp.o:
|
.cpp.o:
|
||||||
$(CXX) -c $(CXXFLAGS) $(CPPFLAGS) $(INCLUDES) $< -o $@
|
$(CXX) -c $(CXXFLAGS) $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||||
|
|
||||||
.c.o:
|
.c.o:
|
||||||
$(CC) -c $(CFLAGS) $(CPPFLAGS) $(INCLUDES) $< -o $@
|
$(CC) -c $(CFLAGS) $(CPPFLAGS) $(INCLUDES) $< -o $@
|
||||||
|
|
||||||
all:$(EXE)
|
all:$(EXE)
|
||||||
|
|
||||||
$(EXE):$(OBJS) main.o
|
$(EXE):$(OBJS) main.o
|
||||||
$(CXX) $(CXXFLAGS) $^ -o $@ $(LIBS)
|
$(CXX) $(CXXFLAGS) $^ -o $@ $(LIBS)
|
||||||
|
|
||||||
clean:
|
clean:
|
||||||
rm -fr gmon.out *.o a.out $(EXE) *~ *.a *.dSYM
|
rm -fr gmon.out *.o a.out $(EXE) *~ *.a *.dSYM
|
||||||
|
|
||||||
depend:
|
depend:
|
||||||
(LC_ALL=C; export LC_ALL; makedepend -Y -- $(CPPFLAGS) $(DFLAGS) -- *.cpp)
|
(LC_ALL=C; export LC_ALL; makedepend -Y -- $(CPPFLAGS) $(DFLAGS) -- *.cpp)
|
||||||
|
|
||||||
# DO NOT DELETE
|
# DO NOT DELETE
|
||||||
|
|
||||||
Assembly.o: Assembly.h CommandLines.h Process_Read.h Overlaps.h kvec.h kdq.h
|
Assembly.o: Assembly.h CommandLines.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||||
Assembly.o: Hash_Table.h htab.h POA.h Correct.h Levenshtein_distance.h
|
Assembly.o: Hash_Table.h htab.h POA.h Correct.h Levenshtein_distance.h
|
||||||
Assembly.o: kthread.h ecovlp.h
|
Assembly.o: kthread.h ecovlp.h
|
||||||
CommandLines.o: CommandLines.h ketopt.h
|
CommandLines.o: CommandLines.h ketopt.h
|
||||||
Correct.o: Correct.h Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h
|
Correct.o: Correct.h Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h
|
||||||
Correct.o: kdq.h CommandLines.h Levenshtein_distance.h POA.h Assembly.h
|
Correct.o: kdq.h CommandLines.h Levenshtein_distance.h POA.h Assembly.h
|
||||||
Correct.o: ksort.h
|
Correct.o: ksort.h
|
||||||
Hash_Table.o: Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h kdq.h
|
Hash_Table.o: Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||||
Hash_Table.o: CommandLines.h ksort.h
|
Hash_Table.o: CommandLines.h ksort.h
|
||||||
Levenshtein_distance.o: Levenshtein_distance.h
|
Levenshtein_distance.o: Levenshtein_distance.h
|
||||||
Output.o: Output.h CommandLines.h
|
Output.o: Output.h CommandLines.h
|
||||||
Overlaps.o: Overlaps.h kvec.h kdq.h ksort.h Process_Read.h CommandLines.h
|
Overlaps.o: Overlaps.h kvec.h kdq.h ksort.h Process_Read.h CommandLines.h
|
||||||
Overlaps.o: Hash_Table.h htab.h Correct.h Levenshtein_distance.h POA.h
|
Overlaps.o: Hash_Table.h htab.h Correct.h Levenshtein_distance.h POA.h
|
||||||
Overlaps.o: Purge_Dups.h
|
Overlaps.o: Purge_Dups.h
|
||||||
POA.o: POA.h Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h kdq.h
|
POA.o: POA.h Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||||
POA.o: CommandLines.h Correct.h Levenshtein_distance.h
|
POA.o: CommandLines.h Correct.h Levenshtein_distance.h
|
||||||
Process_Read.o: Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
Process_Read.o: Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||||
Purge_Dups.o: ksort.h Purge_Dups.h kvec.h kdq.h Overlaps.h Hash_Table.h
|
Purge_Dups.o: ksort.h Purge_Dups.h kvec.h kdq.h Overlaps.h Hash_Table.h
|
||||||
Purge_Dups.o: htab.h Process_Read.h CommandLines.h Correct.h
|
Purge_Dups.o: htab.h Process_Read.h CommandLines.h Correct.h
|
||||||
Purge_Dups.o: Levenshtein_distance.h POA.h kthread.h
|
Purge_Dups.o: Levenshtein_distance.h POA.h kthread.h
|
||||||
ecovlp.o: Hash_Table.h Process_Read.h Overlaps.h kthread.h
|
ecovlp.o: Hash_Table.h Process_Read.h Overlaps.h kthread.h
|
||||||
Trio.o: khashl.h kthread.h kseq.h Process_Read.h Overlaps.h kvec.h kdq.h
|
Trio.o: khashl.h kthread.h kseq.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||||
Trio.o: CommandLines.h htab.h
|
Trio.o: CommandLines.h htab.h
|
||||||
anchor.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
anchor.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||||
anchor.o: ksort.h Hash_Table.h
|
anchor.o: ksort.h Hash_Table.h
|
||||||
extract.o: Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h khashl.h
|
extract.o: Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h khashl.h
|
||||||
extract.o: kseq.h
|
extract.o: kseq.h
|
||||||
hist.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
hist.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||||
htab.o: kthread.h khashl.h kseq.h ksort.h htab.h Process_Read.h Overlaps.h
|
htab.o: kthread.h khashl.h kseq.h ksort.h htab.h Process_Read.h Overlaps.h
|
||||||
htab.o: kvec.h kdq.h CommandLines.h
|
htab.o: kvec.h kdq.h CommandLines.h
|
||||||
kthread.o: kthread.h
|
kthread.o: kthread.h
|
||||||
main.o: CommandLines.h Process_Read.h Overlaps.h kvec.h kdq.h Assembly.h
|
main.o: CommandLines.h Process_Read.h Overlaps.h kvec.h kdq.h Assembly.h
|
||||||
main.o: Levenshtein_distance.h htab.h
|
main.o: Levenshtein_distance.h htab.h
|
||||||
sketch.o: kvec.h htab.h Process_Read.h Overlaps.h kdq.h CommandLines.h
|
sketch.o: kvec.h htab.h Process_Read.h Overlaps.h kdq.h CommandLines.h
|
||||||
sys.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
sys.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||||
hic.o: hic.h
|
hic.o: hic.h
|
||||||
rcut.o: rcut.h
|
rcut.o: rcut.h
|
||||||
horder.o: horder.h
|
horder.o: horder.h
|
||||||
tovlp.o: tovlp.h
|
tovlp.o: tovlp.h
|
||||||
inter.o: inter.h Process_Read.h
|
inter.o: inter.h Process_Read.h
|
||||||
kalloc.o: kalloc.h
|
kalloc.o: kalloc.h
|
||||||
gfa_ut.o: Overlaps.h
|
gfa_ut.o: Overlaps.h
|
||||||
gchain_map.o: gchain_map.h
|
gchain_map.o: gchain_map.h
|
||||||
+234
-234
@@ -1,235 +1,235 @@
|
|||||||
#include "Output.h"
|
#include "Output.h"
|
||||||
#include "CommandLines.h"
|
#include "CommandLines.h"
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
|
|
||||||
pthread_mutex_t o_queueMutex;
|
pthread_mutex_t o_queueMutex;
|
||||||
pthread_cond_t o_flushCond;
|
pthread_cond_t o_flushCond;
|
||||||
pthread_cond_t o_stallCond;
|
pthread_cond_t o_stallCond;
|
||||||
pthread_mutex_t o_doneMutex;
|
pthread_mutex_t o_doneMutex;
|
||||||
|
|
||||||
|
|
||||||
Output_buffer buffer_out;
|
Output_buffer buffer_out;
|
||||||
Output_buffer_sub_block tmp_buffer_sub_block;
|
Output_buffer_sub_block tmp_buffer_sub_block;
|
||||||
|
|
||||||
void init_buffer_sub_block(Output_buffer_sub_block* sub_block)
|
void init_buffer_sub_block(Output_buffer_sub_block* sub_block)
|
||||||
{
|
{
|
||||||
sub_block->length = 0;
|
sub_block->length = 0;
|
||||||
sub_block->size = SUB_BLOCK_INIT_SIZE;
|
sub_block->size = SUB_BLOCK_INIT_SIZE;
|
||||||
sub_block->buffer = (char*)malloc(sub_block->size);
|
sub_block->buffer = (char*)malloc(sub_block->size);
|
||||||
}
|
}
|
||||||
|
|
||||||
void destory_buffer_sub_block(Output_buffer_sub_block* sub_block)
|
void destory_buffer_sub_block(Output_buffer_sub_block* sub_block)
|
||||||
{
|
{
|
||||||
free(sub_block->buffer);
|
free(sub_block->buffer);
|
||||||
}
|
}
|
||||||
|
|
||||||
void destory_output_buffer()
|
void destory_output_buffer()
|
||||||
{
|
{
|
||||||
for (int i = 0; i < buffer_out.sub_block_size; i++)
|
for (int i = 0; i < buffer_out.sub_block_size; i++)
|
||||||
{
|
{
|
||||||
destory_buffer_sub_block(&(buffer_out.sub_buffer[i]));
|
destory_buffer_sub_block(&(buffer_out.sub_buffer[i]));
|
||||||
}
|
}
|
||||||
|
|
||||||
free(buffer_out.sub_buffer);
|
free(buffer_out.sub_buffer);
|
||||||
}
|
}
|
||||||
|
|
||||||
void init_output_buffer(int thread_number)
|
void init_output_buffer(int thread_number)
|
||||||
{
|
{
|
||||||
buffer_out.sub_block_size = OUTPUT_BUFFER_SIZE * thread_number;
|
buffer_out.sub_block_size = OUTPUT_BUFFER_SIZE * thread_number;
|
||||||
buffer_out.sub_block_number = 0;
|
buffer_out.sub_block_number = 0;
|
||||||
|
|
||||||
buffer_out.sub_buffer = (Output_buffer_sub_block*)malloc(sizeof(Output_buffer_sub_block)*buffer_out.sub_block_size);
|
buffer_out.sub_buffer = (Output_buffer_sub_block*)malloc(sizeof(Output_buffer_sub_block)*buffer_out.sub_block_size);
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
for (int i = 0; i < buffer_out.sub_block_size; i++)
|
for (int i = 0; i < buffer_out.sub_block_size; i++)
|
||||||
{
|
{
|
||||||
init_buffer_sub_block(&(buffer_out.sub_buffer[i]));
|
init_buffer_sub_block(&(buffer_out.sub_buffer[i]));
|
||||||
}
|
}
|
||||||
|
|
||||||
buffer_out.all_buffer_end = 0;
|
buffer_out.all_buffer_end = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
inline int if_empty_buffer()
|
inline int if_empty_buffer()
|
||||||
{
|
{
|
||||||
|
|
||||||
if (buffer_out.sub_block_number == 0)
|
if (buffer_out.sub_block_number == 0)
|
||||||
{
|
{
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
inline int if_full_buffer()
|
inline int if_full_buffer()
|
||||||
{
|
{
|
||||||
|
|
||||||
if (buffer_out.sub_block_number >= buffer_out.sub_block_size)
|
if (buffer_out.sub_block_number >= buffer_out.sub_block_size)
|
||||||
{
|
{
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
inline void pop_single_buffer(Output_buffer_sub_block* curr_sub_block)
|
inline void pop_single_buffer(Output_buffer_sub_block* curr_sub_block)
|
||||||
{
|
{
|
||||||
buffer_out.sub_block_number--;
|
buffer_out.sub_block_number--;
|
||||||
char *k;
|
char *k;
|
||||||
k = buffer_out.sub_buffer[buffer_out.sub_block_number].buffer;
|
k = buffer_out.sub_buffer[buffer_out.sub_block_number].buffer;
|
||||||
buffer_out.sub_buffer[buffer_out.sub_block_number].buffer = curr_sub_block->buffer;
|
buffer_out.sub_buffer[buffer_out.sub_block_number].buffer = curr_sub_block->buffer;
|
||||||
curr_sub_block->buffer = k;
|
curr_sub_block->buffer = k;
|
||||||
|
|
||||||
|
|
||||||
long long tmp_size;
|
long long tmp_size;
|
||||||
tmp_size = curr_sub_block->size;
|
tmp_size = curr_sub_block->size;
|
||||||
curr_sub_block->size = buffer_out.sub_buffer[buffer_out.sub_block_number].size;
|
curr_sub_block->size = buffer_out.sub_buffer[buffer_out.sub_block_number].size;
|
||||||
buffer_out.sub_buffer[buffer_out.sub_block_number].size = tmp_size;
|
buffer_out.sub_buffer[buffer_out.sub_block_number].size = tmp_size;
|
||||||
|
|
||||||
|
|
||||||
curr_sub_block->length = buffer_out.sub_buffer[buffer_out.sub_block_number].length;
|
curr_sub_block->length = buffer_out.sub_buffer[buffer_out.sub_block_number].length;
|
||||||
buffer_out.sub_buffer[buffer_out.sub_block_number].length = 0;
|
buffer_out.sub_buffer[buffer_out.sub_block_number].length = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
void add_segment_to_sub_buffer(Output_buffer_sub_block* current_sub_buffer, char* seg, long long segLen)
|
void add_segment_to_sub_buffer(Output_buffer_sub_block* current_sub_buffer, char* seg, long long segLen)
|
||||||
{
|
{
|
||||||
if(current_sub_buffer->length + segLen + 2 > current_sub_buffer->size)
|
if(current_sub_buffer->length + segLen + 2 > current_sub_buffer->size)
|
||||||
{
|
{
|
||||||
current_sub_buffer->size = current_sub_buffer->length + segLen + 2;
|
current_sub_buffer->size = current_sub_buffer->length + segLen + 2;
|
||||||
current_sub_buffer->buffer = (char*)realloc(current_sub_buffer->buffer, current_sub_buffer->size);
|
current_sub_buffer->buffer = (char*)realloc(current_sub_buffer->buffer, current_sub_buffer->size);
|
||||||
}
|
}
|
||||||
|
|
||||||
memcpy(current_sub_buffer->buffer + current_sub_buffer->length, seg, segLen);
|
memcpy(current_sub_buffer->buffer + current_sub_buffer->length, seg, segLen);
|
||||||
current_sub_buffer->length += segLen;
|
current_sub_buffer->length += segLen;
|
||||||
current_sub_buffer->buffer[current_sub_buffer->length] = '\0';
|
current_sub_buffer->buffer[current_sub_buffer->length] = '\0';
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
void add_base_to_sub_buffer(Output_buffer_sub_block* current_sub_buffer, char base)
|
void add_base_to_sub_buffer(Output_buffer_sub_block* current_sub_buffer, char base)
|
||||||
{
|
{
|
||||||
if(current_sub_buffer->length + 2 > current_sub_buffer->size)
|
if(current_sub_buffer->length + 2 > current_sub_buffer->size)
|
||||||
{
|
{
|
||||||
current_sub_buffer->size = current_sub_buffer->length + 2;
|
current_sub_buffer->size = current_sub_buffer->length + 2;
|
||||||
current_sub_buffer->buffer = (char*)realloc(current_sub_buffer->buffer, current_sub_buffer->size);
|
current_sub_buffer->buffer = (char*)realloc(current_sub_buffer->buffer, current_sub_buffer->size);
|
||||||
}
|
}
|
||||||
|
|
||||||
current_sub_buffer->buffer[current_sub_buffer->length] = base;
|
current_sub_buffer->buffer[current_sub_buffer->length] = base;
|
||||||
current_sub_buffer->length++;
|
current_sub_buffer->length++;
|
||||||
current_sub_buffer->buffer[current_sub_buffer->length] = '\0';
|
current_sub_buffer->buffer[current_sub_buffer->length] = '\0';
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
inline void push_single_buffer(Output_buffer_sub_block* curr_sub_block)
|
inline void push_single_buffer(Output_buffer_sub_block* curr_sub_block)
|
||||||
{
|
{
|
||||||
|
|
||||||
char *k;
|
char *k;
|
||||||
k = buffer_out.sub_buffer[buffer_out.sub_block_number].buffer;
|
k = buffer_out.sub_buffer[buffer_out.sub_block_number].buffer;
|
||||||
buffer_out.sub_buffer[buffer_out.sub_block_number].buffer = curr_sub_block->buffer;
|
buffer_out.sub_buffer[buffer_out.sub_block_number].buffer = curr_sub_block->buffer;
|
||||||
curr_sub_block->buffer = k;
|
curr_sub_block->buffer = k;
|
||||||
|
|
||||||
long long tmp_size;
|
long long tmp_size;
|
||||||
tmp_size = curr_sub_block->size;
|
tmp_size = curr_sub_block->size;
|
||||||
curr_sub_block->size = buffer_out.sub_buffer[buffer_out.sub_block_number].size;
|
curr_sub_block->size = buffer_out.sub_buffer[buffer_out.sub_block_number].size;
|
||||||
buffer_out.sub_buffer[buffer_out.sub_block_number].size = tmp_size;
|
buffer_out.sub_buffer[buffer_out.sub_block_number].size = tmp_size;
|
||||||
|
|
||||||
buffer_out.sub_buffer[buffer_out.sub_block_number].length = curr_sub_block->length;
|
buffer_out.sub_buffer[buffer_out.sub_block_number].length = curr_sub_block->length;
|
||||||
curr_sub_block->length = 0;
|
curr_sub_block->length = 0;
|
||||||
|
|
||||||
buffer_out.sub_block_number++;
|
buffer_out.sub_block_number++;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
void* pop_buffer(void*)
|
void* pop_buffer(void*)
|
||||||
{
|
{
|
||||||
|
|
||||||
FILE* output_file = fopen(asm_opt.output_file_name, "w");
|
FILE* output_file = fopen(asm_opt.output_file_name, "w");
|
||||||
|
|
||||||
init_buffer_sub_block(&tmp_buffer_sub_block);
|
init_buffer_sub_block(&tmp_buffer_sub_block);
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
while (buffer_out.all_buffer_end < asm_opt.thread_num)
|
while (buffer_out.all_buffer_end < asm_opt.thread_num)
|
||||||
{
|
{
|
||||||
|
|
||||||
pthread_mutex_lock(&o_queueMutex);
|
pthread_mutex_lock(&o_queueMutex);
|
||||||
|
|
||||||
while (if_empty_buffer() && (buffer_out.all_buffer_end < asm_opt.thread_num))
|
while (if_empty_buffer() && (buffer_out.all_buffer_end < asm_opt.thread_num))
|
||||||
{
|
{
|
||||||
pthread_cond_signal(&o_stallCond);
|
pthread_cond_signal(&o_stallCond);
|
||||||
pthread_cond_wait(&o_flushCond, &o_queueMutex);
|
pthread_cond_wait(&o_flushCond, &o_queueMutex);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!if_empty_buffer())
|
if (!if_empty_buffer())
|
||||||
{
|
{
|
||||||
pop_single_buffer(&tmp_buffer_sub_block);
|
pop_single_buffer(&tmp_buffer_sub_block);
|
||||||
}
|
}
|
||||||
|
|
||||||
pthread_cond_signal(&o_stallCond);
|
pthread_cond_signal(&o_stallCond);
|
||||||
pthread_mutex_unlock(&o_queueMutex);
|
pthread_mutex_unlock(&o_queueMutex);
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
if (tmp_buffer_sub_block.length != 0)
|
if (tmp_buffer_sub_block.length != 0)
|
||||||
{
|
{
|
||||||
fprintf(output_file, "%s", tmp_buffer_sub_block.buffer);
|
fprintf(output_file, "%s", tmp_buffer_sub_block.buffer);
|
||||||
}
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
while (buffer_out.sub_block_number>0)
|
while (buffer_out.sub_block_number>0)
|
||||||
{
|
{
|
||||||
buffer_out.sub_block_number--;
|
buffer_out.sub_block_number--;
|
||||||
fprintf(output_file, "%s", buffer_out.sub_buffer[buffer_out.sub_block_number].buffer);
|
fprintf(output_file, "%s", buffer_out.sub_buffer[buffer_out.sub_block_number].buffer);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
destory_buffer_sub_block(&tmp_buffer_sub_block);
|
destory_buffer_sub_block(&tmp_buffer_sub_block);
|
||||||
|
|
||||||
fclose(output_file);
|
fclose(output_file);
|
||||||
|
|
||||||
return NULL;
|
return NULL;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
void push_results_to_buffer(Output_buffer_sub_block* sub_block)
|
void push_results_to_buffer(Output_buffer_sub_block* sub_block)
|
||||||
{
|
{
|
||||||
|
|
||||||
pthread_mutex_lock(&o_queueMutex);
|
pthread_mutex_lock(&o_queueMutex);
|
||||||
|
|
||||||
while (if_full_buffer())
|
while (if_full_buffer())
|
||||||
{
|
{
|
||||||
pthread_cond_signal(&o_flushCond);
|
pthread_cond_signal(&o_flushCond);
|
||||||
pthread_cond_wait(&o_stallCond, &o_queueMutex);
|
pthread_cond_wait(&o_stallCond, &o_queueMutex);
|
||||||
}
|
}
|
||||||
|
|
||||||
push_single_buffer(sub_block);
|
push_single_buffer(sub_block);
|
||||||
pthread_cond_signal(&o_flushCond);
|
pthread_cond_signal(&o_flushCond);
|
||||||
pthread_mutex_unlock(&o_queueMutex);
|
pthread_mutex_unlock(&o_queueMutex);
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
void finish_output_buffer()
|
void finish_output_buffer()
|
||||||
{
|
{
|
||||||
|
|
||||||
pthread_mutex_lock(&o_doneMutex);
|
pthread_mutex_lock(&o_doneMutex);
|
||||||
buffer_out.all_buffer_end++;
|
buffer_out.all_buffer_end++;
|
||||||
|
|
||||||
|
|
||||||
if (buffer_out.all_buffer_end == asm_opt.thread_num)
|
if (buffer_out.all_buffer_end == asm_opt.thread_num)
|
||||||
{
|
{
|
||||||
pthread_cond_signal(&o_flushCond);
|
pthread_cond_signal(&o_flushCond);
|
||||||
}
|
}
|
||||||
|
|
||||||
pthread_mutex_unlock(&o_doneMutex);
|
pthread_mutex_unlock(&o_doneMutex);
|
||||||
}
|
}
|
||||||
@@ -1,38 +1,38 @@
|
|||||||
#ifndef __OUTPUT__
|
#ifndef __OUTPUT__
|
||||||
#define __OUTPUT__
|
#define __OUTPUT__
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include<stdint.h>
|
#include<stdint.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <zlib.h>
|
#include <zlib.h>
|
||||||
|
|
||||||
typedef struct
|
typedef struct
|
||||||
{
|
{
|
||||||
char* buffer;
|
char* buffer;
|
||||||
long long size;
|
long long size;
|
||||||
long long length;
|
long long length;
|
||||||
} Output_buffer_sub_block;
|
} Output_buffer_sub_block;
|
||||||
|
|
||||||
typedef struct Output_buffer
|
typedef struct Output_buffer
|
||||||
{
|
{
|
||||||
Output_buffer_sub_block* sub_buffer;
|
Output_buffer_sub_block* sub_buffer;
|
||||||
long long sub_block_size;
|
long long sub_block_size;
|
||||||
long long sub_block_number;
|
long long sub_block_number;
|
||||||
int all_buffer_end;
|
int all_buffer_end;
|
||||||
} Output_buffer;
|
} Output_buffer;
|
||||||
|
|
||||||
#define OUTPUT_BUFFER_SIZE 100
|
#define OUTPUT_BUFFER_SIZE 100
|
||||||
#define SUB_BLOCK_INIT_SIZE 10000
|
#define SUB_BLOCK_INIT_SIZE 10000
|
||||||
|
|
||||||
void init_buffer_sub_block(Output_buffer_sub_block* sub_block);
|
void init_buffer_sub_block(Output_buffer_sub_block* sub_block);
|
||||||
void* pop_buffer(void*);
|
void* pop_buffer(void*);
|
||||||
void add_segment_to_sub_buffer(Output_buffer_sub_block* current_sub_buffer, char* seg, long long segLen);
|
void add_segment_to_sub_buffer(Output_buffer_sub_block* current_sub_buffer, char* seg, long long segLen);
|
||||||
void add_base_to_sub_buffer(Output_buffer_sub_block* current_sub_buffer, char base);
|
void add_base_to_sub_buffer(Output_buffer_sub_block* current_sub_buffer, char base);
|
||||||
void push_results_to_buffer(Output_buffer_sub_block* sub_block);
|
void push_results_to_buffer(Output_buffer_sub_block* sub_block);
|
||||||
void finish_output_buffer();
|
void finish_output_buffer();
|
||||||
void destory_buffer_sub_block(Output_buffer_sub_block* sub_block);
|
void destory_buffer_sub_block(Output_buffer_sub_block* sub_block);
|
||||||
void destory_output_buffer();
|
void destory_output_buffer();
|
||||||
void init_output_buffer(int thread_number);
|
void init_output_buffer(int thread_number);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
+40262
-39737
File diff suppressed because it is too large
Load Diff
+1253
-1253
File diff suppressed because it is too large
Load Diff
@@ -1,359 +1,359 @@
|
|||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include "POA.h"
|
#include "POA.h"
|
||||||
#include "Correct.h"
|
#include "Correct.h"
|
||||||
#include "Process_Read.h"
|
#include "Process_Read.h"
|
||||||
#define INIT_EDGE_SIZE 50
|
#define INIT_EDGE_SIZE 50
|
||||||
#define INCREASE_EDGE_SIZE 5
|
#define INCREASE_EDGE_SIZE 5
|
||||||
#define INIT_NODE_SIZE 16000
|
#define INIT_NODE_SIZE 16000
|
||||||
|
|
||||||
/********
|
/********
|
||||||
* Edge *
|
* Edge *
|
||||||
********/
|
********/
|
||||||
|
|
||||||
void init_Edge_alloc(Edge_alloc* list)
|
void init_Edge_alloc(Edge_alloc* list)
|
||||||
{
|
{
|
||||||
if (list->list == NULL) {
|
if (list->list == NULL) {
|
||||||
list->size = INIT_EDGE_SIZE;
|
list->size = INIT_EDGE_SIZE;
|
||||||
list->length = 0;
|
list->length = 0;
|
||||||
list->delete_length = 0;
|
list->delete_length = 0;
|
||||||
list->list = (Edge*)malloc(sizeof(Edge)*list->size);
|
list->list = (Edge*)malloc(sizeof(Edge)*list->size);
|
||||||
} else {
|
} else {
|
||||||
list->length = 0;
|
list->length = 0;
|
||||||
list->delete_length = 0;
|
list->delete_length = 0;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void clear_Edge_alloc(Edge_alloc* list)
|
void clear_Edge_alloc(Edge_alloc* list)
|
||||||
{
|
{
|
||||||
list->length = 0;
|
list->length = 0;
|
||||||
list->delete_length = 0;
|
list->delete_length = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
void destory_Edge_alloc(Edge_alloc* list)
|
void destory_Edge_alloc(Edge_alloc* list)
|
||||||
{
|
{
|
||||||
if (list && list->list)
|
if (list && list->list)
|
||||||
free(list->list);
|
free(list->list);
|
||||||
}
|
}
|
||||||
|
|
||||||
void append_Edge_alloc(Edge_alloc* list, uint64_t in_node, uint64_t out_node, uint64_t weight, uint64_t length)
|
void append_Edge_alloc(Edge_alloc* list, uint64_t in_node, uint64_t out_node, uint64_t weight, uint64_t length)
|
||||||
{
|
{
|
||||||
if (list->length + 1 > list->size) {
|
if (list->length + 1 > list->size) {
|
||||||
uint64_t old_size = list->size;
|
uint64_t old_size = list->size;
|
||||||
list->size = list->size + INCREASE_EDGE_SIZE;
|
list->size = list->size + INCREASE_EDGE_SIZE;
|
||||||
list->list = (Edge*)realloc(list->list, sizeof(Edge)*list->size);
|
list->list = (Edge*)realloc(list->list, sizeof(Edge)*list->size);
|
||||||
memset(&list->list[old_size], 0, (list->size - old_size) * sizeof(Edge));
|
memset(&list->list[old_size], 0, (list->size - old_size) * sizeof(Edge));
|
||||||
}
|
}
|
||||||
|
|
||||||
list->list[list->length].in_node = in_node;
|
list->list[list->length].in_node = in_node;
|
||||||
list->list[list->length].out_node = out_node;
|
list->list[list->length].out_node = out_node;
|
||||||
list->list[list->length].weight = weight;
|
list->list[list->length].weight = weight;
|
||||||
list->list[list->length].length = length;
|
list->list[list->length].length = length;
|
||||||
list->list[list->length].num_insertions = 0;
|
list->list[list->length].num_insertions = 0;
|
||||||
list->list[list->length].self_edge_ID = list->length;
|
list->list[list->length].self_edge_ID = list->length;
|
||||||
|
|
||||||
list->length++;
|
list->length++;
|
||||||
}
|
}
|
||||||
|
|
||||||
int add_and_check_bi_direction_edge(Graph* graph, Node* in_node, Node* out_node, uint64_t weight, uint64_t flag)
|
int add_and_check_bi_direction_edge(Graph* graph, Node* in_node, Node* out_node, uint64_t weight, uint64_t flag)
|
||||||
{
|
{
|
||||||
Edge* e_forward;
|
Edge* e_forward;
|
||||||
Edge* e_backward;
|
Edge* e_backward;
|
||||||
|
|
||||||
//if there are no edge from in_node to out_node
|
//if there are no edge from in_node to out_node
|
||||||
if(!get_bi_Edge(graph, in_node, out_node, &e_forward, &e_backward))
|
if(!get_bi_Edge(graph, in_node, out_node, &e_forward, &e_backward))
|
||||||
{
|
{
|
||||||
append_Edge_alloc(&(Output_Edges((*in_node))), (*in_node).ID, (*out_node).ID, weight, flag);
|
append_Edge_alloc(&(Output_Edges((*in_node))), (*in_node).ID, (*out_node).ID, weight, flag);
|
||||||
append_Edge_alloc(&(Input_Edges((*out_node))), (*in_node).ID, (*out_node).ID, weight, flag);
|
append_Edge_alloc(&(Input_Edges((*out_node))), (*in_node).ID, (*out_node).ID, weight, flag);
|
||||||
|
|
||||||
Output_Edges((*in_node)).list[Output_Edges((*in_node)).length - 1].reverse_edge_ID
|
Output_Edges((*in_node)).list[Output_Edges((*in_node)).length - 1].reverse_edge_ID
|
||||||
= Input_Edges((*out_node)).length - 1;
|
= Input_Edges((*out_node)).length - 1;
|
||||||
|
|
||||||
Input_Edges((*out_node)).list[Input_Edges((*out_node)).length - 1].reverse_edge_ID
|
Input_Edges((*out_node)).list[Input_Edges((*out_node)).length - 1].reverse_edge_ID
|
||||||
= Output_Edges((*in_node)).length - 1;
|
= Output_Edges((*in_node)).length - 1;
|
||||||
|
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
else//if there is an edge from in_node to out_node, do nothing
|
else//if there is an edge from in_node to out_node, do nothing
|
||||||
{
|
{
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void add_bi_direction_edge(Graph* graph, Node* in_node, Node* out_node, uint64_t weight, uint64_t flag)
|
void add_bi_direction_edge(Graph* graph, Node* in_node, Node* out_node, uint64_t weight, uint64_t flag)
|
||||||
{
|
{
|
||||||
|
|
||||||
append_Edge_alloc(&(Output_Edges((*in_node))), (*in_node).ID, (*out_node).ID, weight, flag);
|
append_Edge_alloc(&(Output_Edges((*in_node))), (*in_node).ID, (*out_node).ID, weight, flag);
|
||||||
append_Edge_alloc(&(Input_Edges((*out_node))), (*in_node).ID, (*out_node).ID, weight, flag);
|
append_Edge_alloc(&(Input_Edges((*out_node))), (*in_node).ID, (*out_node).ID, weight, flag);
|
||||||
|
|
||||||
Output_Edges((*in_node)).list[Output_Edges((*in_node)).length - 1].reverse_edge_ID
|
Output_Edges((*in_node)).list[Output_Edges((*in_node)).length - 1].reverse_edge_ID
|
||||||
= Input_Edges((*out_node)).length - 1;
|
= Input_Edges((*out_node)).length - 1;
|
||||||
|
|
||||||
Input_Edges((*out_node)).list[Input_Edges((*out_node)).length - 1].reverse_edge_ID
|
Input_Edges((*out_node)).list[Input_Edges((*out_node)).length - 1].reverse_edge_ID
|
||||||
= Output_Edges((*in_node)).length - 1;
|
= Output_Edges((*in_node)).length - 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
int remove_and_check_bi_direction_edge_from_nodes(Graph* graph, Node* in_node, Node* out_node)
|
int remove_and_check_bi_direction_edge_from_nodes(Graph* graph, Node* in_node, Node* out_node)
|
||||||
{
|
{
|
||||||
Edge* e_forward;
|
Edge* e_forward;
|
||||||
Edge* e_backward;
|
Edge* e_backward;
|
||||||
|
|
||||||
//if there are no edge from in_node to out_node
|
//if there are no edge from in_node to out_node
|
||||||
//1. remove these two edges
|
//1. remove these two edges
|
||||||
//2. increase the edge_list.delete_length in both in_node and out_node
|
//2. increase the edge_list.delete_length in both in_node and out_node
|
||||||
if(get_bi_Edge(graph, in_node, out_node, &e_forward, &e_backward))
|
if(get_bi_Edge(graph, in_node, out_node, &e_forward, &e_backward))
|
||||||
{
|
{
|
||||||
e_forward->in_node = (uint64_t)-1;
|
e_forward->in_node = (uint64_t)-1;
|
||||||
e_forward->out_node = (uint64_t)-1;
|
e_forward->out_node = (uint64_t)-1;
|
||||||
e_forward->weight = (uint64_t)-1;
|
e_forward->weight = (uint64_t)-1;
|
||||||
e_forward->length = (uint64_t)-1;
|
e_forward->length = (uint64_t)-1;
|
||||||
e_forward->num_insertions = (uint64_t)-1;
|
e_forward->num_insertions = (uint64_t)-1;
|
||||||
e_forward->self_edge_ID = (uint64_t)-1;
|
e_forward->self_edge_ID = (uint64_t)-1;
|
||||||
e_forward->reverse_edge_ID = (uint64_t)-1;
|
e_forward->reverse_edge_ID = (uint64_t)-1;
|
||||||
|
|
||||||
|
|
||||||
e_backward->in_node = (uint64_t)-1;
|
e_backward->in_node = (uint64_t)-1;
|
||||||
e_backward->out_node = (uint64_t)-1;
|
e_backward->out_node = (uint64_t)-1;
|
||||||
e_backward->weight = (uint64_t)-1;
|
e_backward->weight = (uint64_t)-1;
|
||||||
e_backward->length = (uint64_t)-1;
|
e_backward->length = (uint64_t)-1;
|
||||||
e_backward->num_insertions = (uint64_t)-1;
|
e_backward->num_insertions = (uint64_t)-1;
|
||||||
e_backward->self_edge_ID = (uint64_t)-1;
|
e_backward->self_edge_ID = (uint64_t)-1;
|
||||||
e_backward->reverse_edge_ID = (uint64_t)-1;
|
e_backward->reverse_edge_ID = (uint64_t)-1;
|
||||||
|
|
||||||
Output_Edges(*in_node).delete_length++;
|
Output_Edges(*in_node).delete_length++;
|
||||||
Input_Edges((*out_node)).delete_length++;
|
Input_Edges((*out_node)).delete_length++;
|
||||||
|
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
else//if there is an edge from in_node to out_node, do nothing
|
else//if there is an edge from in_node to out_node, do nothing
|
||||||
{
|
{
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
int remove_and_check_bi_direction_edge_from_edge(Graph* graph, Edge* e)
|
int remove_and_check_bi_direction_edge_from_edge(Graph* graph, Edge* e)
|
||||||
{
|
{
|
||||||
Edge* e_forward;
|
Edge* e_forward;
|
||||||
Edge* e_backward;
|
Edge* e_backward;
|
||||||
|
|
||||||
if(If_Edge_Exist(*e))
|
if(If_Edge_Exist(*e))
|
||||||
{
|
{
|
||||||
get_bi_direction_edges(graph, e, &e_forward, &e_backward);
|
get_bi_direction_edges(graph, e, &e_forward, &e_backward);
|
||||||
Output_Edges(G_Node(*graph, e_forward->in_node)).delete_length++;
|
Output_Edges(G_Node(*graph, e_forward->in_node)).delete_length++;
|
||||||
Input_Edges(G_Node(*graph, e_forward->out_node)).delete_length++;
|
Input_Edges(G_Node(*graph, e_forward->out_node)).delete_length++;
|
||||||
|
|
||||||
e_forward->in_node = (uint64_t)-1;
|
e_forward->in_node = (uint64_t)-1;
|
||||||
e_forward->out_node = (uint64_t)-1;
|
e_forward->out_node = (uint64_t)-1;
|
||||||
e_forward->weight = (uint64_t)-1;
|
e_forward->weight = (uint64_t)-1;
|
||||||
e_forward->length = (uint64_t)-1;
|
e_forward->length = (uint64_t)-1;
|
||||||
e_forward->num_insertions = (uint64_t)-1;
|
e_forward->num_insertions = (uint64_t)-1;
|
||||||
e_forward->self_edge_ID = (uint64_t)-1;
|
e_forward->self_edge_ID = (uint64_t)-1;
|
||||||
e_forward->reverse_edge_ID = (uint64_t)-1;
|
e_forward->reverse_edge_ID = (uint64_t)-1;
|
||||||
|
|
||||||
|
|
||||||
e_backward->in_node = (uint64_t)-1;
|
e_backward->in_node = (uint64_t)-1;
|
||||||
e_backward->out_node = (uint64_t)-1;
|
e_backward->out_node = (uint64_t)-1;
|
||||||
e_backward->weight = (uint64_t)-1;
|
e_backward->weight = (uint64_t)-1;
|
||||||
e_backward->length = (uint64_t)-1;
|
e_backward->length = (uint64_t)-1;
|
||||||
e_backward->num_insertions = (uint64_t)-1;
|
e_backward->num_insertions = (uint64_t)-1;
|
||||||
e_backward->self_edge_ID = (uint64_t)-1;
|
e_backward->self_edge_ID = (uint64_t)-1;
|
||||||
e_backward->reverse_edge_ID = (uint64_t)-1;
|
e_backward->reverse_edge_ID = (uint64_t)-1;
|
||||||
|
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
else
|
else
|
||||||
{
|
{
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/********
|
/********
|
||||||
* Node *
|
* Node *
|
||||||
********/
|
********/
|
||||||
|
|
||||||
void init_Node_alloc(Node_alloc* list)
|
void init_Node_alloc(Node_alloc* list)
|
||||||
{
|
{
|
||||||
memset(list, 0, sizeof(Node_alloc));
|
memset(list, 0, sizeof(Node_alloc));
|
||||||
list->size = INIT_NODE_SIZE;
|
list->size = INIT_NODE_SIZE;
|
||||||
list->list = (Node*)calloc(list->size, sizeof(Node));
|
list->list = (Node*)calloc(list->size, sizeof(Node));
|
||||||
}
|
}
|
||||||
|
|
||||||
void destory_Node_alloc(Node_alloc* list)
|
void destory_Node_alloc(Node_alloc* list)
|
||||||
{
|
{
|
||||||
uint64_t i;
|
uint64_t i;
|
||||||
for (i = 0; i < list->size; i++) {
|
for (i = 0; i < list->size; i++) {
|
||||||
destory_Edge_alloc(&list->list[i].deletion_edges);
|
destory_Edge_alloc(&list->list[i].deletion_edges);
|
||||||
destory_Edge_alloc(&list->list[i].insertion_edges);
|
destory_Edge_alloc(&list->list[i].insertion_edges);
|
||||||
destory_Edge_alloc(&list->list[i].mismatch_edges);
|
destory_Edge_alloc(&list->list[i].mismatch_edges);
|
||||||
}
|
}
|
||||||
free(list->list);
|
free(list->list);
|
||||||
free(list->sort.list);
|
free(list->sort.list);
|
||||||
free(list->sort.visit);
|
free(list->sort.visit);
|
||||||
free(list->sort.iterative_buffer);
|
free(list->sort.iterative_buffer);
|
||||||
free(list->sort.iterative_buffer_visit);
|
free(list->sort.iterative_buffer_visit);
|
||||||
}
|
}
|
||||||
|
|
||||||
void clear_Node_alloc(Node_alloc* list)
|
void clear_Node_alloc(Node_alloc* list)
|
||||||
{
|
{
|
||||||
uint64_t i =0;
|
uint64_t i =0;
|
||||||
for (i = 0; i < list->length; i++) { // TODO: is this list->size or list->length? The original version is list->length.
|
for (i = 0; i < list->length; i++) { // TODO: is this list->size or list->length? The original version is list->length.
|
||||||
clear_Edge_alloc(&list->list[i].insertion_edges);
|
clear_Edge_alloc(&list->list[i].insertion_edges);
|
||||||
clear_Edge_alloc(&list->list[i].mismatch_edges);
|
clear_Edge_alloc(&list->list[i].mismatch_edges);
|
||||||
clear_Edge_alloc(&list->list[i].deletion_edges);
|
clear_Edge_alloc(&list->list[i].deletion_edges);
|
||||||
}
|
}
|
||||||
list->length = 0;
|
list->length = 0;
|
||||||
list->delete_length = 0;
|
list->delete_length = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
uint64_t append_Node_alloc(Node_alloc* list, char base)
|
uint64_t append_Node_alloc(Node_alloc* list, char base)
|
||||||
{
|
{
|
||||||
if (list->length + 1 > list->size) {
|
if (list->length + 1 > list->size) {
|
||||||
uint64_t old_size = list->size;
|
uint64_t old_size = list->size;
|
||||||
list->size = list->size * 2;
|
list->size = list->size * 2;
|
||||||
list->list = (Node*)realloc(list->list, sizeof(Node) * list->size);
|
list->list = (Node*)realloc(list->list, sizeof(Node) * list->size);
|
||||||
memset(&list->list[old_size], 0, (list->size - old_size) * sizeof(Node));
|
memset(&list->list[old_size], 0, (list->size - old_size) * sizeof(Node));
|
||||||
}
|
}
|
||||||
|
|
||||||
list->list[list->length].ID = list->length;
|
list->list[list->length].ID = list->length;
|
||||||
list->list[list->length].base = base;
|
list->list[list->length].base = base;
|
||||||
list->list[list->length].weight = 1;
|
list->list[list->length].weight = 1;
|
||||||
list->list[list->length].num_insertions = 0;
|
list->list[list->length].num_insertions = 0;
|
||||||
init_Edge_alloc(&list->list[list->length].deletion_edges);
|
init_Edge_alloc(&list->list[list->length].deletion_edges);
|
||||||
init_Edge_alloc(&list->list[list->length].insertion_edges);
|
init_Edge_alloc(&list->list[list->length].insertion_edges);
|
||||||
init_Edge_alloc(&list->list[list->length].mismatch_edges);
|
init_Edge_alloc(&list->list[list->length].mismatch_edges);
|
||||||
|
|
||||||
list->length++;
|
list->length++;
|
||||||
|
|
||||||
return list->length - 1;
|
return list->length - 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
/*********
|
/*********
|
||||||
* Graph *
|
* Graph *
|
||||||
*********/
|
*********/
|
||||||
|
|
||||||
void init_Graph(Graph* g)
|
void init_Graph(Graph* g)
|
||||||
{
|
{
|
||||||
init_Node_alloc(&g->g_nodes);
|
init_Node_alloc(&g->g_nodes);
|
||||||
g->g_n_edges = 0;
|
g->g_n_edges = 0;
|
||||||
g->g_n_nodes = 0;
|
g->g_n_nodes = 0;
|
||||||
g->g_next_nodeID = 0;
|
g->g_next_nodeID = 0;
|
||||||
g->s_end_nodeID = 0;
|
g->s_end_nodeID = 0;
|
||||||
g->s_start_nodeID = 0;
|
g->s_start_nodeID = 0;
|
||||||
g->seq = NULL;
|
g->seq = NULL;
|
||||||
g->seqID = (uint64_t)-1;
|
g->seqID = (uint64_t)-1;
|
||||||
|
|
||||||
init_Queue(&(g->node_q));
|
init_Queue(&(g->node_q));
|
||||||
}
|
}
|
||||||
|
|
||||||
void destory_Graph(Graph* g)
|
void destory_Graph(Graph* g)
|
||||||
{
|
{
|
||||||
destory_Node_alloc(&g->g_nodes);
|
destory_Node_alloc(&g->g_nodes);
|
||||||
destory_Queue(&(g->node_q));
|
destory_Queue(&(g->node_q));
|
||||||
}
|
}
|
||||||
|
|
||||||
void clear_Graph(Graph* g)
|
void clear_Graph(Graph* g)
|
||||||
{
|
{
|
||||||
clear_Node_alloc(&g->g_nodes);
|
clear_Node_alloc(&g->g_nodes);
|
||||||
|
|
||||||
g->g_n_edges = 0;
|
g->g_n_edges = 0;
|
||||||
g->g_n_nodes = 0;
|
g->g_n_nodes = 0;
|
||||||
g->g_next_nodeID = 0;
|
g->g_next_nodeID = 0;
|
||||||
g->s_end_nodeID = 0;
|
g->s_end_nodeID = 0;
|
||||||
g->s_start_nodeID = 0;
|
g->s_start_nodeID = 0;
|
||||||
g->seq = NULL;
|
g->seq = NULL;
|
||||||
g->seqID = (uint64_t)-1;
|
g->seqID = (uint64_t)-1;
|
||||||
|
|
||||||
clear_Queue(&(g->node_q));
|
clear_Queue(&(g->node_q));
|
||||||
}
|
}
|
||||||
|
|
||||||
void addUnmatchedSeqToGraph(Graph* g, char* g_read_seq, long long g_read_length, long long* startID, long long* endID)
|
void addUnmatchedSeqToGraph(Graph* g, char* g_read_seq, long long g_read_length, long long* startID, long long* endID)
|
||||||
{
|
{
|
||||||
long long firstID, lastID, nodeID, i;
|
long long firstID, lastID, nodeID, i;
|
||||||
firstID = -1;
|
firstID = -1;
|
||||||
lastID = -1;
|
lastID = -1;
|
||||||
|
|
||||||
if(g_read_length == 0)
|
if(g_read_length == 0)
|
||||||
return;
|
return;
|
||||||
|
|
||||||
///start node
|
///start node
|
||||||
nodeID = add_Node_Graph(g, 'S');
|
nodeID = add_Node_Graph(g, 'S');
|
||||||
firstID = nodeID;
|
firstID = nodeID;
|
||||||
lastID = nodeID;
|
lastID = nodeID;
|
||||||
|
|
||||||
|
|
||||||
for (i = 0; i < g_read_length; i++)
|
for (i = 0; i < g_read_length; i++)
|
||||||
{
|
{
|
||||||
nodeID = add_Node_Graph(g, g_read_seq[i]);
|
nodeID = add_Node_Graph(g, g_read_seq[i]);
|
||||||
|
|
||||||
if (firstID == -1)
|
if (firstID == -1)
|
||||||
{
|
{
|
||||||
firstID = nodeID;
|
firstID = nodeID;
|
||||||
}
|
}
|
||||||
if (lastID != -1)
|
if (lastID != -1)
|
||||||
{
|
{
|
||||||
///the legnth of match edge is 0, while the length of musmatch is 1
|
///the legnth of match edge is 0, while the length of musmatch is 1
|
||||||
append_Edge_alloc(&(g->g_nodes.list[lastID].mismatch_edges), lastID, nodeID, 1, 0);
|
append_Edge_alloc(&(g->g_nodes.list[lastID].mismatch_edges), lastID, nodeID, 1, 0);
|
||||||
}
|
}
|
||||||
|
|
||||||
lastID = nodeID;
|
lastID = nodeID;
|
||||||
}
|
}
|
||||||
|
|
||||||
*startID = firstID;
|
*startID = firstID;
|
||||||
*endID = lastID;
|
*endID = lastID;
|
||||||
|
|
||||||
g->s_start_nodeID = firstID;
|
g->s_start_nodeID = firstID;
|
||||||
g->s_end_nodeID = lastID;
|
g->s_end_nodeID = lastID;
|
||||||
|
|
||||||
}
|
}
|
||||||
|
|
||||||
void addmatchedSeqToGraph(Graph* backbone, long long currentNodeID, char* x_string, long long x_length,
|
void addmatchedSeqToGraph(Graph* backbone, long long currentNodeID, char* x_string, long long x_length,
|
||||||
char* y_string, long long y_length, window_list *cigar_idx, window_list_alloc *cigar_s, long long backbone_start, long long backbone_end)
|
char* y_string, long long y_length, window_list *cigar_idx, window_list_alloc *cigar_s, long long backbone_start, long long backbone_end)
|
||||||
{
|
{
|
||||||
|
|
||||||
int64_t x_i = 0, y_i = 0, c_i = 0, c_n = cigar_idx->clen;
|
int64_t x_i = 0, y_i = 0, c_i = 0, c_n = cigar_idx->clen;
|
||||||
uint32_t i, operLen; uint8_t oper; int8_t last_oper = -1;
|
uint32_t i, operLen; uint8_t oper; int8_t last_oper = -1;
|
||||||
// if(currentNodeID == 366 && x_length == 9 && y_length == 9) {
|
// if(currentNodeID == 366 && x_length == 9 && y_length == 9) {
|
||||||
// fprintf(stderr, "[M::%s] currentNodeID::%lld, x_length::%lld, c_n::%ld, cidx::%u, cigar_s_n::%lld\n", __func__,
|
// fprintf(stderr, "[M::%s] currentNodeID::%lld, x_length::%lld, c_n::%ld, cidx::%u, cigar_s_n::%lld\n", __func__,
|
||||||
// currentNodeID, x_length, c_n, cigar_idx->cidx, (long long)cigar_s->c.n);
|
// currentNodeID, x_length, c_n, cigar_idx->cidx, (long long)cigar_s->c.n);
|
||||||
// }
|
// }
|
||||||
|
|
||||||
///note that node 0 is the start node
|
///note that node 0 is the start node
|
||||||
///0 is match, 1 is mismatch, 2 is up, 3 is left
|
///0 is match, 1 is mismatch, 2 is up, 3 is left
|
||||||
///2 mean y has more bases, while 3 means x has more bases
|
///2 mean y has more bases, while 3 means x has more bases
|
||||||
for (c_i = 0; c_i < c_n; c_i++) {
|
for (c_i = 0; c_i < c_n; c_i++) {
|
||||||
get_cigar_cell(cigar_idx, cigar_s, c_i, &oper, &operLen);
|
get_cigar_cell(cigar_idx, cigar_s, c_i, &oper, &operLen);
|
||||||
|
|
||||||
// if(currentNodeID == 366 && x_length == 9 && y_length == 9) {
|
// if(currentNodeID == 366 && x_length == 9 && y_length == 9) {
|
||||||
// fprintf(stderr, "[M::%s] c_i::%ld, oper::%u, operLen::%u, last_oper::%d\n", __func__, c_i, oper, operLen, last_oper);
|
// fprintf(stderr, "[M::%s] c_i::%ld, oper::%u, operLen::%u, last_oper::%d\n", __func__, c_i, oper, operLen, last_oper);
|
||||||
// }
|
// }
|
||||||
|
|
||||||
if (oper == 0 || oper == 1) { ///match/mismatch
|
if (oper == 0 || oper == 1) { ///match/mismatch
|
||||||
for (i = 0; i < operLen; i++) {
|
for (i = 0; i < operLen; i++) {
|
||||||
///if the previous node is insertion, this node might be mismatch/match
|
///if the previous node is insertion, this node might be mismatch/match
|
||||||
add_mismatchEdge_weight(backbone, currentNodeID, y_string[y_i], last_oper);
|
add_mismatchEdge_weight(backbone, currentNodeID, y_string[y_i], last_oper);
|
||||||
x_i++; y_i++; currentNodeID++;
|
x_i++; y_i++; currentNodeID++;
|
||||||
}
|
}
|
||||||
} else if (oper == 2) { ///insertion
|
} else if (oper == 2) { ///insertion
|
||||||
///the begin and end of cigar cannot be 2, so -1 is right here
|
///the begin and end of cigar cannot be 2, so -1 is right here
|
||||||
///if (operationLen <= CORRECT_INDEL_LENGTH)
|
///if (operationLen <= CORRECT_INDEL_LENGTH)
|
||||||
{
|
{
|
||||||
add_insertionEdge_weight(backbone, currentNodeID, y_string + y_i, operLen);
|
add_insertionEdge_weight(backbone, currentNodeID, y_string + y_i, operLen);
|
||||||
backbone->g_nodes.list[currentNodeID].num_insertions++;
|
backbone->g_nodes.list[currentNodeID].num_insertions++;
|
||||||
}
|
}
|
||||||
y_i += operLen;
|
y_i += operLen;
|
||||||
} else if (oper == 3) {
|
} else if (oper == 3) {
|
||||||
///3 means x has more bases, that means backbone has more bases
|
///3 means x has more bases, that means backbone has more bases
|
||||||
///like a mismatch (-)
|
///like a mismatch (-)
|
||||||
///if (operationLen <= CORRECT_INDEL_LENGTH)
|
///if (operationLen <= CORRECT_INDEL_LENGTH)
|
||||||
{
|
{
|
||||||
add_deletionEdge_weight(backbone, currentNodeID, operLen);
|
add_deletionEdge_weight(backbone, currentNodeID, operLen);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
currentNodeID += operLen;
|
currentNodeID += operLen;
|
||||||
x_i += operLen;
|
x_i += operLen;
|
||||||
}
|
}
|
||||||
|
|
||||||
last_oper = oper;
|
last_oper = oper;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+2251
-2232
File diff suppressed because it is too large
Load Diff
+277
-275
@@ -1,275 +1,277 @@
|
|||||||
#ifndef __READ__
|
#ifndef __READ__
|
||||||
#define __READ__
|
#define __READ__
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include<stdint.h>
|
#include<stdint.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <zlib.h>
|
#include <zlib.h>
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
#include "CommandLines.h"
|
#include "CommandLines.h"
|
||||||
///#include "Hash_Table.h"
|
///#include "Hash_Table.h"
|
||||||
|
|
||||||
#define READ_INIT_NUMBER 1000
|
#define READ_INIT_NUMBER 1000
|
||||||
|
|
||||||
#define READ_BLOCK_SIZE 64
|
#define READ_BLOCK_SIZE 64
|
||||||
#define READ_BLOCK_NUM_PRE_THR 100
|
#define READ_BLOCK_NUM_PRE_THR 100
|
||||||
|
|
||||||
#define IS_FULL(buffer) ((buffer.num >= buffer.size)?1:0)
|
#define IS_FULL(buffer) ((buffer.num >= buffer.size)?1:0)
|
||||||
#define IS_EMPTY(buffer) ((buffer.num == 0)?1:0)
|
#define IS_EMPTY(buffer) ((buffer.num == 0)?1:0)
|
||||||
///#define Get_READ_LENGTH(R_INF, ID) (R_INF.index[ID+1] - R_INF.index[ID])
|
///#define Get_READ_LENGTH(R_INF, ID) (R_INF.index[ID+1] - R_INF.index[ID])
|
||||||
#define Get_READ_LENGTH(R_INF, ID) (R_INF).read_length[(ID)]
|
#define Get_READ_LENGTH(R_INF, ID) (R_INF).read_length[(ID)]
|
||||||
#define Get_NAME_LENGTH(R_INF, ID) ((R_INF).name_index[(ID)+1] - (R_INF).name_index[(ID)])
|
#define Get_NAME_LENGTH(R_INF, ID) ((R_INF).name_index[(ID)+1] - (R_INF).name_index[(ID)])
|
||||||
///#define Get_READ(R_INF, ID) R_INF.read + (R_INF.index[ID]>>2) + ID
|
///#define Get_READ(R_INF, ID) R_INF.read + (R_INF.index[ID]>>2) + ID
|
||||||
#define Get_READ(R_INF, ID) (R_INF).read_sperate[(ID)]
|
#define Get_READ(R_INF, ID) (R_INF).read_sperate[(ID)]
|
||||||
#define Get_QUAL(R_INF, ID) (R_INF).rsc[(ID)]
|
#define Get_QUAL(R_INF, ID) (R_INF).rsc[(ID)]
|
||||||
#define Get_NAME(R_INF, ID) ((R_INF).name + (R_INF).name_index[(ID)])
|
#define Get_NAME(R_INF, ID) ((R_INF).name + (R_INF).name_index[(ID)])
|
||||||
#define CHECK_BY_NAME(R_INF, NAME, ID) (Get_NAME_LENGTH((R_INF),(ID))==strlen((NAME)) && \
|
#define CHECK_BY_NAME(R_INF, NAME, ID) (Get_NAME_LENGTH((R_INF),(ID))==strlen((NAME)) && \
|
||||||
memcmp((NAME), Get_NAME((R_INF), (ID)), Get_NAME_LENGTH((R_INF),(ID))) == 0)
|
memcmp((NAME), Get_NAME((R_INF), (ID)), Get_NAME_LENGTH((R_INF),(ID))) == 0)
|
||||||
#define IS_SCAF_READ(R_INF, ID) ((R_INF).read_sperate[(ID)] == NULL)
|
#define IS_SCAF_READ(R_INF, ID) ((R_INF).read_sperate[(ID)] == NULL)
|
||||||
|
|
||||||
extern uint8_t seq_nt6_table[256];
|
extern uint8_t seq_nt6_table[256];
|
||||||
extern char bit_t_seq_table[256][4];
|
extern char bit_t_seq_table[256][4];
|
||||||
extern char bit_t_seq_table_rc[256][4];
|
extern char bit_t_seq_table_rc[256][4];
|
||||||
extern char s_H[5];
|
extern char s_H[5];
|
||||||
extern char rc_Table[6];
|
extern char rc_Table[6];
|
||||||
|
|
||||||
|
|
||||||
#define RC_CHAR(x) rc_Table[seq_nt6_table[(uint8_t)x]]
|
#define RC_CHAR(x) rc_Table[seq_nt6_table[(uint8_t)x]]
|
||||||
|
|
||||||
void init_aux_table();
|
void init_aux_table();
|
||||||
|
|
||||||
typedef struct { size_t n, m; uint8_t *a; } asg8_v;
|
typedef struct { size_t n, m; uint8_t *a; } asg8_v;
|
||||||
|
|
||||||
typedef struct
|
typedef struct
|
||||||
{
|
{
|
||||||
uint64_t x_id;
|
uint64_t x_id;
|
||||||
uint64_t x_pos_s;
|
uint64_t x_pos_s;
|
||||||
uint64_t x_pos_e;
|
uint64_t x_pos_e;
|
||||||
uint8_t x_pos_strand;
|
uint8_t x_pos_strand;
|
||||||
|
|
||||||
uint64_t y_id;
|
uint64_t y_id;
|
||||||
uint64_t y_pos_s;
|
uint64_t y_pos_s;
|
||||||
uint64_t y_pos_e;
|
uint64_t y_pos_e;
|
||||||
uint8_t y_pos_strand;
|
uint8_t y_pos_strand;
|
||||||
|
|
||||||
uint64_t matchLen;
|
uint64_t matchLen;
|
||||||
uint64_t totalLen;
|
uint64_t totalLen;
|
||||||
|
|
||||||
} PAF;
|
} PAF;
|
||||||
|
|
||||||
typedef struct
|
typedef struct
|
||||||
{
|
{
|
||||||
PAF* list;
|
PAF* list;
|
||||||
uint64_t size;
|
uint64_t size;
|
||||||
uint64_t length;
|
uint64_t length;
|
||||||
} PAF_alloc;
|
} PAF_alloc;
|
||||||
|
|
||||||
|
|
||||||
inline void init_PAF_alloc(PAF_alloc* list)
|
inline void init_PAF_alloc(PAF_alloc* list)
|
||||||
{
|
{
|
||||||
list->size = 15;
|
list->size = 15;
|
||||||
list->length = 0;
|
list->length = 0;
|
||||||
list->list = (PAF*)malloc(sizeof(PAF)*list->size);
|
list->list = (PAF*)malloc(sizeof(PAF)*list->size);
|
||||||
}
|
}
|
||||||
|
|
||||||
inline void append_PAF_alloc(PAF_alloc* list, PAF* e)
|
inline void append_PAF_alloc(PAF_alloc* list, PAF* e)
|
||||||
{
|
{
|
||||||
if(list->length+1 > list->size)
|
if(list->length+1 > list->size)
|
||||||
{
|
{
|
||||||
list->size = list->size * 2;
|
list->size = list->size * 2;
|
||||||
list->list = (PAF*)realloc(list->list, sizeof(PAF)*list->size);
|
list->list = (PAF*)realloc(list->list, sizeof(PAF)*list->size);
|
||||||
}
|
}
|
||||||
|
|
||||||
list->list[list->length] = (*e);
|
list->list[list->length] = (*e);
|
||||||
list->length++;
|
list->length++;
|
||||||
}
|
}
|
||||||
|
|
||||||
typedef struct
|
typedef struct
|
||||||
{
|
{
|
||||||
/**[0-1] bits are type:**/
|
/**[0-1] bits are type:**/
|
||||||
/**[2-31] bits are length**/
|
/**[2-31] bits are length**/
|
||||||
uint32_t* record;
|
uint32_t* record;
|
||||||
uint32_t length;
|
uint32_t length;
|
||||||
uint32_t size;
|
uint32_t size;
|
||||||
|
|
||||||
char* lost_base;
|
char* lost_base;
|
||||||
uint32_t lost_base_length;
|
uint32_t lost_base_length;
|
||||||
uint32_t lost_base_size;
|
uint32_t lost_base_size;
|
||||||
uint32_t new_length;
|
uint32_t new_length;
|
||||||
} Compressed_Cigar_record;
|
} Compressed_Cigar_record;
|
||||||
|
|
||||||
|
|
||||||
#define AMBIGU 0
|
#define AMBIGU 0
|
||||||
#define FATHER 1
|
#define FATHER 1
|
||||||
#define MOTHER 2
|
#define MOTHER 2
|
||||||
#define MIX_TRIO 3
|
#define MIX_TRIO 3
|
||||||
#define NON_TRIO 4
|
#define NON_TRIO 4
|
||||||
#define DROP 5
|
#define DROP 5
|
||||||
#define SET_TRIO 8
|
#define SET_TRIO 8
|
||||||
#define CHAIN_MATCH 1
|
#define CHAIN_MATCH 1
|
||||||
#define CHAIN_UNMATCH 0.334
|
#define CHAIN_UNMATCH 0.334
|
||||||
|
|
||||||
#define NEC 1
|
#define NEC 1
|
||||||
|
|
||||||
typedef struct
|
typedef struct
|
||||||
{
|
{
|
||||||
uint64_t** N_site;
|
uint64_t** N_site;
|
||||||
///uint8_t* read;
|
///uint8_t* read;
|
||||||
char* name;
|
char* name;
|
||||||
|
|
||||||
uint8_t** read_sperate;
|
uint8_t** read_sperate;
|
||||||
uint64_t* read_length;
|
uint64_t* read_length;
|
||||||
uint64_t* read_size;
|
uint64_t* read_size;
|
||||||
uint8_t* trio_flag;
|
uint8_t* trio_flag;
|
||||||
uint8_t** rsc;
|
uint8_t** rsc;
|
||||||
|
|
||||||
///seq start pos in uint8_t* read
|
///seq start pos in uint8_t* read
|
||||||
///do not need it
|
///do not need it
|
||||||
///uint64_t* index;
|
///uint64_t* index;
|
||||||
uint64_t index_size;
|
uint64_t index_size;
|
||||||
|
|
||||||
///name start pos in char* name
|
///name start pos in char* name
|
||||||
uint64_t* name_index;
|
uint64_t* name_index;
|
||||||
uint64_t name_index_size;
|
uint64_t name_index_size;
|
||||||
uint64_t total_reads;
|
uint64_t total_reads;
|
||||||
uint64_t total_reads_bases;
|
uint64_t tqn;
|
||||||
uint64_t total_name_length;
|
uint64_t total_reads_bases;
|
||||||
|
uint64_t total_name_length;
|
||||||
Compressed_Cigar_record* cigars;
|
uint64_t tr[2];
|
||||||
Compressed_Cigar_record* second_round_cigar;
|
|
||||||
|
Compressed_Cigar_record* cigars;
|
||||||
ma_hit_t_alloc* paf;
|
Compressed_Cigar_record* second_round_cigar;
|
||||||
ma_hit_t_alloc* reverse_paf;
|
|
||||||
|
ma_hit_t_alloc* paf;
|
||||||
///kvec_t_u64_warp* pb_regions;
|
ma_hit_t_alloc* reverse_paf;
|
||||||
} All_reads;
|
|
||||||
|
///kvec_t_u64_warp* pb_regions;
|
||||||
extern All_reads R_INF;
|
} All_reads;
|
||||||
|
|
||||||
typedef struct
|
extern All_reads R_INF;
|
||||||
{
|
|
||||||
char* seq;
|
typedef struct
|
||||||
long long length;
|
{
|
||||||
long long size;
|
char* seq;
|
||||||
long long RID;
|
long long length;
|
||||||
} UC_Read;
|
long long size;
|
||||||
|
long long RID;
|
||||||
typedef struct
|
} UC_Read;
|
||||||
{
|
|
||||||
char** read_name;
|
typedef struct
|
||||||
uint64_t *read_id;
|
{
|
||||||
uint64_t query_num;
|
char** read_name;
|
||||||
kvec_t_u64_warp* candidate_count;
|
uint64_t *read_id;
|
||||||
FILE *fp, *fp_r0, *fp_r1;
|
uint64_t query_num;
|
||||||
pthread_mutex_t OutputMutex;
|
kvec_t_u64_warp* candidate_count;
|
||||||
} Debug_reads;
|
FILE *fp, *fp_r0, *fp_r1;
|
||||||
|
pthread_mutex_t OutputMutex;
|
||||||
|
} Debug_reads;
|
||||||
typedef struct
|
|
||||||
{
|
|
||||||
uint32_t hid;
|
typedef struct
|
||||||
uint32_t qs, qe, ts, te; uint32_t pidx, pdis, aidx;///TODO: enable pdis
|
{
|
||||||
uint8_t pchain:5, rev:1, base:1, el:1;
|
uint32_t hid;
|
||||||
} uc_block_t;
|
uint32_t qs, qe, ts, te; uint32_t pidx, pdis, aidx;///TODO: enable pdis
|
||||||
|
uint8_t pchain:5, rev:1, base:1, el:1;
|
||||||
typedef struct
|
} uc_block_t;
|
||||||
{
|
|
||||||
uint32_t *a;
|
typedef struct
|
||||||
size_t n, m;
|
{
|
||||||
} N_t;
|
uint32_t *a;
|
||||||
|
size_t n, m;
|
||||||
typedef struct
|
} N_t;
|
||||||
{
|
|
||||||
char *a; uint32_t n;
|
typedef struct
|
||||||
} nid_t;
|
{
|
||||||
|
char *a; uint32_t n;
|
||||||
typedef struct
|
} nid_t;
|
||||||
{
|
|
||||||
kvec_t(uint8_t) r_base;
|
typedef struct
|
||||||
uint32_t rlen;
|
{
|
||||||
|
kvec_t(uint8_t) r_base;
|
||||||
kvec_t(uc_block_t) bb;
|
uint32_t rlen;
|
||||||
N_t N_site;
|
|
||||||
|
kvec_t(uc_block_t) bb;
|
||||||
uint8_t dd;
|
N_t N_site;
|
||||||
} ul_vec_t;
|
|
||||||
|
uint8_t dd;
|
||||||
typedef struct{
|
} ul_vec_t;
|
||||||
kvec_t(uint32_t) idx;
|
|
||||||
kvec_t(uint64_t) occ;
|
typedef struct{
|
||||||
} ul_vec_rid_t;
|
kvec_t(uint32_t) idx;
|
||||||
|
kvec_t(uint64_t) occ;
|
||||||
typedef struct
|
} ul_vec_rid_t;
|
||||||
{
|
|
||||||
kvec_t(nid_t) nid;
|
typedef struct
|
||||||
ul_vec_rid_t ridx;
|
{
|
||||||
ul_vec_t *a;
|
kvec_t(nid_t) nid;
|
||||||
size_t n, m;
|
ul_vec_rid_t ridx;
|
||||||
All_reads *hR;
|
ul_vec_t *a;
|
||||||
// idx_emask_t *mm;
|
size_t n, m;
|
||||||
// uint32_t mm;
|
All_reads *hR;
|
||||||
} all_ul_t;
|
// idx_emask_t *mm;
|
||||||
|
// uint32_t mm;
|
||||||
|
} all_ul_t;
|
||||||
typedef struct {
|
|
||||||
ul_vec_t *a;
|
|
||||||
size_t n, m;
|
typedef struct {
|
||||||
uint8_t dd;
|
ul_vec_t *a;
|
||||||
} scaf_res_t;
|
size_t n, m;
|
||||||
|
uint8_t dd;
|
||||||
extern all_ul_t UL_INF;
|
} scaf_res_t;
|
||||||
extern all_ul_t ULG_INF;
|
|
||||||
// extern uint32_t *het_cnt;
|
extern all_ul_t UL_INF;
|
||||||
// extern uint32_t debug_out;
|
extern all_ul_t ULG_INF;
|
||||||
|
// extern uint32_t *het_cnt;
|
||||||
void init_All_reads(All_reads* r);
|
// extern uint32_t debug_out;
|
||||||
void malloc_All_reads(All_reads* r);
|
|
||||||
void ha_insert_read_len(All_reads *r, int read_len, int name_len);
|
void init_All_reads(All_reads* r);
|
||||||
void ha_compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_lis, uint64_t N_site_occ);
|
void malloc_All_reads(All_reads* r);
|
||||||
void ha_compress_qual(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitn, uint64_t sc_off);
|
void ha_insert_read_len(All_reads *r, int read_len, int name_len);
|
||||||
void init_UC_Read(UC_Read* r);
|
void ha_compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_lis, uint64_t N_site_occ);
|
||||||
void recover_UC_Read(UC_Read* r, const All_reads *R_INF, uint64_t ID);
|
void ha_compress_qual(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitn, uint64_t sc_off);
|
||||||
void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID);
|
void init_UC_Read(UC_Read* r);
|
||||||
void recover_UC_Read_sub_region(char* r, int64_t start_pos, int64_t length, uint8_t strand, All_reads* R_INF, int64_t ID);
|
void recover_UC_Read(UC_Read* r, const All_reads *R_INF, uint64_t ID);
|
||||||
void destory_UC_Read(UC_Read* r);
|
void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID);
|
||||||
void reverse_complement(char* pattern, uint64_t length);
|
void recover_UC_Read_sub_region(char* r, int64_t start_pos, int64_t length, uint8_t strand, All_reads* R_INF, int64_t ID);
|
||||||
void write_All_reads(All_reads* r, char* read_file_name);
|
void destory_UC_Read(UC_Read* r);
|
||||||
int load_All_reads(All_reads* r, char* read_file_name);
|
void reverse_complement(char* pattern, uint64_t length);
|
||||||
int append_All_reads(All_reads* r, char *idx, uint32_t id);
|
void write_All_reads(All_reads* r, char* read_file_name);
|
||||||
void destory_All_reads(All_reads* r);
|
int load_All_reads(All_reads* r, char* read_file_name);
|
||||||
int destory_read_bin(All_reads* r);
|
int append_All_reads(All_reads* r, char *idx, uint32_t id);
|
||||||
void init_Debug_reads(Debug_reads* x, const char* file);
|
void destory_All_reads(All_reads* r);
|
||||||
void destory_Debug_reads(Debug_reads* x);
|
int destory_read_bin(All_reads* r);
|
||||||
void recover_UC_sub_Read(UC_Read* i_r, long long start_pos, long long length, uint8_t strand, All_reads* R_INF, long long ID);
|
void init_Debug_reads(Debug_reads* x, const char* file);
|
||||||
|
void destory_Debug_reads(Debug_reads* x);
|
||||||
void init_all_ul_t(all_ul_t *x, All_reads *hR);
|
void recover_UC_sub_Read(UC_Read* i_r, long long start_pos, long long length, uint8_t strand, All_reads* R_INF, long long ID);
|
||||||
void destory_all_ul_t(all_ul_t *x);
|
|
||||||
void append_ul_t(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str, int64_t str_l, ul_ov_t *o, int64_t on, float p_chain_rate, const ug_opt_t *uopt, uint32_t save_bases);
|
void init_all_ul_t(all_ul_t *x, All_reads *hR);
|
||||||
void retrieve_ul_t(UC_Read* i_r, char *i_s, all_ul_t *ref, uint64_t ID, uint8_t strand, int64_t s, int64_t l);
|
void destory_all_ul_t(all_ul_t *x);
|
||||||
void retrieve_u_seq(UC_Read* i_r, char* i_s, ma_utg_t *u, uint8_t strand, int64_t s, int64_t l, void *km);
|
void append_ul_t(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str, int64_t str_l, ul_ov_t *o, int64_t on, float p_chain_rate, const ug_opt_t *uopt, uint32_t save_bases);
|
||||||
void debug_retrieve_rc_sub(const ug_opt_t *uopt, all_ul_t *ref, const All_reads *R_INF, ul_idx_t *ul, uint32_t n_step);
|
void retrieve_ul_t(UC_Read* i_r, char *i_s, all_ul_t *ref, uint64_t ID, uint8_t strand, int64_t s, int64_t l);
|
||||||
uint32_t retrieve_u_cov(const ul_idx_t *ul, uint64_t id, uint8_t strand, uint64_t pos, uint8_t dir, int64_t *pi);
|
void retrieve_u_seq(UC_Read* i_r, char* i_s, ma_utg_t *u, uint8_t strand, int64_t s, int64_t l, void *km);
|
||||||
uint64_t retrieve_u_cov_region(const ul_idx_t *ul, uint64_t id, uint8_t strand, uint64_t s, uint64_t e, int64_t *pi);
|
void debug_retrieve_rc_sub(const ug_opt_t *uopt, all_ul_t *ref, const All_reads *R_INF, ul_idx_t *ul, uint32_t n_step);
|
||||||
uint64_t retrieve_r_cov_region(const ul_idx_t *ul, uint64_t id, uint8_t strand, uint64_t s, uint64_t e, int64_t *pi);
|
uint32_t retrieve_u_cov(const ul_idx_t *ul, uint64_t id, uint8_t strand, uint64_t pos, uint8_t dir, int64_t *pi);
|
||||||
void append_ul_t_back(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str, int64_t str_l, ul_ov_t *o, int64_t on, float p_chain_rate);
|
uint64_t retrieve_u_cov_region(const ul_idx_t *ul, uint64_t id, uint8_t strand, uint64_t s, uint64_t e, int64_t *pi);
|
||||||
void write_compress_base_disk(FILE *fp, uint64_t ul_rid, char *str, uint32_t len, ul_vec_t *buf);
|
uint64_t retrieve_r_cov_region(const ul_idx_t *ul, uint64_t id, uint8_t strand, uint64_t s, uint64_t e, int64_t *pi);
|
||||||
int64_t load_compress_base_disk(FILE *fp, uint64_t *ul_rid, char *dest, uint32_t *len, ul_vec_t *buf);
|
void append_ul_t_back(all_ul_t *x, uint64_t *rid, char* id, int64_t id_l, char* str, int64_t str_l, ul_ov_t *o, int64_t on, float p_chain_rate);
|
||||||
scaf_res_t *init_scaf_res_t(uint32_t n);
|
void write_compress_base_disk(FILE *fp, uint64_t ul_rid, char *str, uint32_t len, ul_vec_t *buf);
|
||||||
void destroy_scaf_res_t(scaf_res_t *p);
|
int64_t load_compress_base_disk(FILE *fp, uint64_t *ul_rid, char *dest, uint32_t *len, ul_vec_t *buf);
|
||||||
void read_ma(ma_hit_t* x, FILE* fp);
|
scaf_res_t *init_scaf_res_t(uint32_t n);
|
||||||
|
void destroy_scaf_res_t(scaf_res_t *p);
|
||||||
const uint64_t sc_tb[8] = {
|
void read_ma(ma_hit_t* x, FILE* fp);
|
||||||
10, 20, 30, 40, 50, 60, 70, 80
|
|
||||||
};
|
const uint64_t sc_tb[8] = {
|
||||||
|
10, 20, 30, 40, 50, 60, 70, 80
|
||||||
#define sc_bn 2
|
};
|
||||||
#define sc_bm ((((uint64_t)1)<<sc_bn)-1)
|
|
||||||
#define sc_wn 5
|
#define sc_bn 2
|
||||||
|
#define sc_bm ((((uint64_t)1)<<sc_bn)-1)
|
||||||
void convert_qual(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitu, uint64_t rev, uint64_t sc_off);
|
#define sc_wn 5
|
||||||
int64_t retrive_bqual(asg8_v *dv, uint8_t *ds, uint64_t id, int64_t s, int64_t e, uint8_t rev, int64_t bitn);
|
|
||||||
void print_fastq(FILE *fp, char *id, char *bs, char *qual, uint64_t bitu, uint64_t sc_off);
|
void convert_qual(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitu, uint64_t rev, uint64_t sc_off);
|
||||||
void ha_compress_qual_bit(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitn);
|
int64_t retrive_bqual(asg8_v *dv, uint8_t *ds, uint64_t id, int64_t s, int64_t e, uint8_t rev, int64_t bitn);
|
||||||
|
void print_fastq(FILE *fp, char *id, char *bs, char *qual, uint64_t bitu, uint64_t sc_off);
|
||||||
#endif
|
void ha_compress_qual_bit(uint8_t* dest, char* src, uint64_t src_l, uint64_t bitn);
|
||||||
|
|
||||||
|
#endif
|
||||||
|
|||||||
+5716
-5716
File diff suppressed because it is too large
Load Diff
+95
-95
@@ -1,96 +1,96 @@
|
|||||||
#ifndef __PURGEDUPS__
|
#ifndef __PURGEDUPS__
|
||||||
#define __PURGEDUPS__
|
#define __PURGEDUPS__
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include "kvec.h"
|
#include "kvec.h"
|
||||||
#include "kdq.h"
|
#include "kdq.h"
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
#include "Hash_Table.h"
|
#include "Hash_Table.h"
|
||||||
#define COV_COUNT 1024
|
#define COV_COUNT 1024
|
||||||
#define HOM_PEAK_RATE 1.25
|
#define HOM_PEAK_RATE 1.25
|
||||||
#define HET_PEAK_RATE (HOM_PEAK_RATE*2)
|
#define HET_PEAK_RATE (HOM_PEAK_RATE*2)
|
||||||
#define ALTER_COV_THRES 0.9
|
#define ALTER_COV_THRES 0.9
|
||||||
#define REAL_ALTER_THRES 0.25
|
#define REAL_ALTER_THRES 0.25
|
||||||
#define CHAIN_FILTER_RATE 0.7
|
#define CHAIN_FILTER_RATE 0.7
|
||||||
#define REV_W 8
|
#define REV_W 8
|
||||||
|
|
||||||
#define SELF_EXIST 0
|
#define SELF_EXIST 0
|
||||||
#define REVE_EXIST 1
|
#define REVE_EXIST 1
|
||||||
#define DELETE 2
|
#define DELETE 2
|
||||||
#define MIXED 3
|
#define MIXED 3
|
||||||
#define FLIP 4
|
#define FLIP 4
|
||||||
|
|
||||||
#define X2Y 0
|
#define X2Y 0
|
||||||
#define Y2X 1
|
#define Y2X 1
|
||||||
#define XCY 2
|
#define XCY 2
|
||||||
#define YCX 3
|
#define YCX 3
|
||||||
|
|
||||||
#define Cal_Off(OFF) ((long long)((uint32_t)((OFF)>>32)) - (long long)((uint32_t)((OFF))))
|
#define Cal_Off(OFF) ((long long)((uint32_t)((OFF)>>32)) - (long long)((uint32_t)((OFF))))
|
||||||
#define Get_xOff(OFF) ((long long)((uint32_t)((OFF)>>32)))
|
#define Get_xOff(OFF) ((long long)((uint32_t)((OFF)>>32)))
|
||||||
#define Get_yOff(OFF) ((long long)((uint32_t)((OFF))))
|
#define Get_yOff(OFF) ((long long)((uint32_t)((OFF))))
|
||||||
#define Get_match(x) ((x).weight)
|
#define Get_match(x) ((x).weight)
|
||||||
#define Get_total(x) ((x).index_beg)
|
#define Get_total(x) ((x).index_beg)
|
||||||
#define Get_type(x) ((x).index_end)
|
#define Get_type(x) ((x).index_end)
|
||||||
#define Get_x_beg(x) ((x).x_beg_pos)
|
#define Get_x_beg(x) ((x).x_beg_pos)
|
||||||
#define Get_x_end(x) ((x).x_end_pos)
|
#define Get_x_end(x) ((x).x_end_pos)
|
||||||
#define Get_y_beg(x) ((x).y_beg_pos)
|
#define Get_y_beg(x) ((x).y_beg_pos)
|
||||||
#define Get_y_end(x) ((x).y_end_pos)
|
#define Get_y_end(x) ((x).y_end_pos)
|
||||||
#define Get_rev(x) ((x).rev)
|
#define Get_rev(x) ((x).rev)
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint8_t rev;
|
uint8_t rev;
|
||||||
uint8_t type;
|
uint8_t type;
|
||||||
uint8_t status;
|
uint8_t status;
|
||||||
uint32_t x_beg_pos;
|
uint32_t x_beg_pos;
|
||||||
uint32_t x_end_pos;
|
uint32_t x_end_pos;
|
||||||
uint32_t y_beg_pos;
|
uint32_t y_beg_pos;
|
||||||
uint32_t y_end_pos;
|
uint32_t y_end_pos;
|
||||||
uint32_t x_beg_id;
|
uint32_t x_beg_id;
|
||||||
uint32_t x_end_id;
|
uint32_t x_end_id;
|
||||||
uint32_t y_beg_id;
|
uint32_t y_beg_id;
|
||||||
uint32_t y_end_id;
|
uint32_t y_end_id;
|
||||||
uint32_t xUid;
|
uint32_t xUid;
|
||||||
uint32_t yUid;
|
uint32_t yUid;
|
||||||
uint32_t weight;
|
uint32_t weight;
|
||||||
long long score;
|
long long score;
|
||||||
float s;
|
float s;
|
||||||
}hap_overlaps;
|
}hap_overlaps;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
kvec_t(hap_overlaps) a;
|
kvec_t(hap_overlaps) a;
|
||||||
}kvec_hap_overlaps;
|
}kvec_hap_overlaps;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
kvec_hap_overlaps* x;
|
kvec_hap_overlaps* x;
|
||||||
uint32_t num;
|
uint32_t num;
|
||||||
}hap_overlaps_list;
|
}hap_overlaps_list;
|
||||||
|
|
||||||
void purge_dups(ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut, ma_hit_t_alloc* sources,
|
void purge_dups(ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut, ma_hit_t_alloc* sources,
|
||||||
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, kvec_asg_arc_t_warp* edge, float density,
|
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, kvec_asg_arc_t_warp* edge, float density,
|
||||||
uint32_t purege_minLen, int max_hang, int min_ovlp, float drop_ratio, uint32_t just_contain,
|
uint32_t purege_minLen, int max_hang, int min_ovlp, float drop_ratio, uint32_t just_contain,
|
||||||
uint32_t just_coverage, hap_cov_t *cov, uint32_t collect_p_trans, uint32_t collect_p_trans_f);
|
uint32_t just_coverage, hap_cov_t *cov, uint32_t collect_p_trans, uint32_t collect_p_trans_f);
|
||||||
void fill_unitig(uint64_t* buffer, uint32_t bufferLen, asg_t* read_g, kvec_asg_arc_t_warp* edge,
|
void fill_unitig(uint64_t* buffer, uint32_t bufferLen, asg_t* read_g, kvec_asg_arc_t_warp* edge,
|
||||||
uint32_t is_circle, uint64_t* rLen);
|
uint32_t is_circle, uint64_t* rLen);
|
||||||
void get_contig_length(ma_ug_t *ug, asg_t *g, uint64_t* primaryLen, uint64_t* alterLen);
|
void get_contig_length(ma_ug_t *ug, asg_t *g, uint64_t* primaryLen, uint64_t* alterLen);
|
||||||
void enable_debug_mode(uint32_t mode);
|
void enable_debug_mode(uint32_t mode);
|
||||||
hap_cov_t* init_hap_cov_t(ma_ug_t *ug, asg_t* read_g, ma_hit_t_alloc* sources, R_to_U* ruIndex, ma_hit_t_alloc* reverse_sources,
|
hap_cov_t* init_hap_cov_t(ma_ug_t *ug, asg_t* read_g, ma_hit_t_alloc* sources, R_to_U* ruIndex, ma_hit_t_alloc* reverse_sources,
|
||||||
ma_sub_t *coverage_cut, int max_hang, int min_ovlp, uint32_t is_collect_trans);
|
ma_sub_t *coverage_cut, int max_hang, int min_ovlp, uint32_t is_collect_trans);
|
||||||
void destory_hap_cov_t(hap_cov_t **x);
|
void destory_hap_cov_t(hap_cov_t **x);
|
||||||
void chain_trans_ovlp(hap_cov_t *cov, utg_trans_t *o, ma_ug_t *ug, asg_t *read_sg, buf_t* xReads, uint32_t targetBaseLen, uint32_t* xEnd);
|
void chain_trans_ovlp(hap_cov_t *cov, utg_trans_t *o, ma_ug_t *ug, asg_t *read_sg, buf_t* xReads, uint32_t targetBaseLen, uint32_t* xEnd);
|
||||||
int get_specific_hap_overlap(kvec_hap_overlaps* x, uint32_t qn, uint32_t tn);
|
int get_specific_hap_overlap(kvec_hap_overlaps* x, uint32_t qn, uint32_t tn);
|
||||||
void set_reverse_hap_overlap(hap_overlaps* dest, hap_overlaps* source, uint32_t* types);
|
void set_reverse_hap_overlap(hap_overlaps* dest, hap_overlaps* source, uint32_t* types);
|
||||||
void print_hap_paf(ma_ug_t *ug, hap_overlaps* ovlp);
|
void print_hap_paf(ma_ug_t *ug, hap_overlaps* ovlp);
|
||||||
uint64_t get_xy_pos_by_pos(asg_t *read_g, asg_arc_t* t, uint32_t v_in_unitig, uint32_t w_in_unitig,
|
uint64_t get_xy_pos_by_pos(asg_t *read_g, asg_arc_t* t, uint32_t v_in_unitig, uint32_t w_in_unitig,
|
||||||
uint32_t v_in_pos, uint32_t w_in_pos, uint32_t xUnitigLen, uint32_t yUnitigLen, uint8_t* rev);
|
uint32_t v_in_pos, uint32_t w_in_pos, uint32_t xUnitigLen, uint32_t yUnitigLen, uint8_t* rev);
|
||||||
void quick_LIS(asg_arc_t_offset* x, uint32_t n, kvec_t_i32_warp* tailIndex, kvec_t_i32_warp* prevIndex);
|
void quick_LIS(asg_arc_t_offset* x, uint32_t n, kvec_t_i32_warp* tailIndex, kvec_t_i32_warp* prevIndex);
|
||||||
uint32_t classify_hap_overlap(long long xBeg, long long xEnd, long long xLen,
|
uint32_t classify_hap_overlap(long long xBeg, long long xEnd, long long xLen,
|
||||||
long long yBeg, long long yEnd, long long yLen, long long* r_xBeg, long long* r_xEnd,
|
long long yBeg, long long yEnd, long long yLen, long long* r_xBeg, long long* r_xEnd,
|
||||||
long long* r_yBeg, long long* r_yEnd);
|
long long* r_yBeg, long long* r_yEnd);
|
||||||
int cmp_hap_alignment_chaining(const void * a, const void * b);
|
int cmp_hap_alignment_chaining(const void * a, const void * b);
|
||||||
uint32_t classify_hap_overlap(long long xBeg, long long xEnd, long long xLen,
|
uint32_t classify_hap_overlap(long long xBeg, long long xEnd, long long xLen,
|
||||||
long long yBeg, long long yEnd, long long yLen, long long* r_xBeg, long long* r_xEnd,
|
long long yBeg, long long yEnd, long long yLen, long long* r_xBeg, long long* r_xEnd,
|
||||||
long long* r_yBeg, long long* r_yEnd);
|
long long* r_yBeg, long long* r_yEnd);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
@@ -1,296 +1,296 @@
|
|||||||
## <a name="started"></a>Getting Started
|
## <a name="started"></a>Getting Started
|
||||||
|
|
||||||
```sh
|
```sh
|
||||||
# Install hifiasm (requiring g++ and zlib)
|
# Install hifiasm (requiring g++ and zlib)
|
||||||
git clone https://github.com/chhylp123/hifiasm
|
git clone https://github.com/chhylp123/hifiasm
|
||||||
cd hifiasm && make
|
cd hifiasm && make
|
||||||
|
|
||||||
# Run on test data (use -f0 for small datasets)
|
# Run on test data (use -f0 for small datasets)
|
||||||
wget https://github.com/chhylp123/hifiasm/releases/download/v0.7/chr11-2M.fa.gz
|
wget https://github.com/chhylp123/hifiasm/releases/download/v0.7/chr11-2M.fa.gz
|
||||||
./hifiasm -o test -t4 -f0 chr11-2M.fa.gz 2> test.log
|
./hifiasm -o test -t4 -f0 chr11-2M.fa.gz 2> test.log
|
||||||
awk '/^S/{print ">"$2;print $3}' test.bp.p_ctg.gfa > test.p_ctg.fa # get primary contigs in FASTA
|
awk '/^S/{print ">"$2;print $3}' test.bp.p_ctg.gfa > test.p_ctg.fa # get primary contigs in FASTA
|
||||||
|
|
||||||
# Assemble inbred/homozygous genomes (-l0 disables duplication purging)
|
# Assemble inbred/homozygous genomes (-l0 disables duplication purging)
|
||||||
hifiasm -o CHM13.asm -t32 -l0 CHM13-HiFi.fa.gz 2> CHM13.asm.log
|
hifiasm -o CHM13.asm -t32 -l0 CHM13-HiFi.fa.gz 2> CHM13.asm.log
|
||||||
# Assemble heterozygous genomes with built-in duplication purging
|
# Assemble heterozygous genomes with built-in duplication purging
|
||||||
hifiasm -o HG002.asm -t32 HG002-file1.fq.gz HG002-file2.fq.gz
|
hifiasm -o HG002.asm -t32 HG002-file1.fq.gz HG002-file2.fq.gz
|
||||||
|
|
||||||
# Assemble genomes with ONT R10 reads rather than PacBio HiFi reads using the latest release of hifiasm (>0.21.0-r686)
|
# Assemble genomes with ONT R10 reads rather than PacBio HiFi reads using the latest release of hifiasm (>0.21.0-r686)
|
||||||
hifiasm -o HG002.asm --ont -t32 HG002-ont.fq.gz
|
hifiasm -o HG002.asm --ont -t32 HG002-ont.fq.gz
|
||||||
|
|
||||||
# Hi-C phasing with paired-end short reads in two FASTQ files
|
# Hi-C phasing with paired-end short reads in two FASTQ files
|
||||||
hifiasm -o HG002.asm --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
hifiasm -o HG002.asm --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
||||||
|
|
||||||
# Trio binning assembly (requiring https://github.com/lh3/yak)
|
# Trio binning assembly (requiring https://github.com/lh3/yak)
|
||||||
yak count -b37 -t16 -o pat.yak <(cat pat_1.fq.gz pat_2.fq.gz) <(cat pat_1.fq.gz pat_2.fq.gz)
|
yak count -b37 -t16 -o pat.yak <(cat pat_1.fq.gz pat_2.fq.gz) <(cat pat_1.fq.gz pat_2.fq.gz)
|
||||||
yak count -b37 -t16 -o mat.yak <(cat mat_1.fq.gz mat_2.fq.gz) <(cat mat_1.fq.gz mat_2.fq.gz)
|
yak count -b37 -t16 -o mat.yak <(cat mat_1.fq.gz mat_2.fq.gz) <(cat mat_1.fq.gz mat_2.fq.gz)
|
||||||
hifiasm -o HG002.asm -t32 -1 pat.yak -2 mat.yak HG002-HiFi.fa.gz
|
hifiasm -o HG002.asm -t32 -1 pat.yak -2 mat.yak HG002-HiFi.fa.gz
|
||||||
|
|
||||||
# Improve contiguity for diploid genome assembly by self-scaffolding (`--dual-scaf`)
|
# Improve contiguity for diploid genome assembly by self-scaffolding (`--dual-scaf`)
|
||||||
hifiasm -o HG002.asm --dual-scaf --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
hifiasm -o HG002.asm --dual-scaf --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
||||||
|
|
||||||
# Preserve more telomeres for human genomes (`--telo-m CCCTAA`)
|
# Preserve more telomeres for human genomes (`--telo-m CCCTAA`)
|
||||||
hifiasm -o HG002.asm --telo-m CCCTAA --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
hifiasm -o HG002.asm --telo-m CCCTAA --h1 read1.fq.gz --h2 read2.fq.gz HG002-HiFi.fq.gz
|
||||||
|
|
||||||
# Hybrid assembly with HiFi, ultralong and Hi-C reads
|
# Hybrid assembly with HiFi, ultralong and Hi-C reads
|
||||||
hifiasm -o HG002.asm --h1 read1.fq.gz --h2 read2.fq.gz --ul ul.fq.gz HG002-HiFi.fq.gz
|
hifiasm -o HG002.asm --h1 read1.fq.gz --h2 read2.fq.gz --ul ul.fq.gz HG002-HiFi.fq.gz
|
||||||
|
|
||||||
# Single-sample telomere-to-telomere assembly for diploid human genomes
|
# Single-sample telomere-to-telomere assembly for diploid human genomes
|
||||||
hifiasm -o HG002.asm --dual-scaf --telo-m CCCTAA --h1 read1.fq.gz --h2 read2.fq.gz --ul ul.fq.gz HG002-HiFi.fq.gz
|
hifiasm -o HG002.asm --dual-scaf --telo-m CCCTAA --h1 read1.fq.gz --h2 read2.fq.gz --ul ul.fq.gz HG002-HiFi.fq.gz
|
||||||
|
|
||||||
```
|
```
|
||||||
See [tutorial][tutorial] for more details.
|
See [tutorial][tutorial] for more details.
|
||||||
|
|
||||||
## Table of Contents
|
## Table of Contents
|
||||||
|
|
||||||
- [Getting Started](#started)
|
- [Getting Started](#started)
|
||||||
- [Introduction](#intro)
|
- [Introduction](#intro)
|
||||||
- [Why Hifiasm?](#why)
|
- [Why Hifiasm?](#why)
|
||||||
- [Usage](#use)
|
- [Usage](#use)
|
||||||
- [Assembling HiFi reads without additional data types](#hifionly)
|
- [Assembling HiFi reads without additional data types](#hifionly)
|
||||||
- [Assembling ONT reads](#ontonly)
|
- [Assembling ONT reads](#ontonly)
|
||||||
- [Hi-C integration](#hic)
|
- [Hi-C integration](#hic)
|
||||||
- [Trio binning](#trio)
|
- [Trio binning](#trio)
|
||||||
- [Ultra-long ONT integration](#ul)
|
- [Ultra-long ONT integration](#ul)
|
||||||
- [Output files](#output)
|
- [Output files](#output)
|
||||||
- [Results](#results)
|
- [Results](#results)
|
||||||
- [Getting Help](#help)
|
- [Getting Help](#help)
|
||||||
- [Limitations](#limit)
|
- [Limitations](#limit)
|
||||||
- [Citing Hifiasm](#cite)
|
- [Citing Hifiasm](#cite)
|
||||||
|
|
||||||
## <a name="intro"></a>Introduction
|
## <a name="intro"></a>Introduction
|
||||||
|
|
||||||
Hifiasm is a fast haplotype-resolved de novo assembler initially designed for PacBio HiFi reads.
|
Hifiasm is a fast haplotype-resolved de novo assembler initially designed for PacBio HiFi reads.
|
||||||
Its latest release could support the telomere-to-telomere assembly by utilizing ultralong Oxford Nanopore reads. Hifiasm produces arguably the best single-sample telomere-to-telomere assemblies combing HiFi, ultralong and Hi-C reads, and it is one of the best haplotype-resolved assemblers for the trio-binning assembly given parental short reads. For a human genome, hifiasm can produce the telomere-to-telomere assembly in one day.
|
Its latest release could support the telomere-to-telomere assembly by utilizing ultralong Oxford Nanopore reads. Hifiasm produces arguably the best single-sample telomere-to-telomere assemblies combing HiFi, ultralong and Hi-C reads, and it is one of the best haplotype-resolved assemblers for the trio-binning assembly given parental short reads. For a human genome, hifiasm can produce the telomere-to-telomere assembly in one day.
|
||||||
|
|
||||||
## <a name="why"></a>Why Hifiasm?
|
## <a name="why"></a>Why Hifiasm?
|
||||||
|
|
||||||
* Hifiasm delivers high-quality telomere-to-telomere assemblies. It tends to generate longer contigs
|
* Hifiasm delivers high-quality telomere-to-telomere assemblies. It tends to generate longer contigs
|
||||||
and resolve more segmental duplications than other assemblers.
|
and resolve more segmental duplications than other assemblers.
|
||||||
|
|
||||||
* Given Hi-C reads or short reads from the parents, hifiasm can produce overall the best
|
* Given Hi-C reads or short reads from the parents, hifiasm can produce overall the best
|
||||||
haplotype-resolved assembly so far. It is the assembler of choice by the
|
haplotype-resolved assembly so far. It is the assembler of choice by the
|
||||||
[Human Pangenome Project][hpp] for the first batch of samples.
|
[Human Pangenome Project][hpp] for the first batch of samples.
|
||||||
|
|
||||||
* Hifiasm can purge duplications between haplotigs without relying on
|
* Hifiasm can purge duplications between haplotigs without relying on
|
||||||
third-party tools such as purge\_dups. Hifiasm does not need polishing tools
|
third-party tools such as purge\_dups. Hifiasm does not need polishing tools
|
||||||
like pilon or racon, either. This simplifies the assembly pipeline and saves
|
like pilon or racon, either. This simplifies the assembly pipeline and saves
|
||||||
running time.
|
running time.
|
||||||
|
|
||||||
* Hifiasm is fast. It can assemble a human genome in half a day and assemble a
|
* Hifiasm is fast. It can assemble a human genome in half a day and assemble a
|
||||||
~30Gb redwood genome in three days. No genome is too large for hifiasm.
|
~30Gb redwood genome in three days. No genome is too large for hifiasm.
|
||||||
|
|
||||||
* Hifiasm is trivial to install and easy to use. It does not required Python,
|
* Hifiasm is trivial to install and easy to use. It does not required Python,
|
||||||
R or C++11 compilers, and can be compiled into a single executable. The
|
R or C++11 compilers, and can be compiled into a single executable. The
|
||||||
default setting works well with a variety of genomes.
|
default setting works well with a variety of genomes.
|
||||||
|
|
||||||
[hpp]: https://humanpangenome.org
|
[hpp]: https://humanpangenome.org
|
||||||
|
|
||||||
## <a name="use"></a>Usage
|
## <a name="use"></a>Usage
|
||||||
|
|
||||||
### <a name="hifionly"></a>Assembling HiFi reads without additional data types
|
### <a name="hifionly"></a>Assembling HiFi reads without additional data types
|
||||||
|
|
||||||
A typical hifiasm command line looks like:
|
A typical hifiasm command line looks like:
|
||||||
```sh
|
```sh
|
||||||
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
||||||
```
|
```
|
||||||
where `NA12878.fq.gz` provides the input reads, `-t` sets the number of CPUs in
|
where `NA12878.fq.gz` provides the input reads, `-t` sets the number of CPUs in
|
||||||
use and `-o` specifies the prefix of output files. For this example, the
|
use and `-o` specifies the prefix of output files. For this example, the
|
||||||
primary contigs are written to `NA12878.asm.bp.p_ctg.gfa`.
|
primary contigs are written to `NA12878.asm.bp.p_ctg.gfa`.
|
||||||
Since v0.15, hifiasm also produces two sets of
|
Since v0.15, hifiasm also produces two sets of
|
||||||
partially phased contigs at `NA12878.asm.bp.hap?.p_ctg.gfa`. This pair of files
|
partially phased contigs at `NA12878.asm.bp.hap?.p_ctg.gfa`. This pair of files
|
||||||
can be thought to represent the two haplotypes in a diploid genome, though with
|
can be thought to represent the two haplotypes in a diploid genome, though with
|
||||||
occasional switch errors. The frequency of switches is determined by the
|
occasional switch errors. The frequency of switches is determined by the
|
||||||
heterozygosity of the input sample.
|
heterozygosity of the input sample.
|
||||||
|
|
||||||
At the first run, hifiasm saves corrected reads and
|
At the first run, hifiasm saves corrected reads and
|
||||||
overlaps to disk as `NA12878.asm.*.bin`. It reuses the saved results to avoid
|
overlaps to disk as `NA12878.asm.*.bin`. It reuses the saved results to avoid
|
||||||
the time-consuming all-vs-all overlap calculation next time. You may specify
|
the time-consuming all-vs-all overlap calculation next time. You may specify
|
||||||
`-i` to ignore precomputed overlaps and redo overlapping from raw reads.
|
`-i` to ignore precomputed overlaps and redo overlapping from raw reads.
|
||||||
You can also dump error corrected reads in FASTA and read overlaps in PAF with
|
You can also dump error corrected reads in FASTA and read overlaps in PAF with
|
||||||
```sh
|
```sh
|
||||||
hifiasm -o NA12878.asm -t 32 --write-paf --write-ec /dev/null
|
hifiasm -o NA12878.asm -t 32 --write-paf --write-ec /dev/null
|
||||||
```
|
```
|
||||||
|
|
||||||
Hifiasm purges haplotig duplications by default. For inbred or homozygous
|
Hifiasm purges haplotig duplications by default. For inbred or homozygous
|
||||||
genomes, you may disable purging with option `-l0`. Old HiFi reads may contain
|
genomes, you may disable purging with option `-l0`. Old HiFi reads may contain
|
||||||
short adapter sequences at the ends of reads. You can specify `-z20` to trim
|
short adapter sequences at the ends of reads. You can specify `-z20` to trim
|
||||||
both ends of reads by 20bp. For small genomes, use `-f0` to disable the initial
|
both ends of reads by 20bp. For small genomes, use `-f0` to disable the initial
|
||||||
bloom filter which takes 16GB memory at the beginning. For genomes much larger
|
bloom filter which takes 16GB memory at the beginning. For genomes much larger
|
||||||
than human, applying `-f38` or even `-f39` is preferred to save memory on k-mer
|
than human, applying `-f38` or even `-f39` is preferred to save memory on k-mer
|
||||||
counting.
|
counting.
|
||||||
|
|
||||||
### <a name="ontonly"></a>Assembling ONT reads
|
### <a name="ontonly"></a>Assembling ONT reads
|
||||||
|
|
||||||
Since version 0.21.0 (r686), hifiasm can support ONT assembly using ONT simplex R10 reads.
|
Since version 0.21.0 (r686), hifiasm can support ONT assembly using ONT simplex R10 reads.
|
||||||
To enable this feature, add the `--ont` option as shown below:
|
To enable this feature, add the `--ont` option as shown below:
|
||||||
```sh
|
```sh
|
||||||
hifiasm -t64 --ont -o ONT.asm ONT.read.fastq.gz
|
hifiasm -t64 --ont -o ONT.asm ONT.read.fastq.gz
|
||||||
```
|
```
|
||||||
Please note that this module requires input reads in FASTQ format.
|
Please note that this module requires input reads in FASTQ format.
|
||||||
|
|
||||||
|
|
||||||
### <a name="hic"></a>Hi-C integration
|
### <a name="hic"></a>Hi-C integration
|
||||||
|
|
||||||
Hifiasm can generate a pair of haplotype-resolved assemblies with paired-end
|
Hifiasm can generate a pair of haplotype-resolved assemblies with paired-end
|
||||||
Hi-C reads:
|
Hi-C reads:
|
||||||
```sh
|
```sh
|
||||||
hifiasm -o NA12878.asm -t32 --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
|
hifiasm -o NA12878.asm -t32 --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
|
||||||
```
|
```
|
||||||
In this mode, each contig is supposed to be a haplotig, which by definition
|
In this mode, each contig is supposed to be a haplotig, which by definition
|
||||||
comes from one parental haplotype only. Hifiasm often puts all contigs from the
|
comes from one parental haplotype only. Hifiasm often puts all contigs from the
|
||||||
same parental chromosome in one assembly. It has cleanly separated chrX and
|
same parental chromosome in one assembly. It has cleanly separated chrX and
|
||||||
chrY for a human male dataset. Nonetheless, phasing across centromeres is
|
chrY for a human male dataset. Nonetheless, phasing across centromeres is
|
||||||
challenging. Hifiasm is often able to phase entire chromosomes but it may fail
|
challenging. Hifiasm is often able to phase entire chromosomes but it may fail
|
||||||
in rare cases. Also, contigs from different parental chromosomes are randomly mixed as
|
in rare cases. Also, contigs from different parental chromosomes are randomly mixed as
|
||||||
it is just not possible to phase across chromosomes with Hi-C.
|
it is just not possible to phase across chromosomes with Hi-C.
|
||||||
|
|
||||||
Hifiasm does not perform scaffolding for now. You need to run a standalone
|
Hifiasm does not perform scaffolding for now. You need to run a standalone
|
||||||
scaffolder such as SALSA or 3D-DNA to scaffold phased haplotigs.
|
scaffolder such as SALSA or 3D-DNA to scaffold phased haplotigs.
|
||||||
|
|
||||||
### <a name="trio"></a>Trio binning
|
### <a name="trio"></a>Trio binning
|
||||||
|
|
||||||
When parental short reads are available, hifiasm can also generate a pair of
|
When parental short reads are available, hifiasm can also generate a pair of
|
||||||
haplotype-resolved assemblies with trio binning. To perform such assembly, you
|
haplotype-resolved assemblies with trio binning. To perform such assembly, you
|
||||||
need to count k-mers first with [yak][yak] first and then do assembly:
|
need to count k-mers first with [yak][yak] first and then do assembly:
|
||||||
```sh
|
```sh
|
||||||
yak count -k31 -b37 -t16 -o pat.yak paternal.fq.gz
|
yak count -k31 -b37 -t16 -o pat.yak paternal.fq.gz
|
||||||
yak count -k31 -b37 -t16 -o mat.yak maternal.fq.gz
|
yak count -k31 -b37 -t16 -o mat.yak maternal.fq.gz
|
||||||
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak NA12878.fq.gz
|
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak NA12878.fq.gz
|
||||||
```
|
```
|
||||||
Here `NA12878.asm.dip.hap1.p_ctg.gfa` and `NA12878.asm.dip.hap2.p_ctg.gfa` give the two
|
Here `NA12878.asm.dip.hap1.p_ctg.gfa` and `NA12878.asm.dip.hap2.p_ctg.gfa` give the two
|
||||||
haplotype assemblies. In the binning mode, hifiasm does not purge haplotig
|
haplotype assemblies. In the binning mode, hifiasm does not purge haplotig
|
||||||
duplicates by default. Because hifiasm reuses saved overlaps, you can
|
duplicates by default. Because hifiasm reuses saved overlaps, you can
|
||||||
generate both primary/alternate assemblies and trio binning assemblies with
|
generate both primary/alternate assemblies and trio binning assemblies with
|
||||||
```sh
|
```sh
|
||||||
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz 2> NA12878.asm.pri.log
|
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz 2> NA12878.asm.pri.log
|
||||||
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak /dev/null 2> NA12878.asm.trio.log
|
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak /dev/null 2> NA12878.asm.trio.log
|
||||||
```
|
```
|
||||||
The second command line will run much faster than the first.
|
The second command line will run much faster than the first.
|
||||||
|
|
||||||
### <a name="ul"></a>Ultra-long ONT integration
|
### <a name="ul"></a>Ultra-long ONT integration
|
||||||
|
|
||||||
Hifiasm could integrate ultra-long ONT reads to produce the telomere-to-telomere assembly:
|
Hifiasm could integrate ultra-long ONT reads to produce the telomere-to-telomere assembly:
|
||||||
```sh
|
```sh
|
||||||
hifiasm -o NA12878.asm -t32 --ul ul.fq.gz HiFi-reads.fq.gz
|
hifiasm -o NA12878.asm -t32 --ul ul.fq.gz HiFi-reads.fq.gz
|
||||||
```
|
```
|
||||||
For the single-sample telomere-to-telomere assembly with Hi-C reads:
|
For the single-sample telomere-to-telomere assembly with Hi-C reads:
|
||||||
```sh
|
```sh
|
||||||
hifiasm -o NA12878.asm -t32 --ul ul.fq.gz --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
|
hifiasm -o NA12878.asm -t32 --ul ul.fq.gz --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
|
||||||
```
|
```
|
||||||
For the trio-binning telomere-to-telomere assembly:
|
For the trio-binning telomere-to-telomere assembly:
|
||||||
```sh
|
```sh
|
||||||
hifiasm -o NA12878.asm -t32 --ul ul.fq.gz -1 pat.yak -2 mat.yak HiFi-reads.fq.gz
|
hifiasm -o NA12878.asm -t32 --ul ul.fq.gz -1 pat.yak -2 mat.yak HiFi-reads.fq.gz
|
||||||
```
|
```
|
||||||
|
|
||||||
### <a name="ul"></a>Self-scaffolding
|
### <a name="ul"></a>Self-scaffolding
|
||||||
|
|
||||||
For diploid haplotype-resolved genome assembly, hifiasm can further enhance assembly contiguity
|
For diploid haplotype-resolved genome assembly, hifiasm can further enhance assembly contiguity
|
||||||
by introducing scaffolding. It leverages the assemblies of the two haplotypes to scaffold each other.
|
by introducing scaffolding. It leverages the assemblies of the two haplotypes to scaffold each other.
|
||||||
Specifically, if there is a gap within the haplotype 1 assembly, hifiasm will use the corresponding
|
Specifically, if there is a gap within the haplotype 1 assembly, hifiasm will use the corresponding
|
||||||
homologous region in haplotype 2 to scaffold haplotype 1. Below is an example using the `--dual-scaf` option.
|
homologous region in haplotype 2 to scaffold haplotype 1. Below is an example using the `--dual-scaf` option.
|
||||||
```sh
|
```sh
|
||||||
hifiasm -o NA12878.asm -t32 --dual-scaf HiFi-reads.fq.gz
|
hifiasm -o NA12878.asm -t32 --dual-scaf HiFi-reads.fq.gz
|
||||||
```
|
```
|
||||||
|
|
||||||
### <a name="ul"></a>Preserve more telomeres for T2T assemblies
|
### <a name="ul"></a>Preserve more telomeres for T2T assemblies
|
||||||
|
|
||||||
Hifiasm can preserve more telomeres by specifying the telomere motif using the `--telo-m` option.
|
Hifiasm can preserve more telomeres by specifying the telomere motif using the `--telo-m` option.
|
||||||
Below is an example applied to human genome assembly.
|
Below is an example applied to human genome assembly.
|
||||||
```sh
|
```sh
|
||||||
hifiasm -o NA12878.asm -t32 --telo-m CCCTAA HiFi-reads.fq.gz
|
hifiasm -o NA12878.asm -t32 --telo-m CCCTAA HiFi-reads.fq.gz
|
||||||
```
|
```
|
||||||
|
|
||||||
### <a name="output"></a>Output files
|
### <a name="output"></a>Output files
|
||||||
|
|
||||||
Hifiasm generates different types of assemblies based on the input data.
|
Hifiasm generates different types of assemblies based on the input data.
|
||||||
It also writes error corrected reads to the *prefix*.ec.bin binary file and
|
It also writes error corrected reads to the *prefix*.ec.bin binary file and
|
||||||
writes overlaps to *prefix*.ovlp.source.bin and *prefix*.ovlp.reverse.bin.
|
writes overlaps to *prefix*.ovlp.source.bin and *prefix*.ovlp.reverse.bin.
|
||||||
For more details, please see the complete [documentation][tutorial_output].
|
For more details, please see the complete [documentation][tutorial_output].
|
||||||
|
|
||||||
## <a name="results"></a>Results
|
## <a name="results"></a>Results
|
||||||
|
|
||||||
The following table shows the statistics of several hifiasm primary assemblies assembled with v0.12:
|
The following table shows the statistics of several hifiasm primary assemblies assembled with v0.12:
|
||||||
|
|
||||||
|<sub>Dataset<sub>|<sub>Size<sub>|<sub>Cov.<sub>|<sub>Asm options<sub>|<sub>CPU time<sub>|<sub>Wall time<sub>|<sub>RAM<sub>|<sub> N50<sub>|
|
|<sub>Dataset<sub>|<sub>Size<sub>|<sub>Cov.<sub>|<sub>Asm options<sub>|<sub>CPU time<sub>|<sub>Wall time<sub>|<sub>RAM<sub>|<sub> N50<sub>|
|
||||||
|:---------------|-----:|-----:|:---------------------|-------:|--------:|----:|----------------:|
|
|:---------------|-----:|-----:|:---------------------|-------:|--------:|----:|----------------:|
|
||||||
|<sub>[Mouse (C57/BL6J)][mouse-data]</sub>|<sub>2.6Gb</sub> |<sub>×25</sub>|<sub>-t48 -l0</sub> |<sub>172.9h</sub> |<sub>4.8h</sub> |<sub>76G</sub> |<sub>21.1Mb</sub>|
|
|<sub>[Mouse (C57/BL6J)][mouse-data]</sub>|<sub>2.6Gb</sub> |<sub>×25</sub>|<sub>-t48 -l0</sub> |<sub>172.9h</sub> |<sub>4.8h</sub> |<sub>76G</sub> |<sub>21.1Mb</sub>|
|
||||||
|<sub>[Maize (B73)][maize-data]</sub> |<sub>2.2Gb</sub> |<sub>×22</sub>|<sub>-t48 -l0</sub> |<sub>203.2h</sub> |<sub>5.1h</sub> |<sub>68G</sub> |<sub>36.7Mb</sub>|
|
|<sub>[Maize (B73)][maize-data]</sub> |<sub>2.2Gb</sub> |<sub>×22</sub>|<sub>-t48 -l0</sub> |<sub>203.2h</sub> |<sub>5.1h</sub> |<sub>68G</sub> |<sub>36.7Mb</sub>|
|
||||||
|<sub>[Strawberry][strawberry-data]</sub> |<sub>0.8Gb</sub> |<sub>×36</sub>|<sub>-t48 -D10</sub>|<sub>152.7h</sub> |<sub>3.7h</sub> |<sub>91G</sub> |<sub>17.8Mb</sub>|
|
|<sub>[Strawberry][strawberry-data]</sub> |<sub>0.8Gb</sub> |<sub>×36</sub>|<sub>-t48 -D10</sub>|<sub>152.7h</sub> |<sub>3.7h</sub> |<sub>91G</sub> |<sub>17.8Mb</sub>|
|
||||||
|<sub>[Frog][frog-data]</sub> |<sub>9.5Gb</sub> |<sub>×29</sub>|<sub>-t48</sub> |<sub>2834.3h</sub>|<sub>69.0h</sub>|<sub>463G</sub>|<sub>9.3Mb</sub>|
|
|<sub>[Frog][frog-data]</sub> |<sub>9.5Gb</sub> |<sub>×29</sub>|<sub>-t48</sub> |<sub>2834.3h</sub>|<sub>69.0h</sub>|<sub>463G</sub>|<sub>9.3Mb</sub>|
|
||||||
|<sub>[Redwood][redwood-data]</sub> |<sub>35.6Gb</sub>|<sub>×28</sub>|<sub>-t80</sub> |<sub>3890.3h</sub>|<sub>65.5h</sub>|<sub>699G</sub>|<sub>5.4Mb</sub>|
|
|<sub>[Redwood][redwood-data]</sub> |<sub>35.6Gb</sub>|<sub>×28</sub>|<sub>-t80</sub> |<sub>3890.3h</sub>|<sub>65.5h</sub>|<sub>699G</sub>|<sub>5.4Mb</sub>|
|
||||||
|<sub>[Human (CHM13)][CHM13-data]</sub> |<sub>3.1Gb</sub> |<sub>×32</sub>|<sub>-t48 -l0</sub> |<sub>310.7h</sub> |<sub>8.2h</sub> |<sub>114G</sub>|<sub>88.9Mb</sub>|
|
|<sub>[Human (CHM13)][CHM13-data]</sub> |<sub>3.1Gb</sub> |<sub>×32</sub>|<sub>-t48 -l0</sub> |<sub>310.7h</sub> |<sub>8.2h</sub> |<sub>114G</sub>|<sub>88.9Mb</sub>|
|
||||||
|<sub>[Human (HG00733)][HG00733-data]</sub>|<sub>3.1Gb</sub>|<sub>×33</sub>|<sub>-t48</sub> |<sub>269.1h</sub> |<sub>6.9h</sub> |<sub>135G</sub>|<sub>69.9Mb</sub>|
|
|<sub>[Human (HG00733)][HG00733-data]</sub>|<sub>3.1Gb</sub>|<sub>×33</sub>|<sub>-t48</sub> |<sub>269.1h</sub> |<sub>6.9h</sub> |<sub>135G</sub>|<sub>69.9Mb</sub>|
|
||||||
|<sub>[Human (HG002)][NA24385-data]</sub> |<sub>3.1Gb</sub> |<sub>×36</sub>|<sub>-t48</sub> |<sub>305.4h</sub> |<sub>7.7h</sub> |<sub>137G</sub>|<sub>98.7Mb</sub>|
|
|<sub>[Human (HG002)][NA24385-data]</sub> |<sub>3.1Gb</sub> |<sub>×36</sub>|<sub>-t48</sub> |<sub>305.4h</sub> |<sub>7.7h</sub> |<sub>137G</sub>|<sub>98.7Mb</sub>|
|
||||||
|
|
||||||
[mouse-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRR11606870
|
[mouse-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRR11606870
|
||||||
[maize-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRR11606869
|
[maize-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRR11606869
|
||||||
[strawberry-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRR11606867
|
[strawberry-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRR11606867
|
||||||
[frog-data]: https://www.ncbi.nlm.nih.gov/sra?term=(SRR11606868)%20OR%20SRR12048570
|
[frog-data]: https://www.ncbi.nlm.nih.gov/sra?term=(SRR11606868)%20OR%20SRR12048570
|
||||||
[redwood-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRP251156
|
[redwood-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRP251156
|
||||||
[CHM13-data]: https://www.ncbi.nlm.nih.gov/sra?term=(((SRR11292120)%20OR%20SRR11292121)%20OR%20SRR11292122)%20OR%20SRR11292123
|
[CHM13-data]: https://www.ncbi.nlm.nih.gov/sra?term=(((SRR11292120)%20OR%20SRR11292121)%20OR%20SRR11292122)%20OR%20SRR11292123
|
||||||
|
|
||||||
Hifiasm can assemble a 3.1Gb human genome in several hours or a ~30Gb hexaploid
|
Hifiasm can assemble a 3.1Gb human genome in several hours or a ~30Gb hexaploid
|
||||||
redwood genome in a few days on a single machine. For trio binning assembly:
|
redwood genome in a few days on a single machine. For trio binning assembly:
|
||||||
|
|
||||||
|<sub>Dataset<sub>|<sub>Cov.<sub>|<sub>CPU time<sub>|<sub>Elapsed time<sub>|<sub>RAM<sub>|<sub> N50<sub>|
|
|<sub>Dataset<sub>|<sub>Cov.<sub>|<sub>CPU time<sub>|<sub>Elapsed time<sub>|<sub>RAM<sub>|<sub> N50<sub>|
|
||||||
|:---------------|-----:|-------:|--------:|----:|----------------:|
|
|:---------------|-----:|-------:|--------:|----:|----------------:|
|
||||||
|<sub>[HG00733][HG00733-data], [\[father\]][HG00731-data], [\[mother\]][HG00732-data]</sub>|<sub>×33</sub>|<sub>269.1h</sub>|<sub>6.9h</sub>|<sub>135G</sub>|<sub>35.1Mb (paternal), 34.9Mb (maternal)</sub>|
|
|<sub>[HG00733][HG00733-data], [\[father\]][HG00731-data], [\[mother\]][HG00732-data]</sub>|<sub>×33</sub>|<sub>269.1h</sub>|<sub>6.9h</sub>|<sub>135G</sub>|<sub>35.1Mb (paternal), 34.9Mb (maternal)</sub>|
|
||||||
|<sub>[HG002][NA24385-data], [\[father\]][NA24149-data], [\[mother\]][NA24143-data]</sup>|<sub>×36</sub>|<sub>305.4h</sub>|<sub>7.7h</sub>|<sub>137G</sub>|<sub>41.0Mb (paternal), 40.8Mb (maternal)</sub>|
|
|<sub>[HG002][NA24385-data], [\[father\]][NA24149-data], [\[mother\]][NA24143-data]</sup>|<sub>×36</sub>|<sub>305.4h</sub>|<sub>7.7h</sub>|<sub>137G</sub>|<sub>41.0Mb (paternal), 40.8Mb (maternal)</sub>|
|
||||||
|
|
||||||
<!--
|
<!--
|
||||||
|<sub>[NA12878][NA12878-data], [\[father\]][NA12891-data], [\[mother\]][NA12892-data]</sub>|<sub>×30</sub>|<sub>180.8h</sub>|<sub>4.9h</sub>|<sub>123G</sub>|<sub>27.7Mb (paternal), 27.0Mb (maternal)</sub>|
|
|<sub>[NA12878][NA12878-data], [\[father\]][NA12891-data], [\[mother\]][NA12892-data]</sub>|<sub>×30</sub>|<sub>180.8h</sub>|<sub>4.9h</sub>|<sub>123G</sub>|<sub>27.7Mb (paternal), 27.0Mb (maternal)</sub>|
|
||||||
-->
|
-->
|
||||||
|
|
||||||
[HG00733-data]: https://www.ebi.ac.uk/ena/data/view/ERX3831682
|
[HG00733-data]: https://www.ebi.ac.uk/ena/data/view/ERX3831682
|
||||||
[HG00731-data]: https://www.ebi.ac.uk/ena/data/view/ERR3241754
|
[HG00731-data]: https://www.ebi.ac.uk/ena/data/view/ERR3241754
|
||||||
[HG00732-data]: https://www.ebi.ac.uk/ena/data/view/ERR3241755
|
[HG00732-data]: https://www.ebi.ac.uk/ena/data/view/ERR3241755
|
||||||
[NA24385-data]: https://www.ncbi.nlm.nih.gov/sra?term=(((SRR10382244)%20OR%20SRR10382245)%20OR%20SRR10382248)%20OR%20SRR10382249
|
[NA24385-data]: https://www.ncbi.nlm.nih.gov/sra?term=(((SRR10382244)%20OR%20SRR10382245)%20OR%20SRR10382248)%20OR%20SRR10382249
|
||||||
[NA24149-data]: https://ftp-trace.ncbi.nlm.nih.gov/giab/ftp/data/AshkenazimTrio/HG003_NA24149_father/NIST_HiSeq_HG003_Homogeneity-12389378/HG003Run01-13262252/
|
[NA24149-data]: https://ftp-trace.ncbi.nlm.nih.gov/giab/ftp/data/AshkenazimTrio/HG003_NA24149_father/NIST_HiSeq_HG003_Homogeneity-12389378/HG003Run01-13262252/
|
||||||
[NA24143-data]: https://ftp-trace.ncbi.nlm.nih.gov/giab/ftp/data/AshkenazimTrio/HG004_NA24143_mother/NIST_HiSeq_HG004_Homogeneity-14572558/HG004Run01-15133132/
|
[NA24143-data]: https://ftp-trace.ncbi.nlm.nih.gov/giab/ftp/data/AshkenazimTrio/HG004_NA24143_mother/NIST_HiSeq_HG004_Homogeneity-14572558/HG004Run01-15133132/
|
||||||
[NA12878-data]: https://ftp-trace.ncbi.nlm.nih.gov/giab/ftp/data/NA12878/PacBio_SequelII_CCS_11kb/
|
[NA12878-data]: https://ftp-trace.ncbi.nlm.nih.gov/giab/ftp/data/NA12878/PacBio_SequelII_CCS_11kb/
|
||||||
[NA12891-data]: https://www.ebi.ac.uk/ena/data/view/ERR194160
|
[NA12891-data]: https://www.ebi.ac.uk/ena/data/view/ERR194160
|
||||||
[NA12892-data]: https://www.ebi.ac.uk/ena/data/view/ERR194161
|
[NA12892-data]: https://www.ebi.ac.uk/ena/data/view/ERR194161
|
||||||
|
|
||||||
Human assemblies above can be acquired [from Zenodo][zenodo-human] and
|
Human assemblies above can be acquired [from Zenodo][zenodo-human] and
|
||||||
non-human ones are available [here][zenodo-nonh].
|
non-human ones are available [here][zenodo-nonh].
|
||||||
|
|
||||||
[zenodo-human]: https://zenodo.org/record/4393631
|
[zenodo-human]: https://zenodo.org/record/4393631
|
||||||
[zenodo-nonh]: https://zenodo.org/record/4393750
|
[zenodo-nonh]: https://zenodo.org/record/4393750
|
||||||
[unitig]: http://wgs-assembler.sourceforge.net/wiki/index.php/Celera_Assembler_Terminology
|
[unitig]: http://wgs-assembler.sourceforge.net/wiki/index.php/Celera_Assembler_Terminology
|
||||||
[gfa]: https://github.com/pmelsted/GFA-spec/blob/master/GFA-spec.md
|
[gfa]: https://github.com/pmelsted/GFA-spec/blob/master/GFA-spec.md
|
||||||
[paf]: https://github.com/lh3/miniasm/blob/master/PAF.md
|
[paf]: https://github.com/lh3/miniasm/blob/master/PAF.md
|
||||||
[yak]: https://github.com/lh3/yak
|
[yak]: https://github.com/lh3/yak
|
||||||
[tutorial]: https://hifiasm.readthedocs.io/en/latest/index.html
|
[tutorial]: https://hifiasm.readthedocs.io/en/latest/index.html
|
||||||
[tutorial_output]: https://hifiasm.readthedocs.io/en/latest/interpreting-output.html#interpreting-output
|
[tutorial_output]: https://hifiasm.readthedocs.io/en/latest/interpreting-output.html#interpreting-output
|
||||||
|
|
||||||
|
|
||||||
## <a name="help"></a>Getting Help
|
## <a name="help"></a>Getting Help
|
||||||
|
|
||||||
For detailed description of options, please see [tutorial][tutorial] or `man ./hifiasm.1`. The `-h`
|
For detailed description of options, please see [tutorial][tutorial] or `man ./hifiasm.1`. The `-h`
|
||||||
option of hifiasm also provides brief description of options. If you have
|
option of hifiasm also provides brief description of options. If you have
|
||||||
further questions, please raise an issue at the [issue
|
further questions, please raise an issue at the [issue
|
||||||
page](https://github.com/chhylp123/hifiasm/issues).
|
page](https://github.com/chhylp123/hifiasm/issues).
|
||||||
|
|
||||||
## <a name="limit"></a>Limitations
|
## <a name="limit"></a>Limitations
|
||||||
|
|
||||||
1. Purging haplotig duplications may introduce misassemblies.
|
1. Purging haplotig duplications may introduce misassemblies.
|
||||||
|
|
||||||
## <a name="cite"></a>Citating Hifiasm
|
## <a name="cite"></a>Citating Hifiasm
|
||||||
|
|
||||||
If you use hifiasm in your work, please cite:
|
If you use hifiasm in your work, please cite:
|
||||||
|
|
||||||
> Cheng, H., Concepcion, G.T., Feng, X., Zhang, H., Li H. (2021)
|
> Cheng, H., Concepcion, G.T., Feng, X., Zhang, H., Li H. (2021)
|
||||||
> Haplotype-resolved de novo assembly using phased assembly graphs with
|
> Haplotype-resolved de novo assembly using phased assembly graphs with
|
||||||
> hifiasm. *Nat Methods*, **18**:170-175.
|
> hifiasm. *Nat Methods*, **18**:170-175.
|
||||||
> https://doi.org/10.1038/s41592-020-01056-5
|
> https://doi.org/10.1038/s41592-020-01056-5
|
||||||
|
|
||||||
> Cheng, H., Jarvis, E.D., Fedrigo, O., Koepfli, K.P., Urban, L., Gemmell, N.J., Li, H. (2022)
|
> Cheng, H., Jarvis, E.D., Fedrigo, O., Koepfli, K.P., Urban, L., Gemmell, N.J., Li, H. (2022)
|
||||||
> Haplotype-resolved assembly of diploid genomes without parental data.
|
> Haplotype-resolved assembly of diploid genomes without parental data.
|
||||||
> *Nature Biotechnology*, **40**:1332–1335.
|
> *Nature Biotechnology*, **40**:1332–1335.
|
||||||
> https://doi.org/10.1038/s41587-022-01261-x
|
> https://doi.org/10.1038/s41587-022-01261-x
|
||||||
|
|
||||||
> Cheng, H., Asri, M., Lucas, J., Koren, S., Li, H. (2024)
|
> Cheng, H., Asri, M., Lucas, J., Koren, S., Li, H. (2024)
|
||||||
> Scalable telomere-to-telomere assembly for diploid and polyploid genomes with double graph.
|
> Scalable telomere-to-telomere assembly for diploid and polyploid genomes with double graph.
|
||||||
> *Nat Methods*, **21**:967-970.
|
> *Nat Methods*, **21**:967-970.
|
||||||
> https://doi.org/10.1038/s41592-024-02269-8
|
> https://doi.org/10.1038/s41592-024-02269-8
|
||||||
|
|||||||
+4028
-3638
File diff suppressed because it is too large
Load Diff
+30
-30
@@ -1,30 +1,30 @@
|
|||||||
## Contributor Code of Conduct
|
## Contributor Code of Conduct
|
||||||
|
|
||||||
As contributors and maintainers of this project, we pledge to respect all
|
As contributors and maintainers of this project, we pledge to respect all
|
||||||
people who contribute through reporting issues, posting feature requests,
|
people who contribute through reporting issues, posting feature requests,
|
||||||
updating documentation, submitting pull requests or patches, and other
|
updating documentation, submitting pull requests or patches, and other
|
||||||
activities.
|
activities.
|
||||||
|
|
||||||
We are committed to making participation in this project a harassment-free
|
We are committed to making participation in this project a harassment-free
|
||||||
experience for everyone, regardless of level of experience, gender, gender
|
experience for everyone, regardless of level of experience, gender, gender
|
||||||
identity and expression, sexual orientation, disability, personal appearance,
|
identity and expression, sexual orientation, disability, personal appearance,
|
||||||
body size, race, age, or religion.
|
body size, race, age, or religion.
|
||||||
|
|
||||||
Examples of unacceptable behavior by participants include the use of sexual
|
Examples of unacceptable behavior by participants include the use of sexual
|
||||||
language or imagery, derogatory comments or personal attacks, trolling, public
|
language or imagery, derogatory comments or personal attacks, trolling, public
|
||||||
or private harassment, insults, or other unprofessional conduct.
|
or private harassment, insults, or other unprofessional conduct.
|
||||||
|
|
||||||
Project maintainers have the right and responsibility to remove, edit, or
|
Project maintainers have the right and responsibility to remove, edit, or
|
||||||
reject comments, commits, code, wiki edits, issues, and other contributions
|
reject comments, commits, code, wiki edits, issues, and other contributions
|
||||||
that are not aligned to this Code of Conduct. Project maintainers or
|
that are not aligned to this Code of Conduct. Project maintainers or
|
||||||
contributors who do not follow the Code of Conduct may be removed from the
|
contributors who do not follow the Code of Conduct may be removed from the
|
||||||
project team.
|
project team.
|
||||||
|
|
||||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||||
reported by opening an issue or contacting the maintainer via email.
|
reported by opening an issue or contacting the maintainer via email.
|
||||||
|
|
||||||
This Code of Conduct is adapted from the [Contributor Covenant][cc], [version
|
This Code of Conduct is adapted from the [Contributor Covenant][cc], [version
|
||||||
1.0.0][v1].
|
1.0.0][v1].
|
||||||
|
|
||||||
[cc]: http://contributor-covenant.org/
|
[cc]: http://contributor-covenant.org/
|
||||||
[v1]: http://contributor-covenant.org/version/1/0/0/
|
[v1]: http://contributor-covenant.org/version/1/0/0/
|
||||||
|
|||||||
+20
-20
@@ -1,20 +1,20 @@
|
|||||||
# Minimal makefile for Sphinx documentation
|
# Minimal makefile for Sphinx documentation
|
||||||
#
|
#
|
||||||
|
|
||||||
# You can set these variables from the command line, and also
|
# You can set these variables from the command line, and also
|
||||||
# from the environment for the first two.
|
# from the environment for the first two.
|
||||||
SPHINXOPTS ?=
|
SPHINXOPTS ?=
|
||||||
SPHINXBUILD ?= sphinx-build
|
SPHINXBUILD ?= sphinx-build
|
||||||
SOURCEDIR = source
|
SOURCEDIR = source
|
||||||
BUILDDIR = build
|
BUILDDIR = build
|
||||||
|
|
||||||
# Put it first so that "make" without argument is like "make help".
|
# Put it first so that "make" without argument is like "make help".
|
||||||
help:
|
help:
|
||||||
@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
||||||
|
|
||||||
.PHONY: help Makefile
|
.PHONY: help Makefile
|
||||||
|
|
||||||
# Catch-all target: route all unknown targets to Sphinx using the new
|
# Catch-all target: route all unknown targets to Sphinx using the new
|
||||||
# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS).
|
# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS).
|
||||||
%: Makefile
|
%: Makefile
|
||||||
@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
||||||
|
|||||||
+256
-256
@@ -1,257 +1,257 @@
|
|||||||
# -*- coding: utf-8 -*-
|
# -*- coding: utf-8 -*-
|
||||||
|
|
||||||
import sys
|
import sys
|
||||||
import os
|
import os
|
||||||
|
|
||||||
# -- General configuration ------------------------------------------------
|
# -- General configuration ------------------------------------------------
|
||||||
|
|
||||||
# If your documentation needs a minimal Sphinx version, state it here.
|
# If your documentation needs a minimal Sphinx version, state it here.
|
||||||
#needs_sphinx = '1.0'
|
#needs_sphinx = '1.0'
|
||||||
|
|
||||||
# Add any Sphinx extension module names here, as strings. They can be
|
# Add any Sphinx extension module names here, as strings. They can be
|
||||||
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
|
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
|
||||||
# ones.
|
# ones.
|
||||||
extensions = [
|
extensions = [
|
||||||
'sphinx.ext.todo',
|
'sphinx.ext.todo',
|
||||||
'sphinx.ext.mathjax',
|
'sphinx.ext.mathjax',
|
||||||
'sphinx.ext.ifconfig',
|
'sphinx.ext.ifconfig',
|
||||||
]
|
]
|
||||||
|
|
||||||
# Add any paths that contain templates here, relative to this directory.
|
# Add any paths that contain templates here, relative to this directory.
|
||||||
templates_path = ['_templates']
|
templates_path = ['_templates']
|
||||||
|
|
||||||
# The suffix of source filenames.
|
# The suffix of source filenames.
|
||||||
source_suffix = '.rst'
|
source_suffix = '.rst'
|
||||||
|
|
||||||
# The encoding of source files.
|
# The encoding of source files.
|
||||||
#source_encoding = 'utf-8-sig'
|
#source_encoding = 'utf-8-sig'
|
||||||
|
|
||||||
# The master toctree document.
|
# The master toctree document.
|
||||||
master_doc = 'index'
|
master_doc = 'index'
|
||||||
|
|
||||||
# General information about the project.
|
# General information about the project.
|
||||||
project = u'hifiasm'
|
project = u'hifiasm'
|
||||||
copyright = u'2021, Haoyu Cheng, Heng Li'
|
copyright = u'2021, Haoyu Cheng, Heng Li'
|
||||||
|
|
||||||
# The version info for the project you're documenting, acts as replacement for
|
# The version info for the project you're documenting, acts as replacement for
|
||||||
# |version| and |release|, also used in various other places throughout the
|
# |version| and |release|, also used in various other places throughout the
|
||||||
# built documents.
|
# built documents.
|
||||||
#
|
#
|
||||||
# The short X.Y version.
|
# The short X.Y version.
|
||||||
version = '0.16.0-r369'
|
version = '0.16.0-r369'
|
||||||
# The full version, including alpha/beta/rc tags.
|
# The full version, including alpha/beta/rc tags.
|
||||||
release = '0.16.0'
|
release = '0.16.0'
|
||||||
|
|
||||||
# The language for content autogenerated by Sphinx. Refer to documentation
|
# The language for content autogenerated by Sphinx. Refer to documentation
|
||||||
# for a list of supported languages.
|
# for a list of supported languages.
|
||||||
#language = None
|
#language = None
|
||||||
|
|
||||||
# There are two options for replacing |today|: either, you set today to some
|
# There are two options for replacing |today|: either, you set today to some
|
||||||
# non-false value, then it is used:
|
# non-false value, then it is used:
|
||||||
#today = ''
|
#today = ''
|
||||||
# Else, today_fmt is used as the format for a strftime call.
|
# Else, today_fmt is used as the format for a strftime call.
|
||||||
#today_fmt = '%B %d, %Y'
|
#today_fmt = '%B %d, %Y'
|
||||||
|
|
||||||
# List of patterns, relative to source directory, that match files and
|
# List of patterns, relative to source directory, that match files and
|
||||||
# directories to ignore when looking for source files.
|
# directories to ignore when looking for source files.
|
||||||
exclude_patterns = []
|
exclude_patterns = []
|
||||||
|
|
||||||
# The reST default role (used for this markup: `text`) to use for all
|
# The reST default role (used for this markup: `text`) to use for all
|
||||||
# documents.
|
# documents.
|
||||||
#default_role = None
|
#default_role = None
|
||||||
|
|
||||||
# If true, '()' will be appended to :func: etc. cross-reference text.
|
# If true, '()' will be appended to :func: etc. cross-reference text.
|
||||||
#add_function_parentheses = True
|
#add_function_parentheses = True
|
||||||
|
|
||||||
# If true, the current module name will be prepended to all description
|
# If true, the current module name will be prepended to all description
|
||||||
# unit titles (such as .. function::).
|
# unit titles (such as .. function::).
|
||||||
#add_module_names = True
|
#add_module_names = True
|
||||||
|
|
||||||
# If true, sectionauthor and moduleauthor directives will be shown in the
|
# If true, sectionauthor and moduleauthor directives will be shown in the
|
||||||
# output. They are ignored by default.
|
# output. They are ignored by default.
|
||||||
#show_authors = False
|
#show_authors = False
|
||||||
|
|
||||||
# The name of the Pygments (syntax highlighting) style to use.
|
# The name of the Pygments (syntax highlighting) style to use.
|
||||||
pygments_style = 'sphinx'
|
pygments_style = 'sphinx'
|
||||||
|
|
||||||
# A list of ignored prefixes for module index sorting.
|
# A list of ignored prefixes for module index sorting.
|
||||||
#modindex_common_prefix = []
|
#modindex_common_prefix = []
|
||||||
|
|
||||||
# If true, keep warnings as "system message" paragraphs in the built documents.
|
# If true, keep warnings as "system message" paragraphs in the built documents.
|
||||||
#keep_warnings = False
|
#keep_warnings = False
|
||||||
|
|
||||||
|
|
||||||
# -- Options for HTML output ----------------------------------------------
|
# -- Options for HTML output ----------------------------------------------
|
||||||
|
|
||||||
# The theme to use for HTML and HTML Help pages. See the documentation for
|
# The theme to use for HTML and HTML Help pages. See the documentation for
|
||||||
# a list of builtin themes.
|
# a list of builtin themes.
|
||||||
html_theme = 'default'
|
html_theme = 'default'
|
||||||
|
|
||||||
# Theme options are theme-specific and customize the look and feel of a theme
|
# Theme options are theme-specific and customize the look and feel of a theme
|
||||||
# further. For a list of options available for each theme, see the
|
# further. For a list of options available for each theme, see the
|
||||||
# documentation.
|
# documentation.
|
||||||
#html_theme_options = {}
|
#html_theme_options = {}
|
||||||
|
|
||||||
# Add any paths that contain custom themes here, relative to this directory.
|
# Add any paths that contain custom themes here, relative to this directory.
|
||||||
#html_theme_path = []
|
#html_theme_path = []
|
||||||
|
|
||||||
# Build using the RTD theme, if not on RTD.
|
# Build using the RTD theme, if not on RTD.
|
||||||
# https://read-the-docs.readthedocs.org/en/latest/theme.html
|
# https://read-the-docs.readthedocs.org/en/latest/theme.html
|
||||||
# https://github.com/snide/sphinx_rtd_theme
|
# https://github.com/snide/sphinx_rtd_theme
|
||||||
#
|
#
|
||||||
on_rtd = os.environ.get('READTHEDOCS', None) == 'True'
|
on_rtd = os.environ.get('READTHEDOCS', None) == 'True'
|
||||||
|
|
||||||
if not on_rtd: # only import and set the theme if we're building docs locally
|
if not on_rtd: # only import and set the theme if we're building docs locally
|
||||||
import sphinx_rtd_theme
|
import sphinx_rtd_theme
|
||||||
html_theme = 'sphinx_rtd_theme'
|
html_theme = 'sphinx_rtd_theme'
|
||||||
html_theme_path = [ "/usr/local/lib/python2.7/site-packages", ]
|
html_theme_path = [ "/usr/local/lib/python2.7/site-packages", ]
|
||||||
|
|
||||||
|
|
||||||
# The name for this set of Sphinx documents. If None, it defaults to
|
# The name for this set of Sphinx documents. If None, it defaults to
|
||||||
# "<project> v<release> documentation".
|
# "<project> v<release> documentation".
|
||||||
#html_title = None
|
#html_title = None
|
||||||
|
|
||||||
# A shorter title for the navigation bar. Default is the same as html_title.
|
# A shorter title for the navigation bar. Default is the same as html_title.
|
||||||
#html_short_title = None
|
#html_short_title = None
|
||||||
|
|
||||||
# The name of an image file (relative to this directory) to place at the top
|
# The name of an image file (relative to this directory) to place at the top
|
||||||
# of the sidebar.
|
# of the sidebar.
|
||||||
#html_logo = None
|
#html_logo = None
|
||||||
|
|
||||||
# The name of an image file (within the static path) to use as favicon of the
|
# The name of an image file (within the static path) to use as favicon of the
|
||||||
# docs. This file should be a Windows icon file (.ico) being 16x16 or 32x32
|
# docs. This file should be a Windows icon file (.ico) being 16x16 or 32x32
|
||||||
# pixels large.
|
# pixels large.
|
||||||
#html_favicon = None
|
#html_favicon = None
|
||||||
|
|
||||||
# Add any paths that contain custom static files (such as style sheets) here,
|
# Add any paths that contain custom static files (such as style sheets) here,
|
||||||
# relative to this directory. They are copied after the builtin static files,
|
# relative to this directory. They are copied after the builtin static files,
|
||||||
# so a file named "default.css" will overwrite the builtin "default.css".
|
# so a file named "default.css" will overwrite the builtin "default.css".
|
||||||
html_static_path = ['_static']
|
html_static_path = ['_static']
|
||||||
|
|
||||||
# Add any extra paths that contain custom files (such as robots.txt or
|
# Add any extra paths that contain custom files (such as robots.txt or
|
||||||
# .htaccess) here, relative to this directory. These files are copied
|
# .htaccess) here, relative to this directory. These files are copied
|
||||||
# directly to the root of the documentation.
|
# directly to the root of the documentation.
|
||||||
#html_extra_path = []
|
#html_extra_path = []
|
||||||
|
|
||||||
# If not '', a 'Last updated on:' timestamp is inserted at every page bottom,
|
# If not '', a 'Last updated on:' timestamp is inserted at every page bottom,
|
||||||
# using the given strftime format.
|
# using the given strftime format.
|
||||||
#html_last_updated_fmt = '%b %d, %Y'
|
#html_last_updated_fmt = '%b %d, %Y'
|
||||||
|
|
||||||
# If true, SmartyPants will be used to convert quotes and dashes to
|
# If true, SmartyPants will be used to convert quotes and dashes to
|
||||||
# typographically correct entities.
|
# typographically correct entities.
|
||||||
#html_use_smartypants = True
|
#html_use_smartypants = True
|
||||||
|
|
||||||
# Custom sidebar templates, maps document names to template names.
|
# Custom sidebar templates, maps document names to template names.
|
||||||
#html_sidebars = {}
|
#html_sidebars = {}
|
||||||
|
|
||||||
# Additional templates that should be rendered to pages, maps page names to
|
# Additional templates that should be rendered to pages, maps page names to
|
||||||
# template names.
|
# template names.
|
||||||
#html_additional_pages = {}
|
#html_additional_pages = {}
|
||||||
|
|
||||||
# If false, no module index is generated.
|
# If false, no module index is generated.
|
||||||
#html_domain_indices = True
|
#html_domain_indices = True
|
||||||
|
|
||||||
# If false, no index is generated.
|
# If false, no index is generated.
|
||||||
#html_use_index = True
|
#html_use_index = True
|
||||||
|
|
||||||
# If true, the index is split into individual pages for each letter.
|
# If true, the index is split into individual pages for each letter.
|
||||||
#html_split_index = False
|
#html_split_index = False
|
||||||
|
|
||||||
# If true, links to the reST sources are added to the pages.
|
# If true, links to the reST sources are added to the pages.
|
||||||
#html_show_sourcelink = True
|
#html_show_sourcelink = True
|
||||||
|
|
||||||
# If true, "Created using Sphinx" is shown in the HTML footer. Default is True.
|
# If true, "Created using Sphinx" is shown in the HTML footer. Default is True.
|
||||||
#html_show_sphinx = True
|
#html_show_sphinx = True
|
||||||
|
|
||||||
# If true, "(C) Copyright ..." is shown in the HTML footer. Default is True.
|
# If true, "(C) Copyright ..." is shown in the HTML footer. Default is True.
|
||||||
#html_show_copyright = True
|
#html_show_copyright = True
|
||||||
|
|
||||||
# If true, an OpenSearch description file will be output, and all pages will
|
# If true, an OpenSearch description file will be output, and all pages will
|
||||||
# contain a <link> tag referring to it. The value of this option must be the
|
# contain a <link> tag referring to it. The value of this option must be the
|
||||||
# base URL from which the finished HTML is served.
|
# base URL from which the finished HTML is served.
|
||||||
#html_use_opensearch = ''
|
#html_use_opensearch = ''
|
||||||
|
|
||||||
# This is the file name suffix for HTML files (e.g. ".xhtml").
|
# This is the file name suffix for HTML files (e.g. ".xhtml").
|
||||||
#html_file_suffix = None
|
#html_file_suffix = None
|
||||||
|
|
||||||
# Output file base name for HTML help builder.
|
# Output file base name for HTML help builder.
|
||||||
htmlhelp_basename = 'hifiasm-doc'
|
htmlhelp_basename = 'hifiasm-doc'
|
||||||
|
|
||||||
|
|
||||||
# -- Options for LaTeX output ---------------------------------------------
|
# -- Options for LaTeX output ---------------------------------------------
|
||||||
|
|
||||||
latex_elements = {
|
latex_elements = {
|
||||||
# The paper size ('letterpaper' or 'a4paper').
|
# The paper size ('letterpaper' or 'a4paper').
|
||||||
#'papersize': 'letterpaper',
|
#'papersize': 'letterpaper',
|
||||||
|
|
||||||
# The font size ('10pt', '11pt' or '12pt').
|
# The font size ('10pt', '11pt' or '12pt').
|
||||||
#'pointsize': '10pt',
|
#'pointsize': '10pt',
|
||||||
|
|
||||||
# Additional stuff for the LaTeX preamble.
|
# Additional stuff for the LaTeX preamble.
|
||||||
#'preamble': '',
|
#'preamble': '',
|
||||||
}
|
}
|
||||||
|
|
||||||
# Grouping the document tree into LaTeX files. List of tuples
|
# Grouping the document tree into LaTeX files. List of tuples
|
||||||
# (source start file, target name, title,
|
# (source start file, target name, title,
|
||||||
# author, documentclass [howto, manual, or own class]).
|
# author, documentclass [howto, manual, or own class]).
|
||||||
latex_documents = [
|
latex_documents = [
|
||||||
('index', 'hifiasm.tex', u'hifiasm Documentation',
|
('index', 'hifiasm.tex', u'hifiasm Documentation',
|
||||||
u'Haoyu Cheng, Heng Li', 'manual'),
|
u'Haoyu Cheng, Heng Li', 'manual'),
|
||||||
]
|
]
|
||||||
|
|
||||||
# The name of an image file (relative to this directory) to place at the top of
|
# The name of an image file (relative to this directory) to place at the top of
|
||||||
# the title page.
|
# the title page.
|
||||||
#latex_logo = None
|
#latex_logo = None
|
||||||
|
|
||||||
# For "manual" documents, if this is true, then toplevel headings are parts,
|
# For "manual" documents, if this is true, then toplevel headings are parts,
|
||||||
# not chapters.
|
# not chapters.
|
||||||
#latex_use_parts = False
|
#latex_use_parts = False
|
||||||
|
|
||||||
# If true, show page references after internal links.
|
# If true, show page references after internal links.
|
||||||
#latex_show_pagerefs = False
|
#latex_show_pagerefs = False
|
||||||
|
|
||||||
# If true, show URL addresses after external links.
|
# If true, show URL addresses after external links.
|
||||||
#latex_show_urls = False
|
#latex_show_urls = False
|
||||||
|
|
||||||
# Documents to append as an appendix to all manuals.
|
# Documents to append as an appendix to all manuals.
|
||||||
#latex_appendices = []
|
#latex_appendices = []
|
||||||
|
|
||||||
# If false, no module index is generated.
|
# If false, no module index is generated.
|
||||||
#latex_domain_indices = True
|
#latex_domain_indices = True
|
||||||
|
|
||||||
|
|
||||||
# -- Options for manual page output ---------------------------------------
|
# -- Options for manual page output ---------------------------------------
|
||||||
|
|
||||||
# One entry per manual page. List of tuples
|
# One entry per manual page. List of tuples
|
||||||
# (source start file, name, description, authors, manual section).
|
# (source start file, name, description, authors, manual section).
|
||||||
man_pages = [
|
man_pages = [
|
||||||
('index', 'hifiasm', u'hifiasm Documentation',
|
('index', 'hifiasm', u'hifiasm Documentation',
|
||||||
[u'Haoyu Cheng, Heng Li'], 1)
|
[u'Haoyu Cheng, Heng Li'], 1)
|
||||||
]
|
]
|
||||||
|
|
||||||
# If true, show URL addresses after external links.
|
# If true, show URL addresses after external links.
|
||||||
#man_show_urls = False
|
#man_show_urls = False
|
||||||
|
|
||||||
|
|
||||||
# -- Options for Texinfo output -------------------------------------------
|
# -- Options for Texinfo output -------------------------------------------
|
||||||
|
|
||||||
# Grouping the document tree into Texinfo files. List of tuples
|
# Grouping the document tree into Texinfo files. List of tuples
|
||||||
# (source start file, target name, title, author,
|
# (source start file, target name, title, author,
|
||||||
# dir menu entry, description, category)
|
# dir menu entry, description, category)
|
||||||
texinfo_documents = [
|
texinfo_documents = [
|
||||||
('index', 'hifiasm', u'hifiasm Documentation',
|
('index', 'hifiasm', u'hifiasm Documentation',
|
||||||
u'Haoyu Cheng, Heng Li', 'hifiasm', 'One line description of project.',
|
u'Haoyu Cheng, Heng Li', 'hifiasm', 'One line description of project.',
|
||||||
'Miscellaneous'),
|
'Miscellaneous'),
|
||||||
]
|
]
|
||||||
|
|
||||||
# Documents to append as an appendix to all manuals.
|
# Documents to append as an appendix to all manuals.
|
||||||
#texinfo_appendices = []
|
#texinfo_appendices = []
|
||||||
|
|
||||||
# If false, no module index is generated.
|
# If false, no module index is generated.
|
||||||
#texinfo_domain_indices = True
|
#texinfo_domain_indices = True
|
||||||
|
|
||||||
# How to display URL addresses: 'footnote', 'no', or 'inline'.
|
# How to display URL addresses: 'footnote', 'no', or 'inline'.
|
||||||
#texinfo_show_urls = 'footnote'
|
#texinfo_show_urls = 'footnote'
|
||||||
|
|
||||||
# If true, do not generate a @detailmenu in the "Top" node's menu.
|
# If true, do not generate a @detailmenu in the "Top" node's menu.
|
||||||
#texinfo_no_detailmenu = False
|
#texinfo_no_detailmenu = False
|
||||||
+122
-122
@@ -1,123 +1,123 @@
|
|||||||
|
|
||||||
.. _faq:
|
.. _faq:
|
||||||
|
|
||||||
Hifiasm FAQ
|
Hifiasm FAQ
|
||||||
===========
|
===========
|
||||||
|
|
||||||
|
|
||||||
.. contents::
|
.. contents::
|
||||||
:local:
|
:local:
|
||||||
|
|
||||||
|
|
||||||
How do I get contigs in FASTA?
|
How do I get contigs in FASTA?
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
The FASTA file can be produced from GFA as follows:
|
The FASTA file can be produced from GFA as follows:
|
||||||
::
|
::
|
||||||
|
|
||||||
awk '/^S/{print ">"$2;print $3}' test.p_ctg.gfa > test.p_ctg.fa
|
awk '/^S/{print ">"$2;print $3}' test.p_ctg.gfa > test.p_ctg.fa
|
||||||
|
|
||||||
Which types of assemblies should I use?
|
Which types of assemblies should I use?
|
||||||
----------------------------------------
|
----------------------------------------
|
||||||
If parental data is available, ``*dip.hap*.p_ctg.gfa`` produced in trio-binning mode should be always preferred. Otherwise if Hi-C data is available, ``*hic.hap*.p_ctg.gfa`` produced in Hi-C mode is the best choice. Both trio-binning mode and Hi-C mode generate fully-phased assemblies.
|
If parental data is available, ``*dip.hap*.p_ctg.gfa`` produced in trio-binning mode should be always preferred. Otherwise if Hi-C data is available, ``*hic.hap*.p_ctg.gfa`` produced in Hi-C mode is the best choice. Both trio-binning mode and Hi-C mode generate fully-phased assemblies.
|
||||||
|
|
||||||
If you only have HiFi reads, hifiasm in default outputs ``*bp.hap*.p_ctg.gfa``. The primary/alternate assemblies can be also produced by using ``--primary``. All these HiFi-only assemblies are not fully-phased. See `blog <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_ here for more details.
|
If you only have HiFi reads, hifiasm in default outputs ``*bp.hap*.p_ctg.gfa``. The primary/alternate assemblies can be also produced by using ``--primary``. All these HiFi-only assemblies are not fully-phased. See `blog <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_ here for more details.
|
||||||
|
|
||||||
Are inbred/homozygous genomes supported?
|
Are inbred/homozygous genomes supported?
|
||||||
--------------------------------------------------------------------------
|
--------------------------------------------------------------------------
|
||||||
|
|
||||||
Yes, please use the ``-l0`` option to disable purge duplication step.
|
Yes, please use the ``-l0`` option to disable purge duplication step.
|
||||||
|
|
||||||
Are diploid genomes supported?
|
Are diploid genomes supported?
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
Yes, most modules of hifiasm are designed for diploid samples, including purge duplication step, partially phased assembly and fully-phased assembly with trio-binning or Hi-C.
|
Yes, most modules of hifiasm are designed for diploid samples, including purge duplication step, partially phased assembly and fully-phased assembly with trio-binning or Hi-C.
|
||||||
|
|
||||||
Are polyploid genomes supported?
|
Are polyploid genomes supported?
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
The ``*r_utg.gfa`` and ``*p_utg.gfa`` are lossless so that they also work for polyploid genomes. However, currently the contig-generation modules of hifiasm are designed for diploid samples, which means both the partially phased assembly and the fully-phased assembly does not directly support polyploid genomes. If it is set to >2, the quality of primary assembly for polyploid genomes might be improved. Please use primary assembly for polyploid samples and run multiple rounds of purging steps using third-party tools such as purge_dups.
|
The ``*r_utg.gfa`` and ``*p_utg.gfa`` are lossless so that they also work for polyploid genomes. However, currently the contig-generation modules of hifiasm are designed for diploid samples, which means both the partially phased assembly and the fully-phased assembly does not directly support polyploid genomes. If it is set to >2, the quality of primary assembly for polyploid genomes might be improved. Please use primary assembly for polyploid samples and run multiple rounds of purging steps using third-party tools such as purge_dups.
|
||||||
|
|
||||||
Why one Hi-C integrated assembly is larger than another one?
|
Why one Hi-C integrated assembly is larger than another one?
|
||||||
------------------------------------------------------------
|
------------------------------------------------------------
|
||||||
|
|
||||||
For some samples like human male, the paternal haplotype should be larger than the maternal haplotype. However, if one assembly is much larger than another one, it should be the issues of hifiasm. To fix it, please set smaller value for ``-s`` (default: 0.55).
|
For some samples like human male, the paternal haplotype should be larger than the maternal haplotype. However, if one assembly is much larger than another one, it should be the issues of hifiasm. To fix it, please set smaller value for ``-s`` (default: 0.55).
|
||||||
|
|
||||||
Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads. For instance, hifiasm prints the following information during assembly:
|
Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads. For instance, hifiasm prints the following information during assembly:
|
||||||
::
|
::
|
||||||
|
|
||||||
[M::purge_dups] homozygous read coverage threshold: 36
|
[M::purge_dups] homozygous read coverage threshold: 36
|
||||||
|
|
||||||
In this example, hifiasm identifies the coverage threshold for homozygous reads as ``36``. If it is significantly smaller than the homozygous coverage peak, hifiasm will generate two unbalanced assemblies. In this case, please set ``--hom-cov`` to homozygous coverage peak. Please note that tuning ``--hom-cov`` may affect ``*p_utg*gfa`` so that ``*hic*.bin`` should be deleted. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
|
In this example, hifiasm identifies the coverage threshold for homozygous reads as ``36``. If it is significantly smaller than the homozygous coverage peak, hifiasm will generate two unbalanced assemblies. In this case, please set ``--hom-cov`` to homozygous coverage peak. Please note that tuning ``--hom-cov`` may affect ``*p_utg*gfa`` so that ``*hic*.bin`` should be deleted. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
For Hi-C integrated assembly, why the assembly size of both haplotypes are much larger than the estimated genome size?
|
For Hi-C integrated assembly, why the assembly size of both haplotypes are much larger than the estimated genome size?
|
||||||
------------------------------------------------------------------------------------------------------------------------------
|
------------------------------------------------------------------------------------------------------------------------------
|
||||||
It is likely that hifiasm misidentifies coverage threshold for homozygous reads. Hifiasm prints the following information for debugging:
|
It is likely that hifiasm misidentifies coverage threshold for homozygous reads. Hifiasm prints the following information for debugging:
|
||||||
::
|
::
|
||||||
|
|
||||||
[M::stat] # heterozygous bases: 645155110; # homozygous bases: 1495396634
|
[M::stat] # heterozygous bases: 645155110; # homozygous bases: 1495396634
|
||||||
|
|
||||||
If most bases of a diploid sample are homozygous, the coverage threshold is wrongly determined by hifiasm. For instance, hifiasm prints the following information during assembly:
|
If most bases of a diploid sample are homozygous, the coverage threshold is wrongly determined by hifiasm. For instance, hifiasm prints the following information during assembly:
|
||||||
::
|
::
|
||||||
|
|
||||||
[M::purge_dups] homozygous read coverage threshold: 36
|
[M::purge_dups] homozygous read coverage threshold: 36
|
||||||
|
|
||||||
In this example, hifiasm identifies the coverage threshold for homozygous reads as ``36``. If it is much smaller than homozygous coverage peak, hifiasm thinks most reads are homozygous and assign them to both assemblies, making both of them much larger than the estimated haploid genome size. In this case, please set ``--hom-cov`` to homozygous coverage peak. Please note that tuning ``--hom-cov`` may affect ``*p_utg*gfa`` so that ``*hic*.bin`` should be deleted. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
|
In this example, hifiasm identifies the coverage threshold for homozygous reads as ``36``. If it is much smaller than homozygous coverage peak, hifiasm thinks most reads are homozygous and assign them to both assemblies, making both of them much larger than the estimated haploid genome size. In this case, please set ``--hom-cov`` to homozygous coverage peak. Please note that tuning ``--hom-cov`` may affect ``*p_utg*gfa`` so that ``*hic*.bin`` should be deleted. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
|
||||||
|
|
||||||
|
|
||||||
.. _hic-iss:
|
.. _hic-iss:
|
||||||
|
|
||||||
How can I tweak parameters to improve Hi-C integrated assembly?
|
How can I tweak parameters to improve Hi-C integrated assembly?
|
||||||
---------------------------------------------------------------
|
---------------------------------------------------------------
|
||||||
Compared with the HiFi-only assembly or the trio-binning assembly, the Hi-C integrated assembly is a little bit more complex so that you need to take care of the results. See `Why one Hi-C integrated assembly is larger than another one?`_ and `For Hi-C integrated assembly, why the assembly size of both haplotypes are much larger than the estimated genome size?`_ for details on how to fix potential issues.
|
Compared with the HiFi-only assembly or the trio-binning assembly, the Hi-C integrated assembly is a little bit more complex so that you need to take care of the results. See `Why one Hi-C integrated assembly is larger than another one?`_ and `For Hi-C integrated assembly, why the assembly size of both haplotypes are much larger than the estimated genome size?`_ for details on how to fix potential issues.
|
||||||
|
|
||||||
There are several other options that may affect the Hi-C integrated assembly. Increasing the values of ``--n-weight``, ``--n-perturb`` and ``--f-perturb`` may improve phasing results but takes longer time. However, tuning ``--l-msjoin`` is tricky. All these options do not affect ``*p_utg*gfa`` so that ``*hic*.bin`` can be reused.
|
There are several other options that may affect the Hi-C integrated assembly. Increasing the values of ``--n-weight``, ``--n-perturb`` and ``--f-perturb`` may improve phasing results but takes longer time. However, tuning ``--l-msjoin`` is tricky. All these options do not affect ``*p_utg*gfa`` so that ``*hic*.bin`` can be reused.
|
||||||
|
|
||||||
.. _p-large:
|
.. _p-large:
|
||||||
|
|
||||||
Why the size of primary assembly or partially phased assembly is much larger than the estimated genome size?
|
Why the size of primary assembly or partially phased assembly is much larger than the estimated genome size?
|
||||||
---------------------------------------------------------------------------------------------------------------
|
---------------------------------------------------------------------------------------------------------------
|
||||||
It could be because the estimated genome size is incorrect. Another possibility is that hifiasm does not perform enough purging. Setting smaller value for ``-s`` (default: 0.55) or turning ``--hom-cov`` should be helpful. See :ref:`loginter` for more details.
|
It could be because the estimated genome size is incorrect. Another possibility is that hifiasm does not perform enough purging. Setting smaller value for ``-s`` (default: 0.55) or turning ``--hom-cov`` should be helpful. See :ref:`loginter` for more details.
|
||||||
|
|
||||||
|
|
||||||
.. _p-hamming:
|
.. _p-hamming:
|
||||||
|
|
||||||
Why the hamming error rate or the swith error rate of trio-binning assembly is very high?
|
Why the hamming error rate or the swith error rate of trio-binning assembly is very high?
|
||||||
---------------------------------------------------------------------------------------------------------------
|
---------------------------------------------------------------------------------------------------------------
|
||||||
In rare cases, a potential issue is that a few contigs may misjoin two haplotypes. For example, half of a contig come from mother while another half come from father. Such misjoined contigs can be fixed by manually breaking. The coordinates of problematic regions can be found by A-lines in GFA file or ``yak trioeval -e`` (see `issue 37 <https://github.com/chhylp123/hifiasm/issues/37>`_ for more details). However, if there are many misjoined contigs or the switch/hamming error rate reported by ``yak trioeval`` is very high, users should check if the parental data is correct (see `issue 130 <https://github.com/chhylp123/hifiasm/issues/130#issuecomment-862347943>`_ for more details).
|
In rare cases, a potential issue is that a few contigs may misjoin two haplotypes. For example, half of a contig come from mother while another half come from father. Such misjoined contigs can be fixed by manually breaking. The coordinates of problematic regions can be found by A-lines in GFA file or ``yak trioeval -e`` (see `issue 37 <https://github.com/chhylp123/hifiasm/issues/37>`_ for more details). However, if there are many misjoined contigs or the switch/hamming error rate reported by ``yak trioeval`` is very high, users should check if the parental data is correct (see `issue 130 <https://github.com/chhylp123/hifiasm/issues/130#issuecomment-862347943>`_ for more details).
|
||||||
|
|
||||||
Another possibility is that there are some unitigs in unitig graph misjoining two haplotypes. Such problematic unitigs might be ignored by the graph-binning strategy. Set smaller value for ``--t-occ`` forcedly remove unitig including unexpected haplotype-specific reads.
|
Another possibility is that there are some unitigs in unitig graph misjoining two haplotypes. Such problematic unitigs might be ignored by the graph-binning strategy. Set smaller value for ``--t-occ`` forcedly remove unitig including unexpected haplotype-specific reads.
|
||||||
|
|
||||||
Why does hifiasm stuck or crash?
|
Why does hifiasm stuck or crash?
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
In most cases, it is caused by the low quality HiFi reads. A good HiFi dataset should have a k-mer plot like `issue10 <https://github.com/chhylp123/hifiasm/issues/10#issuecomment-616213684>`_ or `issue49 <https://github.com/chhylp123/hifiasm/issues/49#issue-729106823>`_. In contrast, low quality HiFi data often lead to weird k-mer plot like `issue93 <https://github.com/chhylp123/hifiasm/issues/93#issue-852259042>`_. Such weird k-mer plots usually indicate insufficient coverage or presence of contaminants. See :ref:`loginter` for more details. If the HiFi data look fine, please raise an issue at the `issue page <https://github.com/chhylp123/hifiasm/issues>`_.
|
In most cases, it is caused by the low quality HiFi reads. A good HiFi dataset should have a k-mer plot like `issue10 <https://github.com/chhylp123/hifiasm/issues/10#issuecomment-616213684>`_ or `issue49 <https://github.com/chhylp123/hifiasm/issues/49#issue-729106823>`_. In contrast, low quality HiFi data often lead to weird k-mer plot like `issue93 <https://github.com/chhylp123/hifiasm/issues/93#issue-852259042>`_. Such weird k-mer plots usually indicate insufficient coverage or presence of contaminants. See :ref:`loginter` for more details. If the HiFi data look fine, please raise an issue at the `issue page <https://github.com/chhylp123/hifiasm/issues>`_.
|
||||||
|
|
||||||
What's the usage of different bin files in hifiasm?
|
What's the usage of different bin files in hifiasm?
|
||||||
----------------------------------------------------
|
----------------------------------------------------
|
||||||
``*ec.bin``, ``*ovlp.reverse.bin`` and ``*ovlp.source.bin`` save the results of error correction step. ``*hic*bin`` saves the results of Hi-C alignment. Please note that ``*hic*.bin`` should be deleted when tuning any parameters affecting ``*p_utg*gfa``. There are several parameters which does not change ``*p_utg*gfa``, including ``-s``, ``--seed``, ``--n-weight``, ``--n-perturb``, ``--f-perturb`` and ``--l-msjoin``. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
|
``*ec.bin``, ``*ovlp.reverse.bin`` and ``*ovlp.source.bin`` save the results of error correction step. ``*hic*bin`` saves the results of Hi-C alignment. Please note that ``*hic*.bin`` should be deleted when tuning any parameters affecting ``*p_utg*gfa``. There are several parameters which does not change ``*p_utg*gfa``, including ``-s``, ``--seed``, ``--n-weight``, ``--n-perturb``, ``--f-perturb`` and ``--l-msjoin``. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically.
|
||||||
|
|
||||||
Can I generate HiFi-only assembly first, and then add Hi-C or trio data later?
|
Can I generate HiFi-only assembly first, and then add Hi-C or trio data later?
|
||||||
----------------------------------------------------------------------------------------
|
----------------------------------------------------------------------------------------
|
||||||
Yes, the HiFi-only assembly, Hi-C phased assembly and trio-binning assembly share the same ``*ec.bin``, ``*ovlp.reverse.bin`` and ``*ovlp.source.bin``.
|
Yes, the HiFi-only assembly, Hi-C phased assembly and trio-binning assembly share the same ``*ec.bin``, ``*ovlp.reverse.bin`` and ``*ovlp.source.bin``.
|
||||||
|
|
||||||
What is the minimum read coverage required for hifiasm?
|
What is the minimum read coverage required for hifiasm?
|
||||||
-------------------------------------------------------
|
-------------------------------------------------------
|
||||||
Usually >=13x HiFi reads per haplotype. Higher coverage might be able to improve the contiguity of assembly.
|
Usually >=13x HiFi reads per haplotype. Higher coverage might be able to improve the contiguity of assembly.
|
||||||
|
|
||||||
Why the primary assembly is more contiguous than the fully-phased assemblies and the partially phased assemblies (i.e. ``*.hap*.p_ctg.gfa``)?
|
Why the primary assembly is more contiguous than the fully-phased assemblies and the partially phased assemblies (i.e. ``*.hap*.p_ctg.gfa``)?
|
||||||
----------------------------------------------------------------------------------------------------------------------------------------------------
|
----------------------------------------------------------------------------------------------------------------------------------------------------
|
||||||
|
|
||||||
For diploid samples, primary assembly usually has greater N50 but at the expense of highly fragmented alternate assembly. From the method view, the primary assembly has an extra joining step, which joins two haplotypes to make primary assembly more contiguous.
|
For diploid samples, primary assembly usually has greater N50 but at the expense of highly fragmented alternate assembly. From the method view, the primary assembly has an extra joining step, which joins two haplotypes to make primary assembly more contiguous.
|
||||||
|
|
||||||
When producing fully-phased assemblies and partially phased assemblies, hifiasm is designed to keep both haplotypes contiguous. It is important for many downstream applications like SV calling.
|
When producing fully-phased assemblies and partially phased assemblies, hifiasm is designed to keep both haplotypes contiguous. It is important for many downstream applications like SV calling.
|
||||||
|
|
||||||
My assembly is fragmented or not contiguous enough, how do I improve it?
|
My assembly is fragmented or not contiguous enough, how do I improve it?
|
||||||
--------------------------------------------------------------------------
|
--------------------------------------------------------------------------
|
||||||
|
|
||||||
Raising ``-D`` or ``-N`` may improve the resolution of repetitive regions but takes longer time. These two options affect all types of assemblies and usually do not have a negative impact on the assembly quality. In contrast, ``--purge-max`` only affects primary assembly. Setting larger value for ``--purge-max`` makes primary assembly more contiguous but may collapse repeats or segmental duplications.
|
Raising ``-D`` or ``-N`` may improve the resolution of repetitive regions but takes longer time. These two options affect all types of assemblies and usually do not have a negative impact on the assembly quality. In contrast, ``--purge-max`` only affects primary assembly. Setting larger value for ``--purge-max`` makes primary assembly more contiguous but may collapse repeats or segmental duplications.
|
||||||
|
|
||||||
If the assembly is too fragmented, users should check if HiFi data is good enough. See `Why does hifiasm stuck or crash?`_ for details.
|
If the assembly is too fragmented, users should check if HiFi data is good enough. See `Why does hifiasm stuck or crash?`_ for details.
|
||||||
|
|
||||||
How do I avoid misassemblies?
|
How do I avoid misassemblies?
|
||||||
--------------------------------------------------------------------------
|
--------------------------------------------------------------------------
|
||||||
Set smaller value for ``--purge-max``, ``-s`` and ``-O``, or use the ``-u`` option.
|
Set smaller value for ``--purge-max``, ``-s`` and ``-O``, or use the ``-u`` option.
|
||||||
@@ -1,16 +1,16 @@
|
|||||||
|
|
||||||
.. _hic-assembly:
|
.. _hic-assembly:
|
||||||
|
|
||||||
Hi-C Integrated Assembly
|
Hi-C Integrated Assembly
|
||||||
========================
|
========================
|
||||||
|
|
||||||
Hifiasm can generate a pair of haplotype-resolved assemblies with paired-end Hi-C reads::
|
Hifiasm can generate a pair of haplotype-resolved assemblies with paired-end Hi-C reads::
|
||||||
|
|
||||||
hifiasm -o NA12878.asm -t32 --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
|
hifiasm -o NA12878.asm -t32 --h1 read1.fq.gz --h2 read2.fq.gz HiFi-reads.fq.gz
|
||||||
|
|
||||||
In this mode, each contig is supposed to be a haplotig, which by definition comes from one parental haplotype only. Hifiasm often puts all contigs from the same parental chromosome in one assembly. It has cleanly separated chrX and chrY for a human male dataset. Nonetheless, phasing across centromeres is challenging. Hifiasm is often able to phase entire chromosomes but it may fail in rare cases. Also, contigs from different parental chromosomes are randomly mixed as it is just not possible to phase across chromosomes with Hi-C. Hifiasm does not perform scaffolding for now. You need to run a standalone scaffolder such as SALSA or 3D-DNA to scaffold phased haplotigs.
|
In this mode, each contig is supposed to be a haplotig, which by definition comes from one parental haplotype only. Hifiasm often puts all contigs from the same parental chromosome in one assembly. It has cleanly separated chrX and chrY for a human male dataset. Nonetheless, phasing across centromeres is challenging. Hifiasm is often able to phase entire chromosomes but it may fail in rare cases. Also, contigs from different parental chromosomes are randomly mixed as it is just not possible to phase across chromosomes with Hi-C. Hifiasm does not perform scaffolding for now. You need to run a standalone scaffolder such as SALSA or 3D-DNA to scaffold phased haplotigs.
|
||||||
|
|
||||||
|
|
||||||
For samples with high heterozygosity rate, a common issue is that one assembly is much larger than another one. To fix this issue, please set smaller value for ``-s`` (default: 0.55). Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads. In this case, please set ``--hom-cov`` to homozygous coverage peak. See :ref:`hic-iss` for more details.
|
For samples with high heterozygosity rate, a common issue is that one assembly is much larger than another one. To fix this issue, please set smaller value for ``-s`` (default: 0.55). Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads. In this case, please set ``--hom-cov`` to homozygous coverage peak. See :ref:`hic-iss` for more details.
|
||||||
|
|
||||||
At the first run, hifiasm saves the alignment of Hi-C reads to disk as ``*hic*.bin``. It reuses the saved results to avoid Hi-C alignment next time. Please note that ``*hic*.bin`` should be deleted when tuning any parameters affecting ``*p_utg*gfa``. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically. There are several parameters which do not change ``*p_utg*gfa``, including ``-s``, ``--seed``, ``--n-weight``, ``--n-perturb``, ``--f-perturb`` and ``--l-msjoin``.
|
At the first run, hifiasm saves the alignment of Hi-C reads to disk as ``*hic*.bin``. It reuses the saved results to avoid Hi-C alignment next time. Please note that ``*hic*.bin`` should be deleted when tuning any parameters affecting ``*p_utg*gfa``. Since v0.15.5, hifiasm can detect such changes and renew Hi-C bin files automatically. There are several parameters which do not change ``*p_utg*gfa``, including ``-s``, ``--seed``, ``--n-weight``, ``--n-perturb``, ``--f-perturb`` and ``--l-msjoin``.
|
||||||
|
|||||||
+80
-80
@@ -1,80 +1,80 @@
|
|||||||
Hifiasm
|
Hifiasm
|
||||||
=======
|
=======
|
||||||
|
|
||||||
.. toctree::
|
.. toctree::
|
||||||
:hidden:
|
:hidden:
|
||||||
|
|
||||||
pa-assembly
|
pa-assembly
|
||||||
trio-assembly
|
trio-assembly
|
||||||
hic-assembly
|
hic-assembly
|
||||||
interpreting-output
|
interpreting-output
|
||||||
faq
|
faq
|
||||||
parameter-reference
|
parameter-reference
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
`Hifiasm <https://github.com/chhylp123/hifiasm>`_ is a fast haplotype-resolved de novo assembler for PacBio HiFi reads. It can assemble a human genome in several hours and assemble a ~30Gb California redwood genome in a few days. Hifiasm emits partially phased assemblies of quality competitive with the best assemblers. Given parental short reads or Hi-C data, it produces arguably the best haplotype-resolved assemblies so far.
|
`Hifiasm <https://github.com/chhylp123/hifiasm>`_ is a fast haplotype-resolved de novo assembler for PacBio HiFi reads. It can assemble a human genome in several hours and assemble a ~30Gb California redwood genome in a few days. Hifiasm emits partially phased assemblies of quality competitive with the best assemblers. Given parental short reads or Hi-C data, it produces arguably the best haplotype-resolved assemblies so far.
|
||||||
|
|
||||||
Publications
|
Publications
|
||||||
============
|
============
|
||||||
|
|
||||||
Hifiasm
|
Hifiasm
|
||||||
Haoyu Cheng, Gregory T. Concepcion, Xiaowen Feng, Haowen Zhang & Heng Li.
|
Haoyu Cheng, Gregory T. Concepcion, Xiaowen Feng, Haowen Zhang & Heng Li.
|
||||||
`Haplotype-resolved de novo assembly using phased assembly graphs with hifiasm <https://doi.org/10.1038/s41592-020-01056-5>`_. Nature Methods. (2021).
|
`Haplotype-resolved de novo assembly using phased assembly graphs with hifiasm <https://doi.org/10.1038/s41592-020-01056-5>`_. Nature Methods. (2021).
|
||||||
|
|
||||||
Install
|
Install
|
||||||
=======
|
=======
|
||||||
The easiest way to get started is to download a `release <https://github.com/chhylp123/hifiasm/releases>`_. Please report any issues on `github issues <https://github.com/chhylp123/hifiasm/issues>`_ page.
|
The easiest way to get started is to download a `release <https://github.com/chhylp123/hifiasm/releases>`_. Please report any issues on `github issues <https://github.com/chhylp123/hifiasm/issues>`_ page.
|
||||||
|
|
||||||
In addition, the latest unreleased version can be found from github:
|
In addition, the latest unreleased version can be found from github:
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
git clone https://github.com/chhylp123/hifiasm
|
git clone https://github.com/chhylp123/hifiasm
|
||||||
cd hifiasm && make
|
cd hifiasm && make
|
||||||
|
|
||||||
Another way is to install hifiasm via `bioconda <https://anaconda.org/bioconda/hifiasm>`_:
|
Another way is to install hifiasm via `bioconda <https://anaconda.org/bioconda/hifiasm>`_:
|
||||||
|
|
||||||
::
|
::
|
||||||
|
|
||||||
conda install -c bioconda hifiasm
|
conda install -c bioconda hifiasm
|
||||||
|
|
||||||
Assembly Concepts
|
Assembly Concepts
|
||||||
=================
|
=================
|
||||||
There are different types of assemblies which are commonly used in practice (see
|
There are different types of assemblies which are commonly used in practice (see
|
||||||
`details <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_).
|
`details <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_).
|
||||||
Hifiasm produces primary/alternate assemblies or partially phased assemblies
|
Hifiasm produces primary/alternate assemblies or partially phased assemblies
|
||||||
only with HiFi reads. Given Hi-C data or trio-binning data, hifiasm produces
|
only with HiFi reads. Given Hi-C data or trio-binning data, hifiasm produces
|
||||||
contiguous fully-phased assemblies, i.e. haplotype-resolved assemblies.
|
contiguous fully-phased assemblies, i.e. haplotype-resolved assemblies.
|
||||||
|
|
||||||
Why Hifiasm?
|
Why Hifiasm?
|
||||||
============
|
============
|
||||||
* Hifiasm delivers high-quality assemblies. It tends to generate longer contigs
|
* Hifiasm delivers high-quality assemblies. It tends to generate longer contigs
|
||||||
and resolve more segmental duplications than other assemblers.
|
and resolve more segmental duplications than other assemblers.
|
||||||
|
|
||||||
* Given Hi-C reads or short reads from the parents, hifiasm can produce overall the best
|
* Given Hi-C reads or short reads from the parents, hifiasm can produce overall the best
|
||||||
haplotype-resolved assembly so far. It is the assembler of choice by the
|
haplotype-resolved assembly so far. It is the assembler of choice by the
|
||||||
`Human Pangenome Project <https://humanpangenome.org/>`_ for the first batch of samples.
|
`Human Pangenome Project <https://humanpangenome.org/>`_ for the first batch of samples.
|
||||||
|
|
||||||
* Hifiasm can purge duplications between haplotigs without relying on
|
* Hifiasm can purge duplications between haplotigs without relying on
|
||||||
third-party tools such as purge\_dups. Hifiasm does not need polishing tools
|
third-party tools such as purge\_dups. Hifiasm does not need polishing tools
|
||||||
like pilon or racon, either. This simplifies the assembly pipeline and saves
|
like pilon or racon, either. This simplifies the assembly pipeline and saves
|
||||||
running time.
|
running time.
|
||||||
|
|
||||||
* Hifiasm is fast. It can assemble a human genome in half a day and assemble a
|
* Hifiasm is fast. It can assemble a human genome in half a day and assemble a
|
||||||
~30Gb redwood genome in three days. No genome is too large for hifiasm.
|
~30Gb redwood genome in three days. No genome is too large for hifiasm.
|
||||||
|
|
||||||
* Hifiasm is trivial to install and easy to use. It does not required Python,
|
* Hifiasm is trivial to install and easy to use. It does not required Python,
|
||||||
R or C++11 compilers, and can be compiled into a single executable. The
|
R or C++11 compilers, and can be compiled into a single executable. The
|
||||||
default setting works well with a variety of genomes.
|
default setting works well with a variety of genomes.
|
||||||
|
|
||||||
Learn
|
Learn
|
||||||
=====
|
=====
|
||||||
|
|
||||||
* :ref:`HiFi-only Assembly <pa-assembly>` - Assembling HiFi reads without additional data types
|
* :ref:`HiFi-only Assembly <pa-assembly>` - Assembling HiFi reads without additional data types
|
||||||
* :ref:`Trio-binning Assembly <trio-assembly>` - Producing fully phased assemblies with HiFi and trio-binning data
|
* :ref:`Trio-binning Assembly <trio-assembly>` - Producing fully phased assemblies with HiFi and trio-binning data
|
||||||
* :ref:`Hi-C Integrated Assembly <hic-assembly>` - Producing fully phased assemblies with HiFi and Hi-C data
|
* :ref:`Hi-C Integrated Assembly <hic-assembly>` - Producing fully phased assemblies with HiFi and Hi-C data
|
||||||
* :ref:`Hifiasm Output <interpreting-output>` - Interpreting results
|
* :ref:`Hifiasm Output <interpreting-output>` - Interpreting results
|
||||||
* :ref:`Hifiasm FAQ <faq>` - Frequently asked questions
|
* :ref:`Hifiasm FAQ <faq>` - Frequently asked questions
|
||||||
* :ref:`Hifiasm Parameters <parameter-reference>` - Parameter reference of hifiasm
|
* :ref:`Hifiasm Parameters <parameter-reference>` - Parameter reference of hifiasm
|
||||||
|
|||||||
+102
-102
@@ -1,102 +1,102 @@
|
|||||||
|
|
||||||
.. _interpreting-output:
|
.. _interpreting-output:
|
||||||
|
|
||||||
Hifiasm Output
|
Hifiasm Output
|
||||||
===============
|
===============
|
||||||
|
|
||||||
.. _outfile:
|
.. _outfile:
|
||||||
|
|
||||||
Output files
|
Output files
|
||||||
---------------------------------------
|
---------------------------------------
|
||||||
|
|
||||||
In general, hifiasm generates the following assembly graphs in the GFA format:
|
In general, hifiasm generates the following assembly graphs in the GFA format:
|
||||||
|
|
||||||
* ```prefix`.r_utg.gfa``: haplotype-resolved raw unitig graph. This graph keeps all haplotype information.
|
* ```prefix`.r_utg.gfa``: haplotype-resolved raw unitig graph. This graph keeps all haplotype information.
|
||||||
* ```prefix`.p_utg.gfa``: haplotype-resolved processed unitig graph without small bubbles. Small bubbles might be caused by somatic mutations or noise in data, which are not the real haplotype information. Hifiasm automatically pops such small bubbles based on coverage. The option ``--hom-cov`` affects the result. See :ref:`homozygous coverage setting <homcov>` for more details. In addition, the option ``-p`` forcedly pops bubbles.
|
* ```prefix`.p_utg.gfa``: haplotype-resolved processed unitig graph without small bubbles. Small bubbles might be caused by somatic mutations or noise in data, which are not the real haplotype information. Hifiasm automatically pops such small bubbles based on coverage. The option ``--hom-cov`` affects the result. See :ref:`homozygous coverage setting <homcov>` for more details. In addition, the option ``-p`` forcedly pops bubbles.
|
||||||
* ```prefix`.p_ctg.gfa``: assembly graph of primary contigs. This graph includes a complete assembly with long stretches of phased blocks.
|
* ```prefix`.p_ctg.gfa``: assembly graph of primary contigs. This graph includes a complete assembly with long stretches of phased blocks.
|
||||||
* ```prefix`.a_ctg.gfa``: assembly graph of alternate contigs. This graph consists of all contigs that are discarded in primary contig graph.
|
* ```prefix`.a_ctg.gfa``: assembly graph of alternate contigs. This graph consists of all contigs that are discarded in primary contig graph.
|
||||||
* ```prefix`.*hap*.p_ctg.gfa``: phased contig graph. This graph keeps the phased contigs.
|
* ```prefix`.*hap*.p_ctg.gfa``: phased contig graph. This graph keeps the phased contigs.
|
||||||
|
|
||||||
|
|
||||||
Hifiasm outputs ``*.r_utg.gfa`` and ``*.p_utg.gfa`` in any cases. Specifically, hifiasm outputs the following assembly graphs in trio-binning mode:
|
Hifiasm outputs ``*.r_utg.gfa`` and ``*.p_utg.gfa`` in any cases. Specifically, hifiasm outputs the following assembly graphs in trio-binning mode:
|
||||||
|
|
||||||
* ```prefix`.dip.hap1.p_ctg.gfa``: fully phased paternal/haplotype1 contig graph keeping the phased paternal/haplotype1 assembly.
|
* ```prefix`.dip.hap1.p_ctg.gfa``: fully phased paternal/haplotype1 contig graph keeping the phased paternal/haplotype1 assembly.
|
||||||
* ```prefix`.dip.hap2.p_ctg.gfa``: fully phased maternal/haplotype2 contig graph keeping the phased maternal/haplotype2 assembly.
|
* ```prefix`.dip.hap2.p_ctg.gfa``: fully phased maternal/haplotype2 contig graph keeping the phased maternal/haplotype2 assembly.
|
||||||
|
|
||||||
With Hi-C partition options, hifiasm outputs:
|
With Hi-C partition options, hifiasm outputs:
|
||||||
|
|
||||||
* ```prefix`.hic.p_ctg.gfa``: assembly graph of primary contigs.
|
* ```prefix`.hic.p_ctg.gfa``: assembly graph of primary contigs.
|
||||||
* ```prefix`.hic.hap1.p_ctg.gfa``: fully phased contig graph of haplotype1 where each contig is fully phased.
|
* ```prefix`.hic.hap1.p_ctg.gfa``: fully phased contig graph of haplotype1 where each contig is fully phased.
|
||||||
* ```prefix`.hic.hap2.p_ctg.gfa``: fully phased contig graph of haplotype2 where each contig is fully phased.
|
* ```prefix`.hic.hap2.p_ctg.gfa``: fully phased contig graph of haplotype2 where each contig is fully phased.
|
||||||
* ```prefix`.hic.a_ctg.gfa`` (optional with ``--primary``): assembly graph of alternate contigs.
|
* ```prefix`.hic.a_ctg.gfa`` (optional with ``--primary``): assembly graph of alternate contigs.
|
||||||
|
|
||||||
Hifiasm generates the following assembly graphs only with HiFi reads in default:
|
Hifiasm generates the following assembly graphs only with HiFi reads in default:
|
||||||
|
|
||||||
* ```prefix`.bp.p_ctg.gfa``: assembly graph of primary contigs.
|
* ```prefix`.bp.p_ctg.gfa``: assembly graph of primary contigs.
|
||||||
* ```prefix`.bp.hap1.p_ctg.gfa``: partially phased contig graph of haplotype1.
|
* ```prefix`.bp.hap1.p_ctg.gfa``: partially phased contig graph of haplotype1.
|
||||||
* ```prefix`.bp.hap2.p_ctg.gfa``: partially phased contig graph of haplotype2.
|
* ```prefix`.bp.hap2.p_ctg.gfa``: partially phased contig graph of haplotype2.
|
||||||
|
|
||||||
If the option ``--primary`` or ``-l0`` is specified, hifiasm outputs:
|
If the option ``--primary`` or ``-l0`` is specified, hifiasm outputs:
|
||||||
|
|
||||||
* ```prefix`.p_ctg.gfa``: assembly graph of primary contigs.
|
* ```prefix`.p_ctg.gfa``: assembly graph of primary contigs.
|
||||||
* ```prefix`.a_ctg.gfa``: assembly graph of alternate contigs.
|
* ```prefix`.a_ctg.gfa``: assembly graph of alternate contigs.
|
||||||
|
|
||||||
For each graph, hifiasm also outputs a simplified version (``*noseq*gfa``) without sequences for the ease of visualization. The coordinates of low quality regions are written to ``*lowQ.bed`` in BED format.
|
For each graph, hifiasm also outputs a simplified version (``*noseq*gfa``) without sequences for the ease of visualization. The coordinates of low quality regions are written to ``*lowQ.bed`` in BED format.
|
||||||
The concepts of different types of assemblies can be found `here <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_.
|
The concepts of different types of assemblies can be found `here <https://lh3.github.io/2021/04/17/concepts-in-phased-assemblies>`_.
|
||||||
|
|
||||||
.. _outformat:
|
.. _outformat:
|
||||||
|
|
||||||
Output file formats
|
Output file formats
|
||||||
---------------------------------------
|
---------------------------------------
|
||||||
Hifiasm broadly follows the specification for `GFA 1.0 <https://github.com/GFA-spec/GFA-spec/blob/master/GFA1.md>`_. There are several fields that are specifically used by hifiasm. For ``S`` segment line:
|
Hifiasm broadly follows the specification for `GFA 1.0 <https://github.com/GFA-spec/GFA-spec/blob/master/GFA1.md>`_. There are several fields that are specifically used by hifiasm. For ``S`` segment line:
|
||||||
|
|
||||||
* ``rd:i:``: read coverage. It is calculated by the reads coming from the same contig/unitig.
|
* ``rd:i:``: read coverage. It is calculated by the reads coming from the same contig/unitig.
|
||||||
|
|
||||||
Hifiasm outputs ``A`` lines including the information of reads which are used to construct contig/unitig. Each ``A`` line is plain-text, tab-separated, and the columns appear in the following order:
|
Hifiasm outputs ``A`` lines including the information of reads which are used to construct contig/unitig. Each ``A`` line is plain-text, tab-separated, and the columns appear in the following order:
|
||||||
|
|
||||||
.. list-table::
|
.. list-table::
|
||||||
:widths: 10 25 50
|
:widths: 10 25 50
|
||||||
:header-rows: 1
|
:header-rows: 1
|
||||||
|
|
||||||
* - Col
|
* - Col
|
||||||
- Type
|
- Type
|
||||||
- Description
|
- Description
|
||||||
* - 1
|
* - 1
|
||||||
- string
|
- string
|
||||||
- Should be always ``A``
|
- Should be always ``A``
|
||||||
* - 2
|
* - 2
|
||||||
- string
|
- string
|
||||||
- Contig/unitig name
|
- Contig/unitig name
|
||||||
* - 3
|
* - 3
|
||||||
- int
|
- int
|
||||||
- Contig/unitig start coordinate of subregion constructed by read
|
- Contig/unitig start coordinate of subregion constructed by read
|
||||||
* - 4
|
* - 4
|
||||||
- char
|
- char
|
||||||
- Read strand: "+" or "-"
|
- Read strand: "+" or "-"
|
||||||
* - 5
|
* - 5
|
||||||
- string
|
- string
|
||||||
- Read name
|
- Read name
|
||||||
* - 6
|
* - 6
|
||||||
- int
|
- int
|
||||||
- Read start coordinate of subregion which is used to construct contig/unitig
|
- Read start coordinate of subregion which is used to construct contig/unitig
|
||||||
* - 7
|
* - 7
|
||||||
- int
|
- int
|
||||||
- Read end coordinate of subregion which is used to construct contig/unitig
|
- Read end coordinate of subregion which is used to construct contig/unitig
|
||||||
* - 8
|
* - 8
|
||||||
- id:i:int
|
- id:i:int
|
||||||
- Read ID
|
- Read ID
|
||||||
* - 9
|
* - 9
|
||||||
- HG:A:char
|
- HG:A:char
|
||||||
- Haplotype status of read. ``HG:A:a``, ``HG:A:p``, ``HG:A:m`` indicate read is non-binnable, father/hap1-specific and mother/hap2-specific, respectively.
|
- Haplotype status of read. ``HG:A:a``, ``HG:A:p``, ``HG:A:m`` indicate read is non-binnable, father/hap1-specific and mother/hap2-specific, respectively.
|
||||||
|
|
||||||
.. _loginter:
|
.. _loginter:
|
||||||
|
|
||||||
Hifiasm log interpretation
|
Hifiasm log interpretation
|
||||||
---------------------------------------
|
---------------------------------------
|
||||||
Hifiasm prints several information for quick debugging, including:
|
Hifiasm prints several information for quick debugging, including:
|
||||||
|
|
||||||
.. _homcov:
|
.. _homcov:
|
||||||
|
|
||||||
* k-mer plot: showing how many k-mers appear a certain number of times. For homozygous samples, there should be one peak around read coverage. For heterozygous samples, there should two peaks, where the smaller peak is around the heterozygous read coverage and the larger peak is around the homozygous read coverage. For example, `issue10 <https://github.com/chhylp123/hifiasm/issues/10#issuecomment-616213684>`_ indicates the heterozygous read coverage and the homozygous read coverage are 28 and 57, respectively. `Issue49 <https://github.com/chhylp123/hifiasm/issues/49#issue-729106823>`_ is another good example. Weird k-mer plot like `issue93 <https://github.com/chhylp123/hifiasm/issues/93#issue-852259042>`_ is often caused by insufficient coverage or presence of contaminants.
|
* k-mer plot: showing how many k-mers appear a certain number of times. For homozygous samples, there should be one peak around read coverage. For heterozygous samples, there should two peaks, where the smaller peak is around the heterozygous read coverage and the larger peak is around the homozygous read coverage. For example, `issue10 <https://github.com/chhylp123/hifiasm/issues/10#issuecomment-616213684>`_ indicates the heterozygous read coverage and the homozygous read coverage are 28 and 57, respectively. `Issue49 <https://github.com/chhylp123/hifiasm/issues/49#issue-729106823>`_ is another good example. Weird k-mer plot like `issue93 <https://github.com/chhylp123/hifiasm/issues/93#issue-852259042>`_ is often caused by insufficient coverage or presence of contaminants.
|
||||||
* homozygous coverage: coverage threshold for homozygous reads. Hifiasm prints it as: ``[M::purge_dups] homozygous read coverage threshold: X``. If it is not around homozygous coverage, the final assembly might be either too large or too small. To fix this issue, please set ``--hom-cov`` to homozygous coverage.
|
* homozygous coverage: coverage threshold for homozygous reads. Hifiasm prints it as: ``[M::purge_dups] homozygous read coverage threshold: X``. If it is not around homozygous coverage, the final assembly might be either too large or too small. To fix this issue, please set ``--hom-cov`` to homozygous coverage.
|
||||||
* number of het/hom bases: how many bases in unitig graph are heterozygous and homozygous during Hi-C phased assembly. Hifiasm prints it as: ``[M::stat] # heterozygous bases: X; # homozygous bases: Y``. Given a heterozygous sample, if there are much more homozygous bases than heterozygous bases, hifiasm fails to identify correct coverage threshold for homozygous reads. In this case, please set ``--hom-cov`` to homozygous coverage.
|
* number of het/hom bases: how many bases in unitig graph are heterozygous and homozygous during Hi-C phased assembly. Hifiasm prints it as: ``[M::stat] # heterozygous bases: X; # homozygous bases: Y``. Given a heterozygous sample, if there are much more homozygous bases than heterozygous bases, hifiasm fails to identify correct coverage threshold for homozygous reads. In this case, please set ``--hom-cov`` to homozygous coverage.
|
||||||
|
|||||||
+47
-47
@@ -1,47 +1,47 @@
|
|||||||
|
|
||||||
.. _pa-assembly:
|
.. _pa-assembly:
|
||||||
|
|
||||||
HiFi-only Assembly
|
HiFi-only Assembly
|
||||||
==================
|
==================
|
||||||
|
|
||||||
A typical hifiasm command line looks like::
|
A typical hifiasm command line looks like::
|
||||||
|
|
||||||
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
||||||
|
|
||||||
where ``NA12878.fq.gz`` provides the input reads, ``-t`` sets the number of CPUs in
|
where ``NA12878.fq.gz`` provides the input reads, ``-t`` sets the number of CPUs in
|
||||||
use and ``-o`` specifies the prefix of output files. Input sequences should be FASTA
|
use and ``-o`` specifies the prefix of output files. Input sequences should be FASTA
|
||||||
or FASTQ format, uncompressed or compressed with gzip (.gz). The quality scores of reads
|
or FASTQ format, uncompressed or compressed with gzip (.gz). The quality scores of reads
|
||||||
in FASTQ are ignored by hifiasm. Hifiasm outputs assemblies in `GFA <https://github.com/pmelsted/GFA-spec/blob/master/GFA-spec.md>`_ format.
|
in FASTQ are ignored by hifiasm. Hifiasm outputs assemblies in `GFA <https://github.com/pmelsted/GFA-spec/blob/master/GFA-spec.md>`_ format.
|
||||||
|
|
||||||
At the first run, hifiasm saves corrected reads and overlaps to disk as ``NA12878.asm.*.bin``. It reuses the saved results to avoid the time-consuming all-vs-all overlap calculation next time. You may specify ``-i`` to ignore precomputed overlaps and redo overlapping from raw reads. You can also dump error corrected reads in FASTA and read overlaps in PAF with::
|
At the first run, hifiasm saves corrected reads and overlaps to disk as ``NA12878.asm.*.bin``. It reuses the saved results to avoid the time-consuming all-vs-all overlap calculation next time. You may specify ``-i`` to ignore precomputed overlaps and redo overlapping from raw reads. You can also dump error corrected reads in FASTA and read overlaps in PAF with::
|
||||||
|
|
||||||
hifiasm -o NA12878.asm -t 32 --write-paf --write-ec /dev/null
|
hifiasm -o NA12878.asm -t 32 --write-paf --write-ec /dev/null
|
||||||
|
|
||||||
Hifiasm purges haplotig duplications by default. For inbred or homozygous genomes, you may disable purging with option ``-l0``. Old HiFi reads may contain short adapter sequences at the ends of reads. You can specify ``-z20`` to trim both ends of reads by 20bp. For small genomes, use ``-f0`` to disable the initial bloom filter which takes 16GB memory at the beginning. For genomes much larger than human, applying ``-f38`` or even ``-f39`` is preferred to save memory on k-mer counting.
|
Hifiasm purges haplotig duplications by default. For inbred or homozygous genomes, you may disable purging with option ``-l0``. Old HiFi reads may contain short adapter sequences at the ends of reads. You can specify ``-z20`` to trim both ends of reads by 20bp. For small genomes, use ``-f0`` to disable the initial bloom filter which takes 16GB memory at the beginning. For genomes much larger than human, applying ``-f38`` or even ``-f39`` is preferred to save memory on k-mer counting.
|
||||||
|
|
||||||
|
|
||||||
Produce two partially phased assemblies
|
Produce two partially phased assemblies
|
||||||
---------------------------------------
|
---------------------------------------
|
||||||
|
|
||||||
|
|
||||||
Since v0.15, hifiasm produces two sets of partially phased contigs in default like::
|
Since v0.15, hifiasm produces two sets of partially phased contigs in default like::
|
||||||
|
|
||||||
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
||||||
|
|
||||||
In this example, the partially phased contigs are written to ``NA12878.asm.bp.hap*.p_ctg.gfa``.
|
In this example, the partially phased contigs are written to ``NA12878.asm.bp.hap*.p_ctg.gfa``.
|
||||||
This pair of files can be thought to represent the two haplotypes in a diploid genome, though with occasional switch errors. The frequency of switches is determined by the heterozygosity of the input sample. Hifiasm also writes the primary contigs to ``NA12878.asm.bp.p_ctg.gfa``.
|
This pair of files can be thought to represent the two haplotypes in a diploid genome, though with occasional switch errors. The frequency of switches is determined by the heterozygosity of the input sample. Hifiasm also writes the primary contigs to ``NA12878.asm.bp.p_ctg.gfa``.
|
||||||
|
|
||||||
For samples with high heterozygosity rate, a common issue is that one set of partially phased contigs is much larger than another set. To fix this issue, please set smaller value for ``-s`` (default: 0.55). Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads.
|
For samples with high heterozygosity rate, a common issue is that one set of partially phased contigs is much larger than another set. To fix this issue, please set smaller value for ``-s`` (default: 0.55). Another possibility is that hifiasm misidentifies coverage threshold for homozygous reads.
|
||||||
In this case, please set ``--hom-cov`` to homozygous coverage. See :ref:`p-large` for more details.
|
In this case, please set ``--hom-cov`` to homozygous coverage. See :ref:`p-large` for more details.
|
||||||
|
|
||||||
|
|
||||||
Produce primary/alternate assemblies
|
Produce primary/alternate assemblies
|
||||||
------------------------------------
|
------------------------------------
|
||||||
|
|
||||||
To get primary/alternate assemblies, the option ``--primary`` should be set::
|
To get primary/alternate assemblies, the option ``--primary`` should be set::
|
||||||
|
|
||||||
hifiasm -o NA12878.asm --primary -t 32 NA12878.fq.gz
|
hifiasm -o NA12878.asm --primary -t 32 NA12878.fq.gz
|
||||||
|
|
||||||
The primary contigs and the alternate contigs are written to ``NA12878.asm.p_ctg.gfa`` and ``NA12878.asm.a_ctg.gfa``, respectively. For inbred or homozygous genomes, the primary/alternate assemblies can be also produced by ``-l0``. Similarly, turning ``-s`` or ``--hom-cov`` should
|
The primary contigs and the alternate contigs are written to ``NA12878.asm.p_ctg.gfa`` and ``NA12878.asm.a_ctg.gfa``, respectively. For inbred or homozygous genomes, the primary/alternate assemblies can be also produced by ``-l0``. Similarly, turning ``-s`` or ``--hom-cov`` should
|
||||||
be helpful if the primary assembly is too large. See :ref:`p-large` for more details.
|
be helpful if the primary assembly is too large. See :ref:`p-large` for more details.
|
||||||
|
|
||||||
|
|||||||
+298
-298
@@ -1,298 +1,298 @@
|
|||||||
|
|
||||||
.. _parameter-reference:
|
.. _parameter-reference:
|
||||||
|
|
||||||
Hifiasm Parameter Reference
|
Hifiasm Parameter Reference
|
||||||
============================
|
============================
|
||||||
|
|
||||||
Synopsis
|
Synopsis
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
Assembly only with HiFi reads:
|
Assembly only with HiFi reads:
|
||||||
::
|
::
|
||||||
|
|
||||||
hifiasm -o [prefix] -t [nThreads] [options] input1.fq [input2.fq [...]]
|
hifiasm -o [prefix] -t [nThreads] [options] input1.fq [input2.fq [...]]
|
||||||
|
|
||||||
Trio binning assembly with yak dumps:
|
Trio binning assembly with yak dumps:
|
||||||
::
|
::
|
||||||
|
|
||||||
yak count -o paternal.yak -b37 [-t nThreads] [-k kmerLen] paternal.fq.gz
|
yak count -o paternal.yak -b37 [-t nThreads] [-k kmerLen] paternal.fq.gz
|
||||||
yak count -o maternal.yak -b37 [-t nThreads] [-k kmerLen] maternal.fq.gz
|
yak count -o maternal.yak -b37 [-t nThreads] [-k kmerLen] maternal.fq.gz
|
||||||
hifiasm [-o prefix] [-t nThreads] [options] -1 paternal.yak -2 maternal.yak child.hifi.fq.gz
|
hifiasm [-o prefix] [-t nThreads] [options] -1 paternal.yak -2 maternal.yak child.hifi.fq.gz
|
||||||
|
|
||||||
Hi-C integrated assembly:
|
Hi-C integrated assembly:
|
||||||
::
|
::
|
||||||
|
|
||||||
hifiasm -o [prefix] -t [nThreads] --h1 [hic_r1.fq.gz,...] --h2 [hic_r2.fq.gz,...] [options] HiFi.read.fq.gz
|
hifiasm -o [prefix] -t [nThreads] --h1 [hic_r1.fq.gz,...] --h2 [hic_r2.fq.gz,...] [options] HiFi.read.fq.gz
|
||||||
|
|
||||||
To get detailed description of options, run:
|
To get detailed description of options, run:
|
||||||
::
|
::
|
||||||
|
|
||||||
hifiasm -h
|
hifiasm -h
|
||||||
|
|
||||||
or:
|
or:
|
||||||
::
|
::
|
||||||
|
|
||||||
man ./hifiasm.1
|
man ./hifiasm.1
|
||||||
|
|
||||||
|
|
||||||
General options
|
General options
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
.. _oopt:
|
.. _oopt:
|
||||||
|
|
||||||
**\-o <FILE=hifiasm.asm>**
|
**\-o <FILE=hifiasm.asm>**
|
||||||
Prefix of output files. See :ref:`outfile` and :ref:`outformat` for more details.
|
Prefix of output files. See :ref:`outfile` and :ref:`outformat` for more details.
|
||||||
|
|
||||||
.. _topt:
|
.. _topt:
|
||||||
|
|
||||||
**\-t <INT=1>**
|
**\-t <INT=1>**
|
||||||
Number of CPU threads used by hifiasm.
|
Number of CPU threads used by hifiasm.
|
||||||
|
|
||||||
.. _hopt:
|
.. _hopt:
|
||||||
|
|
||||||
**\-h**
|
**\-h**
|
||||||
Show help information.
|
Show help information.
|
||||||
|
|
||||||
.. _versionopt:
|
.. _versionopt:
|
||||||
|
|
||||||
**\-\-version**
|
**\-\-version**
|
||||||
Show version number.
|
Show version number.
|
||||||
|
|
||||||
|
|
||||||
Error correction options
|
Error correction options
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
.. _kopt:
|
.. _kopt:
|
||||||
|
|
||||||
**\-k <INT=51>**
|
**\-k <INT=51>**
|
||||||
K-mer length. This option must be less than 64.
|
K-mer length. This option must be less than 64.
|
||||||
|
|
||||||
.. _wopt:
|
.. _wopt:
|
||||||
|
|
||||||
**\-w <INT=51>**
|
**\-w <INT=51>**
|
||||||
Minimizer window size.
|
Minimizer window size.
|
||||||
|
|
||||||
.. _fopt:
|
.. _fopt:
|
||||||
|
|
||||||
**\-f <INT=37>**
|
**\-f <INT=37>**
|
||||||
Number of bits for bloom filter; 0 to disable. This bloom filter is used to filter out singleton k-mers when counting all k-mers. It takes 2\ :sup:`(INT-3)` bytes of memory. A proper setting saves memory. ``-f37`` is recommended for human assembly. For small genomes, use ``-f0`` to disable the initial bloom filter which takes 16GB memory at the beginning. For genomes much larger than human, applying ``-f38`` or even ``-f39`` is preferred to save memory on k-mer counting.
|
Number of bits for bloom filter; 0 to disable. This bloom filter is used to filter out singleton k-mers when counting all k-mers. It takes 2\ :sup:`(INT-3)` bytes of memory. A proper setting saves memory. ``-f37`` is recommended for human assembly. For small genomes, use ``-f0`` to disable the initial bloom filter which takes 16GB memory at the beginning. For genomes much larger than human, applying ``-f38`` or even ``-f39`` is preferred to save memory on k-mer counting.
|
||||||
|
|
||||||
.. _Dopt:
|
.. _Dopt:
|
||||||
|
|
||||||
**\-D <FLOAT=5.0>**
|
**\-D <FLOAT=5.0>**
|
||||||
Drop k-mers occurring ``>FLOAT*coverage`` times. Hifiasm discards these high-frequency k-mers during error correction to reduce running time. The ``coverage`` is determined automatically by hifiasm based on k-mer plot, representing homozygous read coverage. Raising this option may improve the resolution of repetitive regions but takes longer time.
|
Drop k-mers occurring ``>FLOAT*coverage`` times. Hifiasm discards these high-frequency k-mers during error correction to reduce running time. The ``coverage`` is determined automatically by hifiasm based on k-mer plot, representing homozygous read coverage. Raising this option may improve the resolution of repetitive regions but takes longer time.
|
||||||
|
|
||||||
.. _NEopt:
|
.. _NEopt:
|
||||||
|
|
||||||
**\-N <INT=100>**
|
**\-N <INT=100>**
|
||||||
Consider up to ``max(-D*coverage,-N)`` overlaps for each oriented read. The ``coverage`` is determined automatically by hifiasm based on k-mer plot, representing homozygous read coverage. Raising this option may improve the resolution of repetitive regions but takes longer time.
|
Consider up to ``max(-D*coverage,-N)`` overlaps for each oriented read. The ``coverage`` is determined automatically by hifiasm based on k-mer plot, representing homozygous read coverage. Raising this option may improve the resolution of repetitive regions but takes longer time.
|
||||||
|
|
||||||
.. _ropt:
|
.. _ropt:
|
||||||
|
|
||||||
**\-r <INT=3>**
|
**\-r <INT=3>**
|
||||||
Rounds of haplotype-aware error correction. This option affects all outputs of hifiasm. Odd rounds of correction are preferred in practice.
|
Rounds of haplotype-aware error correction. This option affects all outputs of hifiasm. Odd rounds of correction are preferred in practice.
|
||||||
|
|
||||||
|
|
||||||
.. _zopt:
|
.. _zopt:
|
||||||
|
|
||||||
**\-z <INT=0>**
|
**\-z <INT=0>**
|
||||||
Length of adapters that should be removed. This option remove ``INT`` bases from both ends of each read. Some old HiFi reads may consist of short adapters (e.g. 20bp adapter at one end). For such data, trimming short adapters would significantly improve the assembly quality.
|
Length of adapters that should be removed. This option remove ``INT`` bases from both ends of each read. Some old HiFi reads may consist of short adapters (e.g. 20bp adapter at one end). For such data, trimming short adapters would significantly improve the assembly quality.
|
||||||
|
|
||||||
.. _max-kocc-opt:
|
.. _max-kocc-opt:
|
||||||
|
|
||||||
**\-\-max-kocc <INT=2000>**
|
**\-\-max-kocc <INT=2000>**
|
||||||
Employ k-mers occurring < ``INT`` times to rescue repetitive overlaps. This option may improve the resolution of repeats.
|
Employ k-mers occurring < ``INT`` times to rescue repetitive overlaps. This option may improve the resolution of repeats.
|
||||||
|
|
||||||
|
|
||||||
.. _hg-size-opt:
|
.. _hg-size-opt:
|
||||||
|
|
||||||
**\-\-hg-size <INT(k/m/g)>**
|
**\-\-hg-size <INT(k/m/g)>**
|
||||||
Estimated haploid genome size used for inferring read coverage. This option is used to get accurate homozygous read coverage during error correction. Common suffices are required, for example, 100m or 3g.
|
Estimated haploid genome size used for inferring read coverage. This option is used to get accurate homozygous read coverage during error correction. Common suffices are required, for example, 100m or 3g.
|
||||||
|
|
||||||
|
|
||||||
.. _min-hist-cnt-opt:
|
.. _min-hist-cnt-opt:
|
||||||
|
|
||||||
**\-\-min-hist-cnt <INT=5>**
|
**\-\-min-hist-cnt <INT=5>**
|
||||||
When analyzing the k-mer spectrum, ignore counts below ``INT``. For very low coverage of HiFi data, set smaller value for this option. See `issue 45 <https://github.com/chhylp123/hifiasm/issues/49>`_ for example.
|
When analyzing the k-mer spectrum, ignore counts below ``INT``. For very low coverage of HiFi data, set smaller value for this option. See `issue 45 <https://github.com/chhylp123/hifiasm/issues/49>`_ for example.
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Assembly options
|
Assembly options
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
.. _aopt:
|
.. _aopt:
|
||||||
|
|
||||||
**\-a <INT=4>**
|
**\-a <INT=4>**
|
||||||
Rounds of assembly graph cleaning. This option is used with ``-x`` and ``-y``. Note that unlike -r, this option does not affect error corrected reads and all-to-all overlaps.
|
Rounds of assembly graph cleaning. This option is used with ``-x`` and ``-y``. Note that unlike -r, this option does not affect error corrected reads and all-to-all overlaps.
|
||||||
|
|
||||||
|
|
||||||
.. _mopt:
|
.. _mopt:
|
||||||
|
|
||||||
**\-m <INT=10000000>**
|
**\-m <INT=10000000>**
|
||||||
Maximal probing distance for bubble popping when generating primary/alternate contig graphs. Bubbles longer than ``INT`` bases will not be popped.
|
Maximal probing distance for bubble popping when generating primary/alternate contig graphs. Bubbles longer than ``INT`` bases will not be popped.
|
||||||
|
|
||||||
.. _popt:
|
.. _popt:
|
||||||
|
|
||||||
**\-p <INT=0>**
|
**\-p <INT=0>**
|
||||||
Maximal probing distance for bubble popping when generating haplotype-resolved processed unitig graph without small bubbles. Bubbles longer than ``INT`` bases will not be popped. Small bubbles might be caused by somatic mutations or noise in data. Please note that hifiasm automatically pops small bubbles based on coverage, which can be tweaked by ``--hom-cov``.
|
Maximal probing distance for bubble popping when generating haplotype-resolved processed unitig graph without small bubbles. Bubbles longer than ``INT`` bases will not be popped. Small bubbles might be caused by somatic mutations or noise in data. Please note that hifiasm automatically pops small bubbles based on coverage, which can be tweaked by ``--hom-cov``.
|
||||||
|
|
||||||
.. _nopt:
|
.. _nopt:
|
||||||
|
|
||||||
**\-n <INT=3>**
|
**\-n <INT=3>**
|
||||||
A unitig is considered small if it is composed of less than ``INT`` reads. Hifiasm may try to remove small unitigs at various steps.
|
A unitig is considered small if it is composed of less than ``INT`` reads. Hifiasm may try to remove small unitigs at various steps.
|
||||||
|
|
||||||
.. _xyopt:
|
.. _xyopt:
|
||||||
|
|
||||||
**\-x <FLOAT1=0.8>, \-y <FLOAT2=0.2>**
|
**\-x <FLOAT1=0.8>, \-y <FLOAT2=0.2>**
|
||||||
Max and min overlap drop ratio. This option is used with ``-a``. Given a node N in the assembly graph, let max(N) be the length of the longest overlap of N. Hifiasm iteratively drops overlaps of N if their length/max(N) is below a threshold controlled by ``-x`` and ``-y``. Hifiasm applies ``-a`` rounds of short overlap removal with an increasing threshold between ``FLOAT1`` and ``FLOAT2``.
|
Max and min overlap drop ratio. This option is used with ``-a``. Given a node N in the assembly graph, let max(N) be the length of the longest overlap of N. Hifiasm iteratively drops overlaps of N if their length/max(N) is below a threshold controlled by ``-x`` and ``-y``. Hifiasm applies ``-a`` rounds of short overlap removal with an increasing threshold between ``FLOAT1`` and ``FLOAT2``.
|
||||||
|
|
||||||
.. _iopt:
|
.. _iopt:
|
||||||
|
|
||||||
**\-i**
|
**\-i**
|
||||||
Ignore all bin files so that hifiasm will start again from scratch.
|
Ignore all bin files so that hifiasm will start again from scratch.
|
||||||
|
|
||||||
.. _uopt:
|
.. _uopt:
|
||||||
|
|
||||||
**\-u**
|
**\-u**
|
||||||
Disable post-join step for contigs which may improve N50. The post-join step of hifiasm improves contig N50 but may introduce misassemblies.
|
Disable post-join step for contigs which may improve N50. The post-join step of hifiasm improves contig N50 but may introduce misassemblies.
|
||||||
|
|
||||||
|
|
||||||
.. _hom-cov-opt:
|
.. _hom-cov-opt:
|
||||||
|
|
||||||
**\-\-hom-cov <INT>**
|
**\-\-hom-cov <INT>**
|
||||||
Homozygous read coverage inferred automatically in default. This option affects different types of outputs, including Hi-C phased assembly and HiFi-only assembly. For more details, see :ref:`hic-iss`, :ref:`p-large` and :ref:`loginter`.
|
Homozygous read coverage inferred automatically in default. This option affects different types of outputs, including Hi-C phased assembly and HiFi-only assembly. For more details, see :ref:`hic-iss`, :ref:`p-large` and :ref:`loginter`.
|
||||||
|
|
||||||
.. _pri-range-opt:
|
.. _pri-range-opt:
|
||||||
|
|
||||||
**\-\-pri-range <INT1[,INT2]>**
|
**\-\-pri-range <INT1[,INT2]>**
|
||||||
Min and max coverage cutoffs of primary contigs. Keep contigs with coverage in this range at p_ctg.gfa. Inferred automatically in default. If ``INT2`` is not specified, it is set to infinity. Set -1 to disable.
|
Min and max coverage cutoffs of primary contigs. Keep contigs with coverage in this range at p_ctg.gfa. Inferred automatically in default. If ``INT2`` is not specified, it is set to infinity. Set -1 to disable.
|
||||||
|
|
||||||
.. _lowQ-opt:
|
.. _lowQ-opt:
|
||||||
|
|
||||||
**\-\-lowQ <INT=70>**
|
**\-\-lowQ <INT=70>**
|
||||||
Output contig regions with ``>=INT%`` inconsistency to the bed file with suffix lowQ.bed. Set 0 to disable.
|
Output contig regions with ``>=INT%`` inconsistency to the bed file with suffix lowQ.bed. Set 0 to disable.
|
||||||
|
|
||||||
.. _b-cov-opt:
|
.. _b-cov-opt:
|
||||||
|
|
||||||
**\-\-b-cov <INT=0>**
|
**\-\-b-cov <INT=0>**
|
||||||
Break contigs at potential misassemblies with ``<INT``-fold coverage. Work with ``--m-rate``. Set 0 to disable.
|
Break contigs at potential misassemblies with ``<INT``-fold coverage. Work with ``--m-rate``. Set 0 to disable.
|
||||||
|
|
||||||
.. _h-cov-opt:
|
.. _h-cov-opt:
|
||||||
|
|
||||||
**\-\-h-cov <INT=-1>**
|
**\-\-h-cov <INT=-1>**
|
||||||
Break contigs at potential misassemblies with ``>INT``-fold coverage. Work with ``--m-rate``. Set -1 to disable.
|
Break contigs at potential misassemblies with ``>INT``-fold coverage. Work with ``--m-rate``. Set -1 to disable.
|
||||||
|
|
||||||
.. _m-rate-opt:
|
.. _m-rate-opt:
|
||||||
|
|
||||||
**\-\-m-rate <FLOAT=0.75>**
|
**\-\-m-rate <FLOAT=0.75>**
|
||||||
Break contigs with ``<=FLOAT*coverage`` exact overlaps. Only work when ``--b-cov`` and ``--h-cov`` are specified.
|
Break contigs with ``<=FLOAT*coverage`` exact overlaps. Only work when ``--b-cov`` and ``--h-cov`` are specified.
|
||||||
|
|
||||||
.. _primary-opt:
|
.. _primary-opt:
|
||||||
|
|
||||||
**\-\-primary**
|
**\-\-primary**
|
||||||
Output a primary assembly and an alternate assembly. Enable this option or ``-l0`` outputs a primary assembly and an alternate assembly.
|
Output a primary assembly and an alternate assembly. Enable this option or ``-l0`` outputs a primary assembly and an alternate assembly.
|
||||||
|
|
||||||
|
|
||||||
Trio-binning options
|
Trio-binning options
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
.. _1opt:
|
.. _1opt:
|
||||||
|
|
||||||
**\-1 <FILE>**
|
**\-1 <FILE>**
|
||||||
K-mer dump generated by `yak count <https://github.com/lh3/yak>`_ from the paternal/haplotype1 reads.
|
K-mer dump generated by `yak count <https://github.com/lh3/yak>`_ from the paternal/haplotype1 reads.
|
||||||
|
|
||||||
.. _2opt:
|
.. _2opt:
|
||||||
|
|
||||||
**\-2 <FILE>**
|
**\-2 <FILE>**
|
||||||
K-mer dump generated by `yak count <https://github.com/lh3/yak>`_ from the maternal/haplotype2 reads.
|
K-mer dump generated by `yak count <https://github.com/lh3/yak>`_ from the maternal/haplotype2 reads.
|
||||||
|
|
||||||
.. _3opt:
|
.. _3opt:
|
||||||
|
|
||||||
**\-3 <FILE>**
|
**\-3 <FILE>**
|
||||||
List of paternal/haplotype1 read names.
|
List of paternal/haplotype1 read names.
|
||||||
|
|
||||||
.. _4opt:
|
.. _4opt:
|
||||||
|
|
||||||
**\-4 <FILE>**
|
**\-4 <FILE>**
|
||||||
List of maternal/haplotype2 read names.
|
List of maternal/haplotype2 read names.
|
||||||
|
|
||||||
.. _cdopt:
|
.. _cdopt:
|
||||||
|
|
||||||
**\-c <INT1=2>, -d <INT2=5>**
|
**\-c <INT1=2>, -d <INT2=5>**
|
||||||
Lower bound and upper bound of the binned k-mer's frequency. When doing trio binning, a k-mer is said to be differentiating if it occurs >= ``INT2`` times in one sample but occurs < ``INT1`` times in the other sample.
|
Lower bound and upper bound of the binned k-mer's frequency. When doing trio binning, a k-mer is said to be differentiating if it occurs >= ``INT2`` times in one sample but occurs < ``INT1`` times in the other sample.
|
||||||
|
|
||||||
|
|
||||||
.. _t-occ-opt:
|
.. _t-occ-opt:
|
||||||
|
|
||||||
**\-\-t-occ <INT=60>**
|
**\-\-t-occ <INT=60>**
|
||||||
Forcedly remove unitig including ``>INT`` unexpected haplotype-specific reads without considering graph topology. For more details, see :ref:`p-hamming`.
|
Forcedly remove unitig including ``>INT`` unexpected haplotype-specific reads without considering graph topology. For more details, see :ref:`p-hamming`.
|
||||||
|
|
||||||
|
|
||||||
Purge duplication options
|
Purge duplication options
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
.. _ldopt:
|
.. _ldopt:
|
||||||
|
|
||||||
**\-l <INT=3>**
|
**\-l <INT=3>**
|
||||||
Level of purge duplication. 0 to disable, 1 to only purge contained haplotigs, 2 to purge all types of haplotigs, 3 to purge all types of haplotigs in the most aggressive way. In default, 3 for non-trio assembly, 0 for trio-binning assembly. For trio-binning assembly, only level 0 and level 1 are allowed.
|
Level of purge duplication. 0 to disable, 1 to only purge contained haplotigs, 2 to purge all types of haplotigs, 3 to purge all types of haplotigs in the most aggressive way. In default, 3 for non-trio assembly, 0 for trio-binning assembly. For trio-binning assembly, only level 0 and level 1 are allowed.
|
||||||
|
|
||||||
.. _sdopt:
|
.. _sdopt:
|
||||||
|
|
||||||
**\-s <FLOAT=0.55>**
|
**\-s <FLOAT=0.55>**
|
||||||
Similarity threshold for duplicate haplotigs that should be purged. In default, 0.75 for ``-l1/-l2``, 0.55 for ``-l3``. This option affects both HiFi-only assembly and Hi-C phased assembly. For more details, see :ref:`hic-iss` and :ref:`p-large`.
|
Similarity threshold for duplicate haplotigs that should be purged. In default, 0.75 for ``-l1/-l2``, 0.55 for ``-l3``. This option affects both HiFi-only assembly and Hi-C phased assembly. For more details, see :ref:`hic-iss` and :ref:`p-large`.
|
||||||
|
|
||||||
.. _ovlpdopt:
|
.. _ovlpdopt:
|
||||||
|
|
||||||
**\-O <INT=1>**
|
**\-O <INT=1>**
|
||||||
Min number of overlapped reads for duplicate haplotigs that should be purged.
|
Min number of overlapped reads for duplicate haplotigs that should be purged.
|
||||||
|
|
||||||
.. _purgeopt:
|
.. _purgeopt:
|
||||||
|
|
||||||
**\-\-purge-max <INT>**
|
**\-\-purge-max <INT>**
|
||||||
Coverage upper bound of purge duplication, which is inferred automatically in default. If the coverage of a contig is higher than this bound, don't apply purge duplication. Larger value makes assembly more contiguous but may collapse repeats or segmental duplications.
|
Coverage upper bound of purge duplication, which is inferred automatically in default. If the coverage of a contig is higher than this bound, don't apply purge duplication. Larger value makes assembly more contiguous but may collapse repeats or segmental duplications.
|
||||||
|
|
||||||
.. _nhapopt:
|
.. _nhapopt:
|
||||||
|
|
||||||
**\-\-n\-hap <INT=2>**
|
**\-\-n\-hap <INT=2>**
|
||||||
Assumption of haplotype number. If it is set to >2, the quality of primary assembly for polyploid genomes might be improved.
|
Assumption of haplotype number. If it is set to >2, the quality of primary assembly for polyploid genomes might be improved.
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Hi-C integration options
|
Hi-C integration options
|
||||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
.. _h1opt:
|
.. _h1opt:
|
||||||
|
|
||||||
**\-\-h1 <FILEs>**
|
**\-\-h1 <FILEs>**
|
||||||
File names of input Hi-C R1 ``[r1_1.fq,r1_2.fq,...]``.
|
File names of input Hi-C R1 ``[r1_1.fq,r1_2.fq,...]``.
|
||||||
|
|
||||||
.. _h2opt:
|
.. _h2opt:
|
||||||
|
|
||||||
**\-\-h2 <FILEs>**
|
**\-\-h2 <FILEs>**
|
||||||
File names of input Hi-C R2 ``[r2_1.fq,r2_2.fq,...]``.
|
File names of input Hi-C R2 ``[r2_1.fq,r2_2.fq,...]``.
|
||||||
|
|
||||||
.. _n-weightopt:
|
.. _n-weightopt:
|
||||||
|
|
||||||
**\-\-n-weight <INT=3>**
|
**\-\-n-weight <INT=3>**
|
||||||
Rounds of reweighting Hi-C links. Raising this option may improve phasing results but takes longer time.
|
Rounds of reweighting Hi-C links. Raising this option may improve phasing results but takes longer time.
|
||||||
|
|
||||||
.. _n-perturbopt:
|
.. _n-perturbopt:
|
||||||
|
|
||||||
**\-\-n-perturb <INT=10000>**
|
**\-\-n-perturb <INT=10000>**
|
||||||
Rounds of perturbation. Increasing this option may improve phasing results but takes longer time.
|
Rounds of perturbation. Increasing this option may improve phasing results but takes longer time.
|
||||||
|
|
||||||
.. _f-perturbopt:
|
.. _f-perturbopt:
|
||||||
|
|
||||||
**\-\-f-perturb <FLOAT=0.1>**
|
**\-\-f-perturb <FLOAT=0.1>**
|
||||||
Fraction to flip for perturbation. Increasing this option may improve phasing results but takes longer time.
|
Fraction to flip for perturbation. Increasing this option may improve phasing results but takes longer time.
|
||||||
|
|
||||||
.. _seedopt:
|
.. _seedopt:
|
||||||
|
|
||||||
**\-\-seed <INT=11>**
|
**\-\-seed <INT=11>**
|
||||||
RNG seed.
|
RNG seed.
|
||||||
|
|
||||||
|
|
||||||
.. _l-msjoin:
|
.. _l-msjoin:
|
||||||
|
|
||||||
**\-\-l-msjoin <INT=500000>**
|
**\-\-l-msjoin <INT=500000>**
|
||||||
Detect misjoined unitigs of ``>=INT`` in size; 0 to disable.
|
Detect misjoined unitigs of ``>=INT`` in size; 0 to disable.
|
||||||
|
|||||||
@@ -1,29 +1,29 @@
|
|||||||
|
|
||||||
.. _trio-assembly:
|
.. _trio-assembly:
|
||||||
|
|
||||||
Trio-binning Assembly
|
Trio-binning Assembly
|
||||||
=====================
|
=====================
|
||||||
|
|
||||||
When parental short reads are available, hifiasm can also generate a pair of haplotype-resolved assemblies with trio binning. To perform such assembly, you need to count k-mers first with `yak <https://github.com/lh3/yak>`_ and then do assembly::
|
When parental short reads are available, hifiasm can also generate a pair of haplotype-resolved assemblies with trio binning. To perform such assembly, you need to count k-mers first with `yak <https://github.com/lh3/yak>`_ and then do assembly::
|
||||||
|
|
||||||
yak count -k31 -b37 -t16 -o pat.yak paternal.fq.gz
|
yak count -k31 -b37 -t16 -o pat.yak paternal.fq.gz
|
||||||
yak count -k31 -b37 -t16 -o mat.yak maternal.fq.gz
|
yak count -k31 -b37 -t16 -o mat.yak maternal.fq.gz
|
||||||
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak NA12878.fq.gz
|
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak NA12878.fq.gz
|
||||||
|
|
||||||
Here ``NA12878.asm.hap1.p_ctg.gfa`` and ``NA12878.asm.hap2.p_ctg.gfa`` give the assemblies for two haplotypes. In the binning mode, hifiasm does not purge haplotig duplicates by default. Because hifiasm reuses saved overlaps, you can generate both primary/alternate assemblies and trio binning assemblies with::
|
Here ``NA12878.asm.hap1.p_ctg.gfa`` and ``NA12878.asm.hap2.p_ctg.gfa`` give the assemblies for two haplotypes. In the binning mode, hifiasm does not purge haplotig duplicates by default. Because hifiasm reuses saved overlaps, you can generate both primary/alternate assemblies and trio binning assemblies with::
|
||||||
|
|
||||||
hifiasm -o NA12878.asm --primary -t 32 NA12878.fq.gz 2> NA12878.asm.pri.log
|
hifiasm -o NA12878.asm --primary -t 32 NA12878.fq.gz 2> NA12878.asm.pri.log
|
||||||
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak /dev/null 2> NA12878.asm.trio.log
|
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak /dev/null 2> NA12878.asm.trio.log
|
||||||
|
|
||||||
The second command line will run much faster than the first. The phasing switch error rate and hamming error rate are able to be evaluated quickly by `yak <https://github.com/lh3/yak>`_::
|
The second command line will run much faster than the first. The phasing switch error rate and hamming error rate are able to be evaluated quickly by `yak <https://github.com/lh3/yak>`_::
|
||||||
|
|
||||||
yak trioeval -t16 pat.yak mat.yak assembly.fa
|
yak trioeval -t16 pat.yak mat.yak assembly.fa
|
||||||
|
|
||||||
The W-line and H-line reported by ``yak trioeval`` indicate switch error rate and hamming error rate respectively::
|
The W-line and H-line reported by ``yak trioeval`` indicate switch error rate and hamming error rate respectively::
|
||||||
|
|
||||||
W 26714 3029448 0.008818
|
W 26714 3029448 0.008818
|
||||||
H 24315 3029885 0.008025
|
H 24315 3029885 0.008025
|
||||||
|
|
||||||
For this example, the switch error rate is 0.8818% and the hamming error rate is 0.8025%. If the hamming error rate or the swith error rate of trio-binning assembly is very high, it might be caused by hifiasm or the incorrect parental data. To fix it, see :ref:`p-hamming` for more details.
|
For this example, the switch error rate is 0.8818% and the hamming error rate is 0.8025%. If the hamming error rate or the swith error rate of trio-binning assembly is very high, it might be caused by hifiasm or the incorrect parental data. To fix it, see :ref:`p-hamming` for more details.
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
+7203
-6512
File diff suppressed because it is too large
Load Diff
@@ -1,20 +1,20 @@
|
|||||||
#ifndef __ECOVLP_PARSER__
|
#ifndef __ECOVLP_PARSER__
|
||||||
#define __ECOVLP_PARSER__
|
#define __ECOVLP_PARSER__
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include "Hash_Table.h"
|
#include "Hash_Table.h"
|
||||||
#include "Process_Read.h"
|
#include "Process_Read.h"
|
||||||
#include "kdq.h"
|
#include "kdq.h"
|
||||||
|
|
||||||
|
|
||||||
void prt_chain(overlap_region_alloc *o);
|
void prt_chain(overlap_region_alloc *o);
|
||||||
void cal_ec_r(uint64_t n_thre, uint64_t round, uint64_t n_round, uint64_t n_a, uint64_t is_sv, uint64_t *tot_b, uint64_t *tot_e);
|
void cal_ec_r(uint64_t n_thre, uint64_t round, uint64_t n_round, uint64_t n_a, uint64_t is_sv, uint64_t *tot_b, uint64_t *tot_e);
|
||||||
void sl_ec_r(uint64_t n_thre, uint64_t n_a);
|
void sl_ec_r(uint64_t n_thre, uint64_t n_a);
|
||||||
void cal_ov_r(uint64_t n_thre, uint64_t n_a, uint64_t new_idx);
|
void cal_ov_r(uint64_t n_thre, uint64_t n_a, uint64_t new_idx);
|
||||||
void handle_chemical_r(uint64_t n_thre, uint64_t n_a);
|
void handle_chemical_r(uint64_t n_thre, uint64_t n_a);
|
||||||
void handle_chemical_arc(uint64_t n_thre, uint64_t n_a);
|
void handle_chemical_arc(uint64_t n_thre, uint64_t n_a);
|
||||||
uint8_t* gen_chemical_arc_rf(uint64_t n_thre, uint64_t n_a);
|
uint8_t* gen_chemical_arc_rf(uint64_t n_thre, uint64_t n_a);
|
||||||
void cal_ec_r_dbg(uint64_t n_thre, uint64_t n_a);
|
void cal_ec_r_dbg(uint64_t n_thre, uint64_t n_a);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
+173
-173
@@ -1,173 +1,173 @@
|
|||||||
#include <zlib.h>
|
#include <zlib.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include "Process_Read.h"
|
#include "Process_Read.h"
|
||||||
#include "khashl.h"
|
#include "khashl.h"
|
||||||
#include "kseq.h"
|
#include "kseq.h"
|
||||||
|
|
||||||
typedef const char *cstr_t;
|
typedef const char *cstr_t;
|
||||||
KHASHL_CSET_INIT(KH_LOCAL, strset_t, ss, cstr_t, kh_hash_str, kh_eq_str)
|
KHASHL_CSET_INIT(KH_LOCAL, strset_t, ss, cstr_t, kh_hash_str, kh_eq_str)
|
||||||
KHASHL_MAP_INIT(KH_LOCAL, hm64_t, h64, uint64_t, int, kh_hash_uint64, kh_eq_generic)
|
KHASHL_MAP_INIT(KH_LOCAL, hm64_t, h64, uint64_t, int, kh_hash_uint64, kh_eq_generic)
|
||||||
KSTREAM_INIT(gzFile, gzread, 65536)
|
KSTREAM_INIT(gzFile, gzread, 65536)
|
||||||
|
|
||||||
#define GFA_MALLOC(ptr, len) ((ptr) = (__typeof__(ptr))malloc((len) * sizeof(*(ptr))))
|
#define GFA_MALLOC(ptr, len) ((ptr) = (__typeof__(ptr))malloc((len) * sizeof(*(ptr))))
|
||||||
#define GFA_REALLOC(ptr, len) ((ptr) = (__typeof__(ptr))realloc((ptr), (len) * sizeof(*(ptr))))
|
#define GFA_REALLOC(ptr, len) ((ptr) = (__typeof__(ptr))realloc((ptr), (len) * sizeof(*(ptr))))
|
||||||
|
|
||||||
char *gfa_strdup(const char *src)
|
char *gfa_strdup(const char *src)
|
||||||
{
|
{
|
||||||
int32_t len;
|
int32_t len;
|
||||||
char *dst;
|
char *dst;
|
||||||
len = strlen(src);
|
len = strlen(src);
|
||||||
GFA_MALLOC(dst, len + 1);
|
GFA_MALLOC(dst, len + 1);
|
||||||
memcpy(dst, src, len + 1);
|
memcpy(dst, src, len + 1);
|
||||||
return dst;
|
return dst;
|
||||||
}
|
}
|
||||||
|
|
||||||
char *gfa_strndup(const char *src, size_t n)
|
char *gfa_strndup(const char *src, size_t n)
|
||||||
{
|
{
|
||||||
char *dst;
|
char *dst;
|
||||||
GFA_MALLOC(dst, n + 1);
|
GFA_MALLOC(dst, n + 1);
|
||||||
strncpy(dst, src, n);
|
strncpy(dst, src, n);
|
||||||
dst[n] = 0;
|
dst[n] = 0;
|
||||||
return dst;
|
return dst;
|
||||||
}
|
}
|
||||||
|
|
||||||
char **gv_read_list(const char *o, int *n_)
|
char **gv_read_list(const char *o, int *n_)
|
||||||
{
|
{
|
||||||
int n = 0, m = 0;
|
int n = 0, m = 0;
|
||||||
char **s = 0;
|
char **s = 0;
|
||||||
*n_ = 0;
|
*n_ = 0;
|
||||||
if (*o != '@') {
|
if (*o != '@') {
|
||||||
const char *q = o, *p;
|
const char *q = o, *p;
|
||||||
for (p = q;; ++p) {
|
for (p = q;; ++p) {
|
||||||
if (*p == ',' || *p == 0) {
|
if (*p == ',' || *p == 0) {
|
||||||
if (n == m) {
|
if (n == m) {
|
||||||
m = m? m<<1 : 16;
|
m = m? m<<1 : 16;
|
||||||
GFA_REALLOC(s, m);
|
GFA_REALLOC(s, m);
|
||||||
}
|
}
|
||||||
s[n++] = gfa_strndup(q, p - q);
|
s[n++] = gfa_strndup(q, p - q);
|
||||||
if (*p == 0) break;
|
if (*p == 0) break;
|
||||||
q = p + 1;
|
q = p + 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
gzFile fp;
|
gzFile fp;
|
||||||
kstream_t *ks;
|
kstream_t *ks;
|
||||||
kstring_t str = {0,0,0};
|
kstring_t str = {0,0,0};
|
||||||
int dret;
|
int dret;
|
||||||
|
|
||||||
fp = gzopen(o + 1, "r");
|
fp = gzopen(o + 1, "r");
|
||||||
if (fp == 0) return 0;
|
if (fp == 0) return 0;
|
||||||
ks = ks_init(fp);
|
ks = ks_init(fp);
|
||||||
while (ks_getuntil(ks, KS_SEP_LINE, &str, &dret) >= 0) {
|
while (ks_getuntil(ks, KS_SEP_LINE, &str, &dret) >= 0) {
|
||||||
char *p;
|
char *p;
|
||||||
for (p = str.s; *p && !isspace(*p); ++p);
|
for (p = str.s; *p && !isspace(*p); ++p);
|
||||||
if (n == m) {
|
if (n == m) {
|
||||||
m = m? m<<1 : 16;
|
m = m? m<<1 : 16;
|
||||||
GFA_REALLOC(s, m);
|
GFA_REALLOC(s, m);
|
||||||
}
|
}
|
||||||
s[n++] = gfa_strndup(str.s, p - str.s);
|
s[n++] = gfa_strndup(str.s, p - str.s);
|
||||||
}
|
}
|
||||||
ks_destroy(ks);
|
ks_destroy(ks);
|
||||||
gzclose(fp);
|
gzclose(fp);
|
||||||
}
|
}
|
||||||
if (s) s = (char**)realloc(s, n * sizeof(char*));
|
if (s) s = (char**)realloc(s, n * sizeof(char*));
|
||||||
*n_ = n;
|
*n_ = n;
|
||||||
return s;
|
return s;
|
||||||
}
|
}
|
||||||
|
|
||||||
void ha_extract_print(const All_reads *rs, int n_rounds, int n, char **list)
|
void ha_extract_print(const All_reads *rs, int n_rounds, int n, char **list)
|
||||||
{
|
{
|
||||||
hm64_t *h;
|
hm64_t *h;
|
||||||
khint_t k;
|
khint_t k;
|
||||||
int i, absent, m, l;
|
int i, absent, m, l;
|
||||||
uint64_t j;
|
uint64_t j;
|
||||||
const ma_hit_t_alloc *ov[2] = { rs->paf, rs->reverse_paf };
|
const ma_hit_t_alloc *ov[2] = { rs->paf, rs->reverse_paf };
|
||||||
FILE *fp = stdout;
|
FILE *fp = stdout;
|
||||||
|
|
||||||
if (n > 0) {
|
if (n > 0) {
|
||||||
int max_len = 0;
|
int max_len = 0;
|
||||||
char *s = 0;
|
char *s = 0;
|
||||||
strset_t *ss;
|
strset_t *ss;
|
||||||
ss = ss_init();
|
ss = ss_init();
|
||||||
for (i = 0; i < n; ++i)
|
for (i = 0; i < n; ++i)
|
||||||
ss_put(ss, list[i], &absent);
|
ss_put(ss, list[i], &absent);
|
||||||
for (j = 0; j < rs->total_reads; ++j)
|
for (j = 0; j < rs->total_reads; ++j)
|
||||||
if (max_len < (int)Get_NAME_LENGTH(*rs, j))
|
if (max_len < (int)Get_NAME_LENGTH(*rs, j))
|
||||||
max_len = Get_NAME_LENGTH(*rs, j);
|
max_len = Get_NAME_LENGTH(*rs, j);
|
||||||
GFA_MALLOC(s, max_len + 1);
|
GFA_MALLOC(s, max_len + 1);
|
||||||
h = h64_init();
|
h = h64_init();
|
||||||
for (j = 0; j < rs->total_reads; ++j) {
|
for (j = 0; j < rs->total_reads; ++j) {
|
||||||
strncpy(s, Get_NAME(*rs, j), Get_NAME_LENGTH(*rs, j));
|
strncpy(s, Get_NAME(*rs, j), Get_NAME_LENGTH(*rs, j));
|
||||||
s[Get_NAME_LENGTH(*rs, j)] = 0;
|
s[Get_NAME_LENGTH(*rs, j)] = 0;
|
||||||
if (ss_get(ss, s) != kh_end(ss)) {
|
if (ss_get(ss, s) != kh_end(ss)) {
|
||||||
k = h64_put(h, j, &absent);
|
k = h64_put(h, j, &absent);
|
||||||
kh_val(h, k) = -1;
|
kh_val(h, k) = -1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
free(s);
|
free(s);
|
||||||
ss_destroy(ss);
|
ss_destroy(ss);
|
||||||
} else return;
|
} else return;
|
||||||
|
|
||||||
for (m = 0; m < n_rounds; ++m) {
|
for (m = 0; m < n_rounds; ++m) {
|
||||||
for (j = 0; j < rs->total_reads; ++j) {
|
for (j = 0; j < rs->total_reads; ++j) {
|
||||||
for (l = 0; l < 2; ++l) {
|
for (l = 0; l < 2; ++l) {
|
||||||
const ma_hit_t_alloc *o = &ov[l][j];
|
const ma_hit_t_alloc *o = &ov[l][j];
|
||||||
for (i = 0; i < (int)o->length; ++i) {
|
for (i = 0; i < (int)o->length; ++i) {
|
||||||
uint64_t q = Get_qn(o->buffer[i]);
|
uint64_t q = Get_qn(o->buffer[i]);
|
||||||
uint64_t t = Get_tn(o->buffer[i]);
|
uint64_t t = Get_tn(o->buffer[i]);
|
||||||
int q_hit = 0, t_hit = 0;
|
int q_hit = 0, t_hit = 0;
|
||||||
k = h64_get(h, q);
|
k = h64_get(h, q);
|
||||||
q_hit = (k < kh_end(h) && kh_val(h, k) < m);
|
q_hit = (k < kh_end(h) && kh_val(h, k) < m);
|
||||||
k = h64_get(h, t);
|
k = h64_get(h, t);
|
||||||
t_hit = (k < kh_end(h) && kh_val(h, k) < m);
|
t_hit = (k < kh_end(h) && kh_val(h, k) < m);
|
||||||
if ((!q_hit && !t_hit) || (q_hit && t_hit)) continue;
|
if ((!q_hit && !t_hit) || (q_hit && t_hit)) continue;
|
||||||
if (!q_hit) {
|
if (!q_hit) {
|
||||||
k = h64_put(h, q, &absent);
|
k = h64_put(h, q, &absent);
|
||||||
if (absent) kh_val(h, k) = m;
|
if (absent) kh_val(h, k) = m;
|
||||||
}
|
}
|
||||||
if (!t_hit) {
|
if (!t_hit) {
|
||||||
k = h64_put(h, t, &absent);
|
k = h64_put(h, t, &absent);
|
||||||
if (absent) kh_val(h, k) = m;
|
if (absent) kh_val(h, k) = m;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
for (j = 0; j < rs->total_reads; ++j) {
|
for (j = 0; j < rs->total_reads; ++j) {
|
||||||
for (l = 0; l < 2; ++l) {
|
for (l = 0; l < 2; ++l) {
|
||||||
const ma_hit_t_alloc *o = &ov[l][j];
|
const ma_hit_t_alloc *o = &ov[l][j];
|
||||||
for (i = 0; i < (int)o->length; ++i) {
|
for (i = 0; i < (int)o->length; ++i) {
|
||||||
uint64_t q = Get_qn(o->buffer[i]);
|
uint64_t q = Get_qn(o->buffer[i]);
|
||||||
uint64_t t = Get_tn(o->buffer[i]);
|
uint64_t t = Get_tn(o->buffer[i]);
|
||||||
int q_hit = 0, t_hit = 0;
|
int q_hit = 0, t_hit = 0;
|
||||||
q_hit = (h64_get(h, q) < kh_end(h));
|
q_hit = (h64_get(h, q) < kh_end(h));
|
||||||
t_hit = (h64_get(h, t) < kh_end(h));
|
t_hit = (h64_get(h, t) < kh_end(h));
|
||||||
if (!q_hit && !t_hit) continue;
|
if (!q_hit && !t_hit) continue;
|
||||||
fwrite(Get_NAME(*rs, q), 1, Get_NAME_LENGTH(*rs, q), fp);
|
fwrite(Get_NAME(*rs, q), 1, Get_NAME_LENGTH(*rs, q), fp);
|
||||||
fwrite("\t", 1, 1, fp);
|
fwrite("\t", 1, 1, fp);
|
||||||
fprintf(fp, "%lu\t", (unsigned long)Get_READ_LENGTH(*rs, q));
|
fprintf(fp, "%lu\t", (unsigned long)Get_READ_LENGTH(*rs, q));
|
||||||
fprintf(fp, "%d\t", Get_qs(o->buffer[i]));
|
fprintf(fp, "%d\t", Get_qs(o->buffer[i]));
|
||||||
fprintf(fp, "%d\t", Get_qe(o->buffer[i]));
|
fprintf(fp, "%d\t", Get_qe(o->buffer[i]));
|
||||||
fputs(o->buffer[i].rev? "-\t" : "+\t", fp);
|
fputs(o->buffer[i].rev? "-\t" : "+\t", fp);
|
||||||
fwrite(Get_NAME(*rs, t), 1, Get_NAME_LENGTH(*rs, t), fp);
|
fwrite(Get_NAME(*rs, t), 1, Get_NAME_LENGTH(*rs, t), fp);
|
||||||
fwrite("\t", 1, 1, fp);
|
fwrite("\t", 1, 1, fp);
|
||||||
fprintf(fp, "%lu\t", (unsigned long)Get_READ_LENGTH(*rs, t));
|
fprintf(fp, "%lu\t", (unsigned long)Get_READ_LENGTH(*rs, t));
|
||||||
fprintf(fp, "%d\t", Get_ts(o->buffer[i]));
|
fprintf(fp, "%d\t", Get_ts(o->buffer[i]));
|
||||||
fprintf(fp, "%d\t%d\t%d\t%d\n", Get_te(o->buffer[i]), o->buffer[i].ml, o->buffer[i].bl, !l);
|
fprintf(fp, "%d\t%d\t%d\t%d\n", Get_te(o->buffer[i]), o->buffer[i].ml, o->buffer[i].bl, !l);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
h64_destroy(h);
|
h64_destroy(h);
|
||||||
}
|
}
|
||||||
|
|
||||||
void ha_extract_print_list(const All_reads *rs, int n_rounds, const char *o)
|
void ha_extract_print_list(const All_reads *rs, int n_rounds, const char *o)
|
||||||
{
|
{
|
||||||
int i, n;
|
int i, n;
|
||||||
char **list;
|
char **list;
|
||||||
list = gv_read_list(o, &n);
|
list = gv_read_list(o, &n);
|
||||||
ha_extract_print(rs, n_rounds, n, list);
|
ha_extract_print(rs, n_rounds, n, list);
|
||||||
for (i = 0; i < n; ++i) free(list[i]);
|
for (i = 0; i < n; ++i) free(list[i]);
|
||||||
free(list);
|
free(list);
|
||||||
}
|
}
|
||||||
|
|||||||
+203
-203
@@ -1,204 +1,204 @@
|
|||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <assert.h>
|
#include <assert.h>
|
||||||
#include <zlib.h>
|
#include <zlib.h>
|
||||||
#include <math.h>
|
#include <math.h>
|
||||||
#include "kseq.h" // FASTA/Q parser
|
#include "kseq.h" // FASTA/Q parser
|
||||||
#include "kavl.h"
|
#include "kavl.h"
|
||||||
#include "khash.h"
|
#include "khash.h"
|
||||||
#include "kalloc.h"
|
#include "kalloc.h"
|
||||||
#include "kthread.h"
|
#include "kthread.h"
|
||||||
#include "inter.h"
|
#include "inter.h"
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
#include "CommandLines.h"
|
#include "CommandLines.h"
|
||||||
#include "htab.h"
|
#include "htab.h"
|
||||||
#include "Hash_Table.h"
|
#include "Hash_Table.h"
|
||||||
#include "Correct.h"
|
#include "Correct.h"
|
||||||
#include "Process_Read.h"
|
#include "Process_Read.h"
|
||||||
#include "Assembly.h"
|
#include "Assembly.h"
|
||||||
#include "gchain_map.h"
|
#include "gchain_map.h"
|
||||||
KSEQ_INIT(gzFile, gzread)
|
KSEQ_INIT(gzFile, gzread)
|
||||||
void ul_map_lchain(ha_abufl_t *ab, uint32_t rid, char* rs, uint64_t rl, uint64_t mz_w, uint64_t mz_k, const ul_idx_t *uref, overlap_region_alloc *overlap_list, Candidates_list *cl, double bw_thres,
|
void ul_map_lchain(ha_abufl_t *ab, uint32_t rid, char* rs, uint64_t rl, uint64_t mz_w, uint64_t mz_k, const ul_idx_t *uref, overlap_region_alloc *overlap_list, Candidates_list *cl, double bw_thres,
|
||||||
int max_n_chain, int apend_be, kvec_t_u8_warp* k_flag, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct, st_mt_t *sp, uint32_t *high_occ, uint32_t *low_occ, uint32_t is_accurate, uint32_t gen_off, double mcopy_rate, uint32_t chain_cutoff, uint32_t mcopy_khit_cut);
|
int max_n_chain, int apend_be, kvec_t_u8_warp* k_flag, overlap_region* f_cigar, kvec_t_u64_warp* dbg_ct, st_mt_t *sp, uint32_t *high_occ, uint32_t *low_occ, uint32_t is_accurate, uint32_t gen_off, double mcopy_rate, uint32_t chain_cutoff, uint32_t mcopy_khit_cut);
|
||||||
|
|
||||||
typedef struct { // global data structure for kt_pipeline()
|
typedef struct { // global data structure for kt_pipeline()
|
||||||
const void *ha_flt_tab;
|
const void *ha_flt_tab;
|
||||||
const ha_pt_t *ha_idx;
|
const ha_pt_t *ha_idx;
|
||||||
const mg_idxopt_t *opt;
|
const mg_idxopt_t *opt;
|
||||||
const ma_ug_t *ug;
|
const ma_ug_t *ug;
|
||||||
const asg_t *rg;
|
const asg_t *rg;
|
||||||
const ug_opt_t *uopt;
|
const ug_opt_t *uopt;
|
||||||
const ul_idx_t *uu;
|
const ul_idx_t *uu;
|
||||||
ucr_file_t *ucr_s;
|
ucr_file_t *ucr_s;
|
||||||
kseq_t *ks;
|
kseq_t *ks;
|
||||||
int64_t chunk_size;
|
int64_t chunk_size;
|
||||||
uint64_t n_thread;
|
uint64_t n_thread;
|
||||||
uint64_t total_base;
|
uint64_t total_base;
|
||||||
uint64_t total_pair;
|
uint64_t total_pair;
|
||||||
uint64_t num_bases, num_corrected_bases, num_recorrected_bases;
|
uint64_t num_bases, num_corrected_bases, num_recorrected_bases;
|
||||||
uint64_t remap, mini_cut;
|
uint64_t remap, mini_cut;
|
||||||
} gmap_t;
|
} gmap_t;
|
||||||
|
|
||||||
typedef struct { // data structure for each step in kt_pipeline()
|
typedef struct { // data structure for each step in kt_pipeline()
|
||||||
const mg_idxopt_t *opt;
|
const mg_idxopt_t *opt;
|
||||||
const void *ha_flt_tab;
|
const void *ha_flt_tab;
|
||||||
const ha_pt_t *ha_idx;
|
const ha_pt_t *ha_idx;
|
||||||
const ma_ug_t *ug;
|
const ma_ug_t *ug;
|
||||||
const asg_t *rg;
|
const asg_t *rg;
|
||||||
const ug_opt_t *uopt;
|
const ug_opt_t *uopt;
|
||||||
const ul_idx_t *uu;
|
const ul_idx_t *uu;
|
||||||
int n, m, sum_len;
|
int n, m, sum_len;
|
||||||
uint64_t *len, id;
|
uint64_t *len, id;
|
||||||
char **seq;
|
char **seq;
|
||||||
// ha_mzl_v *mzs;///useless
|
// ha_mzl_v *mzs;///useless
|
||||||
// st_mt_t *sps;///useless
|
// st_mt_t *sps;///useless
|
||||||
// mg_gchains_t **gcs;///useless
|
// mg_gchains_t **gcs;///useless
|
||||||
mg_tbuf_t **buf;///useless
|
mg_tbuf_t **buf;///useless
|
||||||
ha_ovec_buf_t **hab;
|
ha_ovec_buf_t **hab;
|
||||||
kv_ul_ov_t *res;
|
kv_ul_ov_t *res;
|
||||||
// glchain_t *ll;
|
// glchain_t *ll;
|
||||||
// gdpchain_t *gdp;
|
// gdpchain_t *gdp;
|
||||||
// glchain_t *sec_ll;
|
// glchain_t *sec_ll;
|
||||||
uint64_t num_bases, num_corrected_bases, num_recorrected_bases, mini_cut;
|
uint64_t num_bases, num_corrected_bases, num_recorrected_bases, mini_cut;
|
||||||
int64_t n_thread;
|
int64_t n_thread;
|
||||||
} sstep_t;
|
} sstep_t;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
static void worker_ul_map(void *data, long i, int tid) // callback for kt_for()
|
static void worker_ul_map(void *data, long i, int tid) // callback for kt_for()
|
||||||
{
|
{
|
||||||
sstep_t *s = (sstep_t*)data;
|
sstep_t *s = (sstep_t*)data;
|
||||||
ha_ovec_buf_t *b = s->hab[tid];
|
ha_ovec_buf_t *b = s->hab[tid];
|
||||||
kv_ul_ov_t *res = (s->res?(&(s->res[tid])):(NULL));
|
kv_ul_ov_t *res = (s->res?(&(s->res[tid])):(NULL));
|
||||||
mg_tbuf_t *buf = (s->buf?s->buf[tid]:NULL);
|
mg_tbuf_t *buf = (s->buf?s->buf[tid]:NULL);
|
||||||
int64_t winLen = MIN((((double)THRESHOLD_MAX_SIZE)/s->opt->diff_ec_ul), WINDOW);
|
int64_t winLen = MIN((((double)THRESHOLD_MAX_SIZE)/s->opt->diff_ec_ul), WINDOW);
|
||||||
int fully_cov, abnormal;
|
int fully_cov, abnormal;
|
||||||
assert(UL_INF.a[s->id+i].rlen == s->len[i]);
|
assert(UL_INF.a[s->id+i].rlen == s->len[i]);
|
||||||
// if(s->id+i!=43) return;
|
// if(s->id+i!=43) return;
|
||||||
|
|
||||||
ul_map_lchain(b->abl, i, s->seq[i], s->len[i], s->opt->w, s->opt->k, s->uu, &b->olist, &b->olist_hp, &b->clist, s->opt->bw_thres,
|
ul_map_lchain(b->abl, i, s->seq[i], s->len[i], s->opt->w, s->opt->k, s->uu, &b->olist, &b->olist_hp, &b->clist, s->opt->bw_thres,
|
||||||
s->opt->max_n_chain, 1, NULL, &(b->tmp_region), NULL, &(b->sp), s->mini_cut, 0);
|
s->opt->max_n_chain, 1, NULL, &(b->tmp_region), NULL, &(b->sp), s->mini_cut, 0);
|
||||||
|
|
||||||
clear_Cigar_record(&b->cigar1);
|
clear_Cigar_record(&b->cigar1);
|
||||||
clear_Round2_alignment(&b->round2);
|
clear_Round2_alignment(&b->round2);
|
||||||
|
|
||||||
b->self_read.seq = s->seq[i]; b->self_read.length = s->len[i]; b->self_read.size = 0;
|
b->self_read.seq = s->seq[i]; b->self_read.length = s->len[i]; b->self_read.size = 0;
|
||||||
lchain_align(&b->olist, s->uu, &b->self_read, &b->correct, &b->ovlp_read, &b->POA_Graph, &b->DAGCon,
|
lchain_align(&b->olist, s->uu, &b->self_read, &b->correct, &b->ovlp_read, &b->POA_Graph, &b->DAGCon,
|
||||||
&b->cigar1, &b->hap, &b->round2, &b->r_buf, &(b->tmp_region.w_list), 0, 1, &fully_cov, &abnormal, s->opt->diff_ec_ul, winLen, NULL);
|
&b->cigar1, &b->hap, &b->round2, &b->r_buf, &(b->tmp_region.w_list), 0, 1, &fully_cov, &abnormal, s->opt->diff_ec_ul, winLen, NULL);
|
||||||
|
|
||||||
// gl_chain_refine(&b->olist, &b->correct, &b->hap, bl, s->uu, s->opt->diff_ec_ul, winLen, s->len[i], km);
|
// gl_chain_refine(&b->olist, &b->correct, &b->hap, bl, s->uu, s->opt->diff_ec_ul, winLen, s->len[i], km);
|
||||||
gl_chain_refine_advance_combine(s->buf[tid], &(UL_INF.a[s->id+i]), &b->olist, &b->correct, &b->hap, &(s->sps[tid]), bl, &(s->gdp[tid]), s->uu, s->opt->diff_ec_ul, winLen, s->len[i], s->uopt, s->id+i, tid, NULL);
|
gl_chain_refine_advance_combine(s->buf[tid], &(UL_INF.a[s->id+i]), &b->olist, &b->correct, &b->hap, &(s->sps[tid]), bl, &(s->gdp[tid]), s->uu, s->opt->diff_ec_ul, winLen, s->len[i], s->uopt, s->id+i, tid, NULL);
|
||||||
|
|
||||||
memset(&b->self_read, 0, sizeof(b->self_read));
|
memset(&b->self_read, 0, sizeof(b->self_read));
|
||||||
if(UL_INF.a[s->id+i].dd) {
|
if(UL_INF.a[s->id+i].dd) {
|
||||||
free(s->seq[i]); s->seq[i] = NULL; b->num_correct_base++;
|
free(s->seq[i]); s->seq[i] = NULL; b->num_correct_base++;
|
||||||
}
|
}
|
||||||
s->hab[tid]->num_read_base++;
|
s->hab[tid]->num_read_base++;
|
||||||
}
|
}
|
||||||
|
|
||||||
static void *worker_gmap_work_ovec_pip(void *data, int step, void *in) // callback for kt_pipeline()
|
static void *worker_gmap_work_ovec_pip(void *data, int step, void *in) // callback for kt_pipeline()
|
||||||
{
|
{
|
||||||
gmap_t *p = (gmap_t*)data;
|
gmap_t *p = (gmap_t*)data;
|
||||||
if (step == 0) { // step 1: read a block of sequences
|
if (step == 0) { // step 1: read a block of sequences
|
||||||
int32_t ret; uint64_t l; sstep_t *s; CALLOC(s, 1);
|
int32_t ret; uint64_t l; sstep_t *s; CALLOC(s, 1);
|
||||||
s->ha_flt_tab = p->ha_flt_tab; s->ha_idx = p->ha_idx; s->id = p->total_pair;
|
s->ha_flt_tab = p->ha_flt_tab; s->ha_idx = p->ha_idx; s->id = p->total_pair;
|
||||||
s->opt = p->opt; s->uu = p->uu; s->uopt = p->uopt; s->rg = p->rg; s->mini_cut = p->mini_cut;///need set
|
s->opt = p->opt; s->uu = p->uu; s->uopt = p->uopt; s->rg = p->rg; s->mini_cut = p->mini_cut;///need set
|
||||||
while ((ret = kseq_read(p->ks)) >= 0) {
|
while ((ret = kseq_read(p->ks)) >= 0) {
|
||||||
if (p->ks->seq.l < (uint64_t)p->opt->k) continue;
|
if (p->ks->seq.l < (uint64_t)p->opt->k) continue;
|
||||||
if (s->n == s->m) {
|
if (s->n == s->m) {
|
||||||
s->m = s->m < 16? 16 : s->m + (s->n>>1);
|
s->m = s->m < 16? 16 : s->m + (s->n>>1);
|
||||||
REALLOC(s->len, s->m);
|
REALLOC(s->len, s->m);
|
||||||
REALLOC(s->seq, s->m);
|
REALLOC(s->seq, s->m);
|
||||||
}
|
}
|
||||||
if(!(p->remap)) {
|
if(!(p->remap)) {
|
||||||
append_ul_t(&UL_INF, NULL, p->ks->name.s, p->ks->name.l, NULL, 0, NULL, 0, P_CHAIN_COV, s->uopt, 0);
|
append_ul_t(&UL_INF, NULL, p->ks->name.s, p->ks->name.l, NULL, 0, NULL, 0, P_CHAIN_COV, s->uopt, 0);
|
||||||
}
|
}
|
||||||
l = p->ks->seq.l;
|
l = p->ks->seq.l;
|
||||||
MALLOC(s->seq[s->n], l);
|
MALLOC(s->seq[s->n], l);
|
||||||
s->sum_len += l;
|
s->sum_len += l;
|
||||||
memcpy(s->seq[s->n], p->ks->seq.s, l);
|
memcpy(s->seq[s->n], p->ks->seq.s, l);
|
||||||
s->len[s->n++] = l;
|
s->len[s->n++] = l;
|
||||||
if (s->sum_len >= p->chunk_size) break;
|
if (s->sum_len >= p->chunk_size) break;
|
||||||
}
|
}
|
||||||
p->total_pair += s->n;
|
p->total_pair += s->n;
|
||||||
if (s->sum_len == 0) free(s);
|
if (s->sum_len == 0) free(s);
|
||||||
else return s;
|
else return s;
|
||||||
}
|
}
|
||||||
else if (step == 1) { // step 2: alignment
|
else if (step == 1) { // step 2: alignment
|
||||||
sstep_t *s = (sstep_t*)in; uint64_t i;
|
sstep_t *s = (sstep_t*)in; uint64_t i;
|
||||||
CALLOC(s->hab, p->n_thread);
|
CALLOC(s->hab, p->n_thread);
|
||||||
CALLOC(s->buf, p->n_thread);
|
CALLOC(s->buf, p->n_thread);
|
||||||
if(!(p->remap)) CALLOC(s->res, p->n_thread);//for results
|
if(!(p->remap)) CALLOC(s->res, p->n_thread);//for results
|
||||||
for (i = 0; i < p->n_thread; ++i) {
|
for (i = 0; i < p->n_thread; ++i) {
|
||||||
s->hab[i] = ha_ovec_init(0, 0, 1); s->buf[i] = mg_tbuf_init();
|
s->hab[i] = ha_ovec_init(0, 0, 1); s->buf[i] = mg_tbuf_init();
|
||||||
}
|
}
|
||||||
// kt_for(p->n_thread, worker_for_ul_scall_alignment, s, s->n);
|
// kt_for(p->n_thread, worker_for_ul_scall_alignment, s, s->n);
|
||||||
|
|
||||||
for (i = 0; i < p->n_thread; ++i) {
|
for (i = 0; i < p->n_thread; ++i) {
|
||||||
s->num_bases += s->hab[i]->num_read_base;
|
s->num_bases += s->hab[i]->num_read_base;
|
||||||
s->num_corrected_bases += s->hab[i]->num_correct_base;
|
s->num_corrected_bases += s->hab[i]->num_correct_base;
|
||||||
s->num_recorrected_bases += s->hab[i]->num_recorrect_base;
|
s->num_recorrected_bases += s->hab[i]->num_recorrect_base;
|
||||||
ha_ovec_destroy(s->hab[i]); mg_tbuf_destroy(s->buf[i]);
|
ha_ovec_destroy(s->hab[i]); mg_tbuf_destroy(s->buf[i]);
|
||||||
}
|
}
|
||||||
free(s->hab); free(s->buf);
|
free(s->hab); free(s->buf);
|
||||||
return s;
|
return s;
|
||||||
}
|
}
|
||||||
else if (step == 2) { // step 3: dump
|
else if (step == 2) { // step 3: dump
|
||||||
sstep_t *s = (sstep_t*)in;
|
sstep_t *s = (sstep_t*)in;
|
||||||
uint64_t i, rid, sn = s->n;
|
uint64_t i, rid, sn = s->n;
|
||||||
p->num_bases += s->num_bases;
|
p->num_bases += s->num_bases;
|
||||||
p->num_corrected_bases += s->num_corrected_bases;
|
p->num_corrected_bases += s->num_corrected_bases;
|
||||||
p->num_recorrected_bases += s->num_recorrected_bases;
|
p->num_recorrected_bases += s->num_recorrected_bases;
|
||||||
if(!(p->remap)) {
|
if(!(p->remap)) {
|
||||||
for (i = 0; i < p->n_thread; ++i) {
|
for (i = 0; i < p->n_thread; ++i) {
|
||||||
push_uc_block_t(s->uopt, &(s->res[i]), s->seq, s->len, s->id);
|
push_uc_block_t(s->uopt, &(s->res[i]), s->seq, s->len, s->id);
|
||||||
kv_destroy(s->res[i]);
|
kv_destroy(s->res[i]);
|
||||||
}
|
}
|
||||||
free(s->res);
|
free(s->res);
|
||||||
|
|
||||||
for (i = 0; i < sn; ++i) {
|
for (i = 0; i < sn; ++i) {
|
||||||
rid = s->id + i;
|
rid = s->id + i;
|
||||||
if((UL_INF.n <= rid) || (UL_INF.n > rid && UL_INF.a[rid].rlen != s->len[i])) {///reads without alignment
|
if((UL_INF.n <= rid) || (UL_INF.n > rid && UL_INF.a[rid].rlen != s->len[i])) {///reads without alignment
|
||||||
append_ul_t(&UL_INF, &rid, NULL, 0, s->seq[i], s->len[i], NULL, 0, P_CHAIN_COV, s->uopt, 0);
|
append_ul_t(&UL_INF, &rid, NULL, 0, s->seq[i], s->len[i], NULL, 0, P_CHAIN_COV, s->uopt, 0);
|
||||||
}
|
}
|
||||||
free(s->seq[i]);
|
free(s->seq[i]);
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
for (i = 0; i < sn; ++i) {
|
for (i = 0; i < sn; ++i) {
|
||||||
rid = s->id + i;
|
rid = s->id + i;
|
||||||
if(UL_INF.a[rid].dd == 0 && p->ucr_s && p->ucr_s->flag == 1) {
|
if(UL_INF.a[rid].dd == 0 && p->ucr_s && p->ucr_s->flag == 1) {
|
||||||
assert(s->seq[i]);
|
assert(s->seq[i]);
|
||||||
///for debug interval
|
///for debug interval
|
||||||
write_compress_base_disk(p->ucr_s->fp, rid, s->seq[i], s->len[i], &(p->ucr_s->u));
|
write_compress_base_disk(p->ucr_s->fp, rid, s->seq[i], s->len[i], &(p->ucr_s->u));
|
||||||
}
|
}
|
||||||
free(s->seq[i]);
|
free(s->seq[i]);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
free(s->len); free(s->seq); free(s);
|
free(s->len); free(s->seq); free(s);
|
||||||
}
|
}
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
**/
|
**/
|
||||||
|
|
||||||
// int gmap_work_ovec(gmap_t* sl, const enzyme *fn)
|
// int gmap_work_ovec(gmap_t* sl, const enzyme *fn)
|
||||||
// {
|
// {
|
||||||
// double index_time = yak_realtime();
|
// double index_time = yak_realtime();
|
||||||
// int i;
|
// int i;
|
||||||
|
|
||||||
// init_all_ul_t(&UL_INF, &R_INF);
|
// init_all_ul_t(&UL_INF, &R_INF);
|
||||||
// for (i = 0; i < fn->n; i++){
|
// for (i = 0; i < fn->n; i++){
|
||||||
// gzFile fp;
|
// gzFile fp;
|
||||||
// if ((fp = gzopen(fn->a[i], "r")) == 0) return 0;
|
// if ((fp = gzopen(fn->a[i], "r")) == 0) return 0;
|
||||||
// sl->ks = kseq_init(fp);
|
// sl->ks = kseq_init(fp);
|
||||||
// kt_pipeline(3, worker_gmap_work_ovec_pip, sl, 3);
|
// kt_pipeline(3, worker_gmap_work_ovec_pip, sl, 3);
|
||||||
// kseq_destroy(sl->ks);
|
// kseq_destroy(sl->ks);
|
||||||
// gzclose(fp);
|
// gzclose(fp);
|
||||||
// }
|
// }
|
||||||
// sl->hits.total_base = sl->total_base;
|
// sl->hits.total_base = sl->total_base;
|
||||||
// sl->hits.total_pair = sl->total_pair;
|
// sl->hits.total_pair = sl->total_pair;
|
||||||
// fprintf(stderr, "[M::%s::%.3f] ==> Qualification\n", __func__, yak_realtime()-index_time);
|
// fprintf(stderr, "[M::%s::%.3f] ==> Qualification\n", __func__, yak_realtime()-index_time);
|
||||||
// fprintf(stderr, "[M::%s::] ==> # reads: %lu, # bases: %lu\n", __func__, UL_INF.n, sl->total_base);
|
// fprintf(stderr, "[M::%s::] ==> # reads: %lu, # bases: %lu\n", __func__, UL_INF.n, sl->total_base);
|
||||||
// fprintf(stderr, "[M::%s::] ==> # bases: %lu; # corrected bases: %lu; # recorrected bases: %lu\n",
|
// fprintf(stderr, "[M::%s::] ==> # bases: %lu; # corrected bases: %lu; # recorrected bases: %lu\n",
|
||||||
// __func__, sl->num_bases, sl->num_corrected_bases, sl->num_recorrected_bases);
|
// __func__, sl->num_bases, sl->num_corrected_bases, sl->num_recorrected_bases);
|
||||||
// gen_ul_vec_rid_t(&UL_INF, &R_INF, NULL);
|
// gen_ul_vec_rid_t(&UL_INF, &R_INF, NULL);
|
||||||
// return 1;
|
// return 1;
|
||||||
// }
|
// }
|
||||||
+6
-6
@@ -1,6 +1,6 @@
|
|||||||
#ifndef __INTER__
|
#ifndef __INTER__
|
||||||
#define __INTER__
|
#define __INTER__
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
#include "Process_Read.h"
|
#include "Process_Read.h"
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
+18238
-18220
File diff suppressed because it is too large
Load Diff
@@ -1,45 +1,45 @@
|
|||||||
#ifndef __GFA_UT__
|
#ifndef __GFA_UT__
|
||||||
#define __GFA_UT__
|
#define __GFA_UT__
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
#include "hic.h"
|
#include "hic.h"
|
||||||
|
|
||||||
#define is_contain_r(ri, z) (((z)<(ri).len)&&((ri).index[(z)]!=(uint32_t)(-1))&&(!((ri).index[(z)]>>31)))
|
#define is_contain_r(ri, z) (((z)<(ri).len)&&((ri).index[(z)]!=(uint32_t)(-1))&&(!((ri).index[(z)]>>31)))
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
asg_t *g;
|
asg_t *g;
|
||||||
ma_hit_t_alloc *src;
|
ma_hit_t_alloc *src;
|
||||||
ma_sub_t *cov;
|
ma_sub_t *cov;
|
||||||
R_to_U* ruIndex;
|
R_to_U* ruIndex;
|
||||||
int64_t max_hang;
|
int64_t max_hang;
|
||||||
int64_t min_ovlp;
|
int64_t min_ovlp;
|
||||||
int64_t ul_occ;
|
int64_t ul_occ;
|
||||||
} sset_aux;
|
} sset_aux;
|
||||||
|
|
||||||
void ul_clean_gfa(ug_opt_t *uopt, asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, int64_t clean_round, double min_ovlp_drop_ratio, double max_ovlp_drop_ratio,
|
void ul_clean_gfa(ug_opt_t *uopt, asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, int64_t clean_round, double min_ovlp_drop_ratio, double max_ovlp_drop_ratio,
|
||||||
double ou_drop_rate, int64_t max_tip, int64_t gap_fuzz, bub_label_t *b_mask_t, int32_t is_ou, int32_t is_trio, uint32_t ou_thres, uint8_t *cmk, char *o_file);
|
double ou_drop_rate, int64_t max_tip, int64_t gap_fuzz, bub_label_t *b_mask_t, int32_t is_ou, int32_t is_trio, uint32_t ou_thres, uint8_t *cmk, char *o_file);
|
||||||
uint32_t asg_arc_cut_tips(asg_t *g, uint32_t max_ext, asg64_v *in, uint32_t is_ou, R_to_U *ru, telo_end_t *te);
|
uint32_t asg_arc_cut_tips(asg_t *g, uint32_t max_ext, asg64_v *in, uint32_t is_ou, R_to_U *ru, telo_end_t *te);
|
||||||
void asg_iterative_semi_circ(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t normal_len, uint32_t pop_chimer, asg64_v *dbg, telo_end_t *te);
|
void asg_iterative_semi_circ(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t normal_len, uint32_t pop_chimer, asg64_v *dbg, telo_end_t *te);
|
||||||
void asg_arc_cut_chimeric(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t ou_thres, telo_end_t *te);
|
void asg_arc_cut_chimeric(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, uint32_t ou_thres, telo_end_t *te);
|
||||||
void asg_arc_cut_inexact(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, int32_t max_ext, uint32_t is_ou, uint32_t is_trio, uint32_t min_diff, float ou_rat/**, asg64_v *dbg**/);
|
void asg_arc_cut_inexact(asg_t *g, ma_hit_t_alloc* src, asg64_v *in, int32_t max_ext, uint32_t is_ou, uint32_t is_trio, uint32_t min_diff, float ou_rat/**, asg64_v *dbg**/);
|
||||||
void asg_arc_cut_length(asg_t *g, asg64_v *in, int32_t max_ext, float len_rat, float ou_rat, uint32_t is_ou, uint32_t is_trio,
|
void asg_arc_cut_length(asg_t *g, asg64_v *in, int32_t max_ext, float len_rat, float ou_rat, uint32_t is_ou, uint32_t is_trio,
|
||||||
uint32_t is_topo, uint32_t min_diff, uint32_t min_ou, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len);
|
uint32_t is_topo, uint32_t min_diff, uint32_t min_ou, ma_hit_t_alloc *rev, R_to_U* rI, uint32_t *max_drop_len);
|
||||||
void asg_arc_cut_bub_links(asg_t *g, asg64_v *in, float len_rat, float sec_len_rat, float ou_rat, uint32_t is_ou, uint64_t check_dist, ma_hit_t_alloc *rev, R_to_U* rI, int32_t max_ext);
|
void asg_arc_cut_bub_links(asg_t *g, asg64_v *in, float len_rat, float sec_len_rat, float ou_rat, uint32_t is_ou, uint64_t check_dist, ma_hit_t_alloc *rev, R_to_U* rI, int32_t max_ext);
|
||||||
void asg_arc_cut_complex_bub_links(asg_t *g, asg64_v *in, float len_rat, float ou_rat, uint32_t is_ou, bub_label_t *b_mask_t);
|
void asg_arc_cut_complex_bub_links(asg_t *g, asg64_v *in, float len_rat, float ou_rat, uint32_t is_ou, bub_label_t *b_mask_t);
|
||||||
uint32_t asg_cut_large_indel(asg_t *g, asg64_v *in, int32_t max_ext, float ou_rat, uint32_t is_ou, uint32_t min_diff);
|
uint32_t asg_cut_large_indel(asg_t *g, asg64_v *in, int32_t max_ext, float ou_rat, uint32_t is_ou, uint32_t min_diff);
|
||||||
uint32_t asg_cut_semi_circ(asg_t *g, uint32_t lim_len, uint32_t is_clean);
|
uint32_t asg_cut_semi_circ(asg_t *g, uint32_t lim_len, uint32_t is_clean);
|
||||||
void ul_realignment_gfa(ug_opt_t *uopt, asg_t *sg, int64_t clean_round, double min_ovlp_drop_ratio,
|
void ul_realignment_gfa(ug_opt_t *uopt, asg_t *sg, int64_t clean_round, double min_ovlp_drop_ratio,
|
||||||
double max_ovlp_drop_ratio, int64_t max_tip, int64_t max_ul_tip, bub_label_t *b_mask_t, uint32_t is_trio, char *o_file, ul_renew_t *ropt, const char *bin_file, uint64_t free_uld, uint64_t is_bridg, uint64_t deep_clean);
|
double max_ovlp_drop_ratio, int64_t max_tip, int64_t max_ul_tip, bub_label_t *b_mask_t, uint32_t is_trio, char *o_file, ul_renew_t *ropt, const char *bin_file, uint64_t free_uld, uint64_t is_bridg, uint64_t deep_clean);
|
||||||
void recover_contain_g(asg_t *g, ma_hit_t_alloc *src, R_to_U* ruIndex, int64_t max_hang, int64_t min_ovlp, int64_t ul_occ);
|
void recover_contain_g(asg_t *g, ma_hit_t_alloc *src, R_to_U* ruIndex, int64_t max_hang, int64_t min_ovlp, int64_t ul_occ);
|
||||||
void normalize_gou(asg_t *g);
|
void normalize_gou(asg_t *g);
|
||||||
void prt_specfic_sge(asg_t *g, uint32_t src, uint32_t dst, const char* cmd);
|
void prt_specfic_sge(asg_t *g, uint32_t src, uint32_t dst, const char* cmd);
|
||||||
asg_t *gen_ng(ma_ug_t *ug, asg_t *sg, ug_opt_t *uopt, ma_sub_t **cov, R_to_U *ruI, uint64_t scaffold_len);
|
asg_t *gen_ng(ma_ug_t *ug, asg_t *sg, ug_opt_t *uopt, ma_sub_t **cov, R_to_U *ruI, uint64_t scaffold_len);
|
||||||
void post_rescue(ug_opt_t *uopt, asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, bub_label_t *b_mask_t, long long no_trio_recover, uint8_t *cmk);
|
void post_rescue(ug_opt_t *uopt, asg_t *sg, ma_hit_t_alloc *src, ma_hit_t_alloc *rev, R_to_U* rI, bub_label_t *b_mask_t, long long no_trio_recover, uint8_t *cmk);
|
||||||
// void print_raw_u2rgfa_seq(all_ul_t *aln, R_to_U* rI, uint32_t is_detail);
|
// void print_raw_u2rgfa_seq(all_ul_t *aln, R_to_U* rI, uint32_t is_detail);
|
||||||
bubble_type *gen_bubble_chain(asg_t *sg, ma_ug_t *ug, ug_opt_t *uopt, uint8_t **ir_het, uint8_t avoid_het);
|
bubble_type *gen_bubble_chain(asg_t *sg, ma_ug_t *ug, ug_opt_t *uopt, uint8_t **ir_het, uint8_t avoid_het);
|
||||||
void filter_sg_by_ug(asg_t *rg, ma_ug_t *ug, ug_opt_t *uopt);
|
void filter_sg_by_ug(asg_t *rg, ma_ug_t *ug, ug_opt_t *uopt);
|
||||||
void ug_ext_gfa(ug_opt_t *uopt, asg_t *sg, uint32_t max_len);
|
void ug_ext_gfa(ug_opt_t *uopt, asg_t *sg, uint32_t max_len);
|
||||||
void update_sg_uo(asg_t *g, ma_hit_t_alloc *src);
|
void update_sg_uo(asg_t *g, ma_hit_t_alloc *src);
|
||||||
uint32_t get_arcs(asg_t *g, uint32_t v, uint32_t* idx, uint32_t idx_n);
|
uint32_t get_arcs(asg_t *g, uint32_t v, uint32_t* idx, uint32_t idx_n);
|
||||||
uint64_t ug_occ_w(uint64_t is, uint64_t ie, ma_utg_t *u);
|
uint64_t ug_occ_w(uint64_t is, uint64_t ie, ma_utg_t *u);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,122 +1,122 @@
|
|||||||
#ifndef __HIC__
|
#ifndef __HIC__
|
||||||
#define __HIC__
|
#define __HIC__
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
|
|
||||||
#define kdq_clear(q) ((q)->count = (q)->front = 0)
|
#define kdq_clear(q) ((q)->count = (q)->front = 0)
|
||||||
#define kv_malloc(v, s) ((v).n = 0, (v).m = (s), MALLOC((v).a, (s)))
|
#define kv_malloc(v, s) ((v).n = 0, (v).m = (s), MALLOC((v).a, (s)))
|
||||||
#define RC_0 0
|
#define RC_0 0
|
||||||
#define RC_1 1
|
#define RC_1 1
|
||||||
#define RC_2 2
|
#define RC_2 2
|
||||||
#define RC_3 3
|
#define RC_3 3
|
||||||
|
|
||||||
hc_edge* get_hc_edge(hc_links* link, uint64_t src, uint64_t dest, uint64_t dir);
|
hc_edge* get_hc_edge(hc_links* link, uint64_t src, uint64_t dest, uint64_t dir);
|
||||||
hc_edge* push_hc_edge(hc_linkeage* x, uint64_t uID, double weight, int dir, uint64_t* d);
|
hc_edge* push_hc_edge(hc_linkeage* x, uint64_t uID, double weight, int dir, uint64_t* d);
|
||||||
void hic_benchmark(ma_ug_t *ug, asg_t* read_g);
|
void hic_benchmark(ma_ug_t *ug, asg_t* read_g);
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
double w;
|
double w;
|
||||||
uint32_t id, occ;
|
uint32_t id, occ;
|
||||||
///uint32_t *bid, bid_n;
|
///uint32_t *bid, bid_n;
|
||||||
ma_utg_t *u;
|
ma_utg_t *u;
|
||||||
uint64_t l_d, r_d;
|
uint64_t l_d, r_d;
|
||||||
}chain_hic_w_type;
|
}chain_hic_w_type;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
size_t n, m;
|
size_t n, m;
|
||||||
chain_hic_w_type* a;
|
chain_hic_w_type* a;
|
||||||
uint32_t max_bub_id;
|
uint32_t max_bub_id;
|
||||||
uint32_t *chain_idx, u_n;
|
uint32_t *chain_idx, u_n;
|
||||||
}chain_hic_warp;
|
}chain_hic_warp;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
long long g_occ, b_occ;
|
long long g_occ, b_occ;
|
||||||
uint64_t id;
|
uint64_t id;
|
||||||
uint8_t del;
|
uint8_t del;
|
||||||
}chain_w_type;
|
}chain_w_type;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint32_t* index, round_id, n_round;
|
uint32_t* index, round_id, n_round;
|
||||||
ma_ug_t* ug;
|
ma_ug_t* ug;
|
||||||
kvec_t(uint32_t) list;
|
kvec_t(uint32_t) list;
|
||||||
kvec_t(uint32_t) num;
|
kvec_t(uint32_t) num;
|
||||||
kvec_t(uint64_t) pathLen;
|
kvec_t(uint64_t) pathLen;
|
||||||
kvec_t(uint64_t) b_s_idx;
|
kvec_t(uint64_t) b_s_idx;
|
||||||
uint64_t s_bub, f_bub, b_bub, b_end_bub, tangle_bub, cross_bub, mess_bub;
|
uint64_t s_bub, f_bub, b_bub, b_end_bub, tangle_bub, cross_bub, mess_bub;
|
||||||
uint32_t check_het;
|
uint32_t check_het;
|
||||||
asg_t *b_g;
|
asg_t *b_g;
|
||||||
ma_ug_t* b_ug;
|
ma_ug_t* b_ug;
|
||||||
kvec_t(chain_w_type) chain_weight;
|
kvec_t(chain_w_type) chain_weight;
|
||||||
chain_hic_warp c_w;
|
chain_hic_warp c_w;
|
||||||
} bubble_type;
|
} bubble_type;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int8_t *s;
|
int8_t *s;
|
||||||
uint64_t xs;
|
uint64_t xs;
|
||||||
uint64_t n;
|
uint64_t n;
|
||||||
} ps_t;
|
} ps_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint64_t s, e, id, len;
|
uint64_t s, e, id, len;
|
||||||
} pe_hit;
|
} pe_hit;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
kvec_t(pe_hit) a;
|
kvec_t(pe_hit) a;
|
||||||
kvec_t(uint64_t) idx;
|
kvec_t(uint64_t) idx;
|
||||||
kvec_t(uint64_t) occ;
|
kvec_t(uint64_t) occ;
|
||||||
uint64_t uID_bits;
|
uint64_t uID_bits;
|
||||||
uint64_t pos_mode;
|
uint64_t pos_mode;
|
||||||
} kvec_pe_hit;
|
} kvec_pe_hit;
|
||||||
|
|
||||||
typedef struct{
|
typedef struct{
|
||||||
kvec_t(uint8_t) vis;
|
kvec_t(uint8_t) vis;
|
||||||
kvec_t(uint64_t) x;
|
kvec_t(uint64_t) x;
|
||||||
kvec_t(uint64_t) dis;
|
kvec_t(uint64_t) dis;
|
||||||
uint64_t uID_mode, uID_shift, tmp_v, tmp_d;
|
uint64_t uID_mode, uID_shift, tmp_v, tmp_d;
|
||||||
}pdq;
|
}pdq;
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
#define P_het(B) ((B).num.n)
|
#define P_het(B) ((B).num.n)
|
||||||
#define M_het(B) ((B).num.n + 1)
|
#define M_het(B) ((B).num.n + 1)
|
||||||
// #define IF_BUB(ID, B) ((B).index[(ID)] < (B).num.n)
|
// #define IF_BUB(ID, B) ((B).index[(ID)] < (B).num.n)
|
||||||
// #define IF_HET(ID, B) ((B).index[(ID)] == (B).num.n)
|
// #define IF_HET(ID, B) ((B).index[(ID)] == (B).num.n)
|
||||||
// #define IF_HOM(ID, B) ((B).index[(ID)] > (B).num.n)
|
// #define IF_HOM(ID, B) ((B).index[(ID)] > (B).num.n)
|
||||||
#define IF_BUB(ID, B) ((B).index[(ID)] < (B).f_bub+1)
|
#define IF_BUB(ID, B) ((B).index[(ID)] < (B).f_bub+1)
|
||||||
#define IF_HET(ID, B) ((B).index[(ID)] == (B).f_bub+1)
|
#define IF_HET(ID, B) ((B).index[(ID)] == (B).f_bub+1)
|
||||||
#define IF_HOM(ID, B) ((B).index[(ID)] > (B).f_bub+1)
|
#define IF_HOM(ID, B) ((B).index[(ID)] > (B).f_bub+1)
|
||||||
#define Get_bub_num(RECORD) ((RECORD).num.n-1)
|
#define Get_bub_num(RECORD) ((RECORD).num.n-1)
|
||||||
void get_bubbles(bubble_type* bub, uint64_t id, uint32_t* beg, uint32_t* sink, uint32_t** a, uint32_t* n, uint64_t* pathBase);
|
void get_bubbles(bubble_type* bub, uint64_t id, uint32_t* beg, uint32_t* sink, uint32_t** a, uint32_t* n, uint64_t* pathBase);
|
||||||
int load_hc_links(hc_links* link, const char *fn);
|
int load_hc_links(hc_links* link, const char *fn);
|
||||||
void write_hc_links(hc_links* link, const char *fn);
|
void write_hc_links(hc_links* link, const char *fn);
|
||||||
void destory_bubbles(bubble_type* bub);
|
void destory_bubbles(bubble_type* bub);
|
||||||
void identify_bubbles(ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, kv_u_trans_t *ref);
|
void identify_bubbles(ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, kv_u_trans_t *ref);
|
||||||
void identify_bubbles_recal(asg_t* sg, ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, ma_hit_t_alloc* sources, R_to_U* ruIndex,
|
void identify_bubbles_recal(asg_t* sg, ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, ma_hit_t_alloc* sources, R_to_U* ruIndex,
|
||||||
kv_u_trans_t *ref);
|
kv_u_trans_t *ref);
|
||||||
void identify_bubbles_recal_poy(asg_t* sg, ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, ma_hit_t_alloc* sources, R_to_U* ruIndex,
|
void identify_bubbles_recal_poy(asg_t* sg, ma_ug_t* ug, bubble_type* bub, uint8_t *r_het_flag, ma_hit_t_alloc* sources, R_to_U* ruIndex,
|
||||||
kv_u_trans_t *ref);
|
kv_u_trans_t *ref);
|
||||||
void resolve_bubble_chain_tangle(ma_ug_t* ug, bubble_type* bub);
|
void resolve_bubble_chain_tangle(ma_ug_t* ug, bubble_type* bub);
|
||||||
uint32_t connect_bub_occ(bubble_type* bub, uint32_t root_id, uint32_t check_het);
|
uint32_t connect_bub_occ(bubble_type* bub, uint32_t root_id, uint32_t check_het);
|
||||||
void get_bub_id(bubble_type* bub, uint32_t root, uint64_t* id0, uint64_t* id1, uint32_t check_het);
|
void get_bub_id(bubble_type* bub, uint32_t root, uint64_t* id0, uint64_t* id1, uint32_t check_het);
|
||||||
void update_bubble_chain(ma_ug_t* ug, bubble_type* bub, uint32_t is_middle, uint32_t is_end);
|
void update_bubble_chain(ma_ug_t* ug, bubble_type* bub, uint32_t is_middle, uint32_t is_end);
|
||||||
void set_b_utg_weight_flag(bubble_type* bub, buf_t* b, uint32_t v, uint8_t* vis_flag, uint32_t flag, uint32_t* occ);
|
void set_b_utg_weight_flag(bubble_type* bub, buf_t* b, uint32_t v, uint8_t* vis_flag, uint32_t flag, uint32_t* occ);
|
||||||
void debug_gfa_space(ma_ug_t* ug, hap_cov_t *cov);
|
void debug_gfa_space(ma_ug_t* ug, hap_cov_t *cov);
|
||||||
void init_ug_idx(ma_ug_t *ug, uint64_t k, uint64_t up_bound, uint64_t low_bound, uint64_t build_idx);
|
void init_ug_idx(ma_ug_t *ug, uint64_t k, uint64_t up_bound, uint64_t low_bound, uint64_t build_idx);
|
||||||
void des_ug_idx();
|
void des_ug_idx();
|
||||||
uint64_t count_unique_k_mers(char *r, uint64_t len, uint64_t query, uint64_t target, uint64_t *all, uint64_t *found);
|
uint64_t count_unique_k_mers(char *r, uint64_t len, uint64_t query, uint64_t target, uint64_t *all, uint64_t *found);
|
||||||
void init_pdq(pdq* q, uint64_t utg_num);
|
void init_pdq(pdq* q, uint64_t utg_num);
|
||||||
void destory_pdq(pdq* q);
|
void destory_pdq(pdq* q);
|
||||||
uint32_t check_trans_relation_by_path(uint32_t v, uint32_t w, pdq* pqv, uint32_t* path_v, buf_t *resv,
|
uint32_t check_trans_relation_by_path(uint32_t v, uint32_t w, pdq* pqv, uint32_t* path_v, buf_t *resv,
|
||||||
pdq* pqw, uint32_t* path_w, buf_t *resw, asg_t *sg, uint8_t *dest, uint8_t df, uint32_t df_occ, double rate,
|
pdq* pqw, uint32_t* path_w, buf_t *resw, asg_t *sg, uint8_t *dest, uint8_t df, uint32_t df_occ, double rate,
|
||||||
long long *dis);
|
long long *dis);
|
||||||
void set_utg_by_dis(uint32_t v, pdq* pq, asg_t *g, kvec_t_u32_warp *res, uint32_t dis);
|
void set_utg_by_dis(uint32_t v, pdq* pq, asg_t *g, kvec_t_u32_warp *res, uint32_t dis);
|
||||||
void dedup_hits(kvec_pe_hit* hits, uint64_t is_dup);
|
void dedup_hits(kvec_pe_hit* hits, uint64_t is_dup);
|
||||||
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, mmhap_t **rh, kvec_pe_hit **rhits);
|
void hic_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, mmhap_t **rh, kvec_pe_hit **rhits);
|
||||||
spg_t *hic_pre_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, kvec_pe_hit **rhits);
|
spg_t *hic_pre_analysis(ma_ug_t *ug, asg_t* read_g, trans_chain* t_ch, ug_opt_t *opt, kvec_pe_hit **rhits);
|
||||||
void prt_bubble_gfa_adv(FILE *fp, bubble_type *bub, const char* utg_pre, const char* bub_pre, const char* chain_pre);
|
void prt_bubble_gfa_adv(FILE *fp, bubble_type *bub, const char* utg_pre, const char* bub_pre, const char* chain_pre);
|
||||||
void bp_solve(ug_opt_t *opt, kv_u_trans_t *ref, ma_ug_t *ug, asg_t *sg, bubble_type *bub, double cis_rate);
|
void bp_solve(ug_opt_t *opt, kv_u_trans_t *ref, ma_ug_t *ug, asg_t *sg, bubble_type *bub, double cis_rate);
|
||||||
void trio_phasing_refine(ma_ug_t *ug, asg_t* sg, kv_u_trans_t *ta, ug_opt_t *opt);
|
void trio_phasing_refine(ma_ug_t *ug, asg_t* sg, kv_u_trans_t *ta, ug_opt_t *opt);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,157 +1,157 @@
|
|||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <assert.h>
|
#include <assert.h>
|
||||||
#include "htab.h"
|
#include "htab.h"
|
||||||
|
|
||||||
static void ha_hist_line(int c, int x, int exceed, int64_t cnt)
|
static void ha_hist_line(int c, int x, int exceed, int64_t cnt)
|
||||||
{
|
{
|
||||||
int j;
|
int j;
|
||||||
if (c >= 0) fprintf(stderr, "[M::%s] %5d: ", __func__, c);
|
if (c >= 0) fprintf(stderr, "[M::%s] %5d: ", __func__, c);
|
||||||
else fprintf(stderr, "[M::%s] %5s: ", __func__, "rest");
|
else fprintf(stderr, "[M::%s] %5s: ", __func__, "rest");
|
||||||
for (j = 0; j < x; ++j) fputc('*', stderr);
|
for (j = 0; j < x; ++j) fputc('*', stderr);
|
||||||
if (exceed) fputc('>', stderr);
|
if (exceed) fputc('>', stderr);
|
||||||
fprintf(stderr, " %lld\n", (long long)cnt);
|
fprintf(stderr, " %lld\n", (long long)cnt);
|
||||||
}
|
}
|
||||||
|
|
||||||
void print_hist_lines(int n_cnt, int start_cnt, const int64_t *cnt)
|
void print_hist_lines(int n_cnt, int start_cnt, const int64_t *cnt)
|
||||||
{
|
{
|
||||||
const int hist_max = 100;
|
const int hist_max = 100;
|
||||||
int i, start, low_i, max_i, max;
|
int i, start, low_i, max_i, max;
|
||||||
// determine the start point
|
// determine the start point
|
||||||
assert(n_cnt > start_cnt);
|
assert(n_cnt > start_cnt);
|
||||||
start = cnt[1] > 0? 1 : 2;
|
start = cnt[1] > 0? 1 : 2;
|
||||||
|
|
||||||
// find the low point from the left
|
// find the low point from the left
|
||||||
low_i = start > start_cnt? start : start_cnt;
|
low_i = start > start_cnt? start : start_cnt;
|
||||||
for (i = low_i; i < n_cnt; ++i)
|
for (i = low_i; i < n_cnt; ++i)
|
||||||
if (cnt[i] > cnt[i-1]) break;
|
if (cnt[i] > cnt[i-1]) break;
|
||||||
low_i = i - 1;
|
low_i = i - 1;
|
||||||
fprintf(stderr, "[M::%s] lowest: count[%d] = %ld\n", __func__, low_i, (long)cnt[low_i]);
|
fprintf(stderr, "[M::%s] lowest: count[%d] = %ld\n", __func__, low_i, (long)cnt[low_i]);
|
||||||
|
|
||||||
// find the highest peak
|
// find the highest peak
|
||||||
max_i = start > start_cnt? start : start_cnt, max = cnt[max_i];
|
max_i = start > start_cnt? start : start_cnt, max = cnt[max_i];
|
||||||
for (i = max_i; i < n_cnt; ++i)
|
for (i = max_i; i < n_cnt; ++i)
|
||||||
if (cnt[i] > max)
|
if (cnt[i] > max)
|
||||||
max = cnt[i], max_i = i;
|
max = cnt[i], max_i = i;
|
||||||
fprintf(stderr, "[M::%s] highest: count[%d] = %ld\n", __func__, max_i, (long)cnt[max_i]);
|
fprintf(stderr, "[M::%s] highest: count[%d] = %ld\n", __func__, max_i, (long)cnt[max_i]);
|
||||||
|
|
||||||
for (i = start; i < n_cnt; ++i) {
|
for (i = start; i < n_cnt; ++i) {
|
||||||
int x, exceed = 0;
|
int x, exceed = 0;
|
||||||
x = (int)((double)hist_max * cnt[i] / cnt[max_i] + .499);
|
x = (int)((double)hist_max * cnt[i] / cnt[max_i] + .499);
|
||||||
if (x > hist_max) exceed = 1, x = hist_max; // may happen if cnt[2] is higher
|
if (x > hist_max) exceed = 1, x = hist_max; // may happen if cnt[2] is higher
|
||||||
if (i > max_i && x == 0) break;
|
if (i > max_i && x == 0) break;
|
||||||
ha_hist_line(i, x, exceed, cnt[i]);
|
ha_hist_line(i, x, exceed, cnt[i]);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
int adj_m_peak_hom(int m_peak_hom, int max_i, int max2_i, int max3_i, int *peak_het)
|
int adj_m_peak_hom(int m_peak_hom, int max_i, int max2_i, int max3_i, int *peak_het)
|
||||||
{
|
{
|
||||||
int64_t mm[3], d, min_i, min_d, i;
|
int64_t mm[3], d, min_i, min_d, i;
|
||||||
mm[0] = max2_i; mm[1] = max_i; mm[2] = max3_i;
|
mm[0] = max2_i; mm[1] = max_i; mm[2] = max3_i;
|
||||||
for (i = 0, min_i = -1, min_d = -1; i < 3; i++){
|
for (i = 0, min_i = -1, min_d = -1; i < 3; i++){
|
||||||
if(mm[i] <= 0) continue;
|
if(mm[i] <= 0) continue;
|
||||||
d = (mm[i] >= m_peak_hom?mm[i]-m_peak_hom:m_peak_hom-mm[i]);
|
d = (mm[i] >= m_peak_hom?mm[i]-m_peak_hom:m_peak_hom-mm[i]);
|
||||||
if(min_d == -1 || min_d > d || (min_d == d && i == 1)){
|
if(min_d == -1 || min_d > d || (min_d == d && i == 1)){
|
||||||
min_d = d; min_i = i;
|
min_d = d; min_i = i;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if(min_i < 0) return m_peak_hom;
|
if(min_i < 0) return m_peak_hom;
|
||||||
if(mm[min_i] < m_peak_hom){
|
if(mm[min_i] < m_peak_hom){
|
||||||
d = m_peak_hom - mm[min_i];
|
d = m_peak_hom - mm[min_i];
|
||||||
if(d >= mm[min_i]*0.51) {
|
if(d >= mm[min_i]*0.51) {
|
||||||
*peak_het = mm[min_i];
|
*peak_het = mm[min_i];
|
||||||
return m_peak_hom;
|
return m_peak_hom;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
for (i = min_i-1; i >= 0; i--){
|
for (i = min_i-1; i >= 0; i--){
|
||||||
if(mm[i] <= 0) continue;
|
if(mm[i] <= 0) continue;
|
||||||
*peak_het = mm[i];
|
*peak_het = mm[i];
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
return mm[min_i];
|
return mm[min_i];
|
||||||
}
|
}
|
||||||
|
|
||||||
int ha_analyze_count(int n_cnt, int start_cnt, int m_peak_hom, const int64_t *cnt, int *peak_het)
|
int ha_analyze_count(int n_cnt, int start_cnt, int m_peak_hom, const int64_t *cnt, int *peak_het)
|
||||||
{
|
{
|
||||||
const int hist_max = 100;
|
const int hist_max = 100;
|
||||||
int i, start, low_i, max_i, max2_i, max3_i;
|
int i, start, low_i, max_i, max2_i, max3_i;
|
||||||
int64_t max, max2, max3, min;
|
int64_t max, max2, max3, min;
|
||||||
|
|
||||||
// determine the start point
|
// determine the start point
|
||||||
assert(n_cnt > start_cnt);
|
assert(n_cnt > start_cnt);
|
||||||
*peak_het = -1;
|
*peak_het = -1;
|
||||||
start = cnt[1] > 0? 1 : 2;
|
start = cnt[1] > 0? 1 : 2;
|
||||||
|
|
||||||
// find the low point from the left
|
// find the low point from the left
|
||||||
low_i = start > start_cnt? start : start_cnt;
|
low_i = start > start_cnt? start : start_cnt;
|
||||||
for (i = low_i + 1; i < n_cnt; ++i)
|
for (i = low_i + 1; i < n_cnt; ++i)
|
||||||
if (cnt[i] > cnt[i-1]) break;
|
if (cnt[i] > cnt[i-1]) break;
|
||||||
low_i = i - 1;
|
low_i = i - 1;
|
||||||
fprintf(stderr, "[M::%s] lowest: count[%d] = %ld\n", __func__, low_i, (long)cnt[low_i]);
|
fprintf(stderr, "[M::%s] lowest: count[%d] = %ld\n", __func__, low_i, (long)cnt[low_i]);
|
||||||
if (low_i == n_cnt - 1) return -1; // low coverage
|
if (low_i == n_cnt - 1) return -1; // low coverage
|
||||||
|
|
||||||
// find the highest peak
|
// find the highest peak
|
||||||
max_i = low_i + 1, max = cnt[max_i];
|
max_i = low_i + 1, max = cnt[max_i];
|
||||||
for (i = low_i + 1; i < n_cnt; ++i)
|
for (i = low_i + 1; i < n_cnt; ++i)
|
||||||
if (cnt[i] > max)
|
if (cnt[i] > max)
|
||||||
max = cnt[i], max_i = i;
|
max = cnt[i], max_i = i;
|
||||||
fprintf(stderr, "[M::%s] highest: count[%d] = %ld\n", __func__, max_i, (long)cnt[max_i]);
|
fprintf(stderr, "[M::%s] highest: count[%d] = %ld\n", __func__, max_i, (long)cnt[max_i]);
|
||||||
|
|
||||||
// print histogram
|
// print histogram
|
||||||
for (i = start; i < n_cnt; ++i) {
|
for (i = start; i < n_cnt; ++i) {
|
||||||
int x, exceed = 0;
|
int x, exceed = 0;
|
||||||
x = (int)((double)hist_max * cnt[i] / cnt[max_i] + .499);
|
x = (int)((double)hist_max * cnt[i] / cnt[max_i] + .499);
|
||||||
if (x > hist_max) exceed = 1, x = hist_max; // may happen if cnt[2] is higher
|
if (x > hist_max) exceed = 1, x = hist_max; // may happen if cnt[2] is higher
|
||||||
if (i > max_i && x == 0) break;
|
if (i > max_i && x == 0) break;
|
||||||
ha_hist_line(i, x, exceed, cnt[i]);
|
ha_hist_line(i, x, exceed, cnt[i]);
|
||||||
}
|
}
|
||||||
{
|
{
|
||||||
int x, exceed = 0;
|
int x, exceed = 0;
|
||||||
int64_t rest = 0;
|
int64_t rest = 0;
|
||||||
for (; i < n_cnt; ++i) rest += cnt[i];
|
for (; i < n_cnt; ++i) rest += cnt[i];
|
||||||
x = (int)((double)hist_max * rest / cnt[max_i] + .499);
|
x = (int)((double)hist_max * rest / cnt[max_i] + .499);
|
||||||
if (x > hist_max) exceed = 1, x = hist_max;
|
if (x > hist_max) exceed = 1, x = hist_max;
|
||||||
ha_hist_line(-1, x, exceed, rest);
|
ha_hist_line(-1, x, exceed, rest);
|
||||||
}
|
}
|
||||||
|
|
||||||
// look for smaller peak on the low end
|
// look for smaller peak on the low end
|
||||||
max2 = -1; max2_i = -1;
|
max2 = -1; max2_i = -1;
|
||||||
for (i = max_i - 1; i > low_i; --i) {
|
for (i = max_i - 1; i > low_i; --i) {
|
||||||
if (cnt[i] >= cnt[i-1] && cnt[i] >= cnt[i+1]) {
|
if (cnt[i] >= cnt[i-1] && cnt[i] >= cnt[i+1]) {
|
||||||
if (cnt[i] > max2) max2 = cnt[i], max2_i = i;
|
if (cnt[i] > max2) max2 = cnt[i], max2_i = i;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (max2_i > low_i && max2_i < max_i) {
|
if (max2_i > low_i && max2_i < max_i) {
|
||||||
for (i = max2_i + 1, min = max; i < max_i; ++i)
|
for (i = max2_i + 1, min = max; i < max_i; ++i)
|
||||||
if (cnt[i] < min) min = cnt[i];
|
if (cnt[i] < min) min = cnt[i];
|
||||||
if (max2 < max * 0.05 || min > max2 * 0.95)
|
if (max2 < max * 0.05 || min > max2 * 0.95)
|
||||||
max2 = -1, max2_i = -1;
|
max2 = -1, max2_i = -1;
|
||||||
}
|
}
|
||||||
if (max2 > 0) fprintf(stderr, "[M::%s] left: count[%d] = %ld\n", __func__, max2_i, (long)cnt[max2_i]);
|
if (max2 > 0) fprintf(stderr, "[M::%s] left: count[%d] = %ld\n", __func__, max2_i, (long)cnt[max2_i]);
|
||||||
else fprintf(stderr, "[M::%s] left: none\n", __func__);
|
else fprintf(stderr, "[M::%s] left: none\n", __func__);
|
||||||
|
|
||||||
// look for smaller peak on the high end
|
// look for smaller peak on the high end
|
||||||
max3 = -1; max3_i = -1;
|
max3 = -1; max3_i = -1;
|
||||||
for (i = max_i + 1; i < n_cnt - 1; ++i) {
|
for (i = max_i + 1; i < n_cnt - 1; ++i) {
|
||||||
if (cnt[i] >= cnt[i-1] && cnt[i] >= cnt[i+1]) {
|
if (cnt[i] >= cnt[i-1] && cnt[i] >= cnt[i+1]) {
|
||||||
if (cnt[i] > max3) max3 = cnt[i], max3_i = i;
|
if (cnt[i] > max3) max3 = cnt[i], max3_i = i;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (max3_i > max_i) {
|
if (max3_i > max_i) {
|
||||||
for (i = max_i + 1, min = max; i < max3_i; ++i)
|
for (i = max_i + 1, min = max; i < max3_i; ++i)
|
||||||
if (cnt[i] < min) min = cnt[i];
|
if (cnt[i] < min) min = cnt[i];
|
||||||
if (max3 < max * 0.05 || min > max3 * 0.95 || max3_i > max_i * 2.5)
|
if (max3 < max * 0.05 || min > max3 * 0.95 || max3_i > max_i * 2.5)
|
||||||
max3 = -1, max3_i = -1;
|
max3 = -1, max3_i = -1;
|
||||||
}
|
}
|
||||||
if (max3 > 0) fprintf(stderr, "[M::%s] right: count[%d] = %ld\n", __func__, max3_i, (long)cnt[max3_i]);
|
if (max3 > 0) fprintf(stderr, "[M::%s] right: count[%d] = %ld\n", __func__, max3_i, (long)cnt[max3_i]);
|
||||||
else fprintf(stderr, "[M::%s] right: none\n", __func__);
|
else fprintf(stderr, "[M::%s] right: none\n", __func__);
|
||||||
|
|
||||||
if(m_peak_hom > 0) return adj_m_peak_hom(m_peak_hom, max_i, max2_i, max3_i, peak_het);
|
if(m_peak_hom > 0) return adj_m_peak_hom(m_peak_hom, max_i, max2_i, max3_i, peak_het);
|
||||||
if (max3_i > 0) {
|
if (max3_i > 0) {
|
||||||
*peak_het = max_i;
|
*peak_het = max_i;
|
||||||
return max3_i;
|
return max3_i;
|
||||||
} else {
|
} else {
|
||||||
if (max2_i > 0) *peak_het = max2_i;
|
if (max2_i > 0) *peak_het = max2_i;
|
||||||
return max_i;
|
return max_i;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
+4574
-4574
File diff suppressed because it is too large
Load Diff
@@ -1,79 +1,79 @@
|
|||||||
#ifndef __HORDER__
|
#ifndef __HORDER__
|
||||||
#define __HORDER__
|
#define __HORDER__
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include "hic.h"
|
#include "hic.h"
|
||||||
#include "rcut.h"
|
#include "rcut.h"
|
||||||
|
|
||||||
#define get_hit_srev(x, k) ((x).a.a[(k)].s>>63)
|
#define get_hit_srev(x, k) ((x).a.a[(k)].s>>63)
|
||||||
#define get_hit_slen(x, k) ((x).a.a[(k)].len>>32)
|
#define get_hit_slen(x, k) ((x).a.a[(k)].len>>32)
|
||||||
#define get_hit_suid(x, k) (((x).a.a[(k)].s<<1)>>(64 - (x).uID_bits))
|
#define get_hit_suid(x, k) (((x).a.a[(k)].s<<1)>>(64 - (x).uID_bits))
|
||||||
#define get_hit_spos(x, k) ((x).a.a[(k)].s & (x).pos_mode)
|
#define get_hit_spos(x, k) ((x).a.a[(k)].s & (x).pos_mode)
|
||||||
#define get_hit_spos_e(x, k) (get_hit_srev((x),(k))?\
|
#define get_hit_spos_e(x, k) (get_hit_srev((x),(k))?\
|
||||||
((get_hit_spos((x),(k))+1>=get_hit_slen((x),(k)))?\
|
((get_hit_spos((x),(k))+1>=get_hit_slen((x),(k)))?\
|
||||||
(get_hit_spos((x),(k))+1-get_hit_slen((x),(k))):0)\
|
(get_hit_spos((x),(k))+1-get_hit_slen((x),(k))):0)\
|
||||||
:(get_hit_spos((x),(k))+get_hit_slen((x),(k))-1))
|
:(get_hit_spos((x),(k))+get_hit_slen((x),(k))-1))
|
||||||
|
|
||||||
#define get_hit_erev(x, k) ((x).a.a[(k)].e>>63)
|
#define get_hit_erev(x, k) ((x).a.a[(k)].e>>63)
|
||||||
#define get_hit_elen(x, k) ((uint32_t)((x).a.a[(k)].len))
|
#define get_hit_elen(x, k) ((uint32_t)((x).a.a[(k)].len))
|
||||||
#define get_hit_euid(x, k) (((x).a.a[(k)].e<<1)>>(64 - (x).uID_bits))
|
#define get_hit_euid(x, k) (((x).a.a[(k)].e<<1)>>(64 - (x).uID_bits))
|
||||||
#define get_hit_epos(x, k) ((x).a.a[(k)].e & (x).pos_mode)
|
#define get_hit_epos(x, k) ((x).a.a[(k)].e & (x).pos_mode)
|
||||||
#define get_hit_epos_e(x, k) (get_hit_erev((x),(k))?\
|
#define get_hit_epos_e(x, k) (get_hit_erev((x),(k))?\
|
||||||
((get_hit_epos((x),(k))+1>=get_hit_elen((x),(k)))?\
|
((get_hit_epos((x),(k))+1>=get_hit_elen((x),(k)))?\
|
||||||
(get_hit_epos((x),(k))+1-get_hit_elen((x),(k))):0)\
|
(get_hit_epos((x),(k))+1-get_hit_elen((x),(k))):0)\
|
||||||
:(get_hit_epos((x),(k))+get_hit_elen((x),(k))-1))
|
:(get_hit_epos((x),(k))+get_hit_elen((x),(k))-1))
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint32_t v;
|
uint32_t v;
|
||||||
uint32_t u;
|
uint32_t u;
|
||||||
uint32_t occ:31, del:1;
|
uint32_t occ:31, del:1;
|
||||||
double w, nw;
|
double w, nw;
|
||||||
} osg_arc_t;
|
} osg_arc_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
double mw[2], ez[2];
|
double mw[2], ez[2];
|
||||||
uint8_t del;
|
uint8_t del;
|
||||||
} osg_seq_t;
|
} osg_seq_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint32_t m_arc, n_arc:31, is_srt:1;
|
uint32_t m_arc, n_arc:31, is_srt:1;
|
||||||
osg_arc_t *arc;
|
osg_arc_t *arc;
|
||||||
|
|
||||||
uint32_t m_seq, n_seq:31, is_symm:1;
|
uint32_t m_seq, n_seq:31, is_symm:1;
|
||||||
osg_seq_t *seq;
|
osg_seq_t *seq;
|
||||||
|
|
||||||
uint64_t *idx;
|
uint64_t *idx;
|
||||||
} osg_t;
|
} osg_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
osg_t *g;
|
osg_t *g;
|
||||||
}scg_t;
|
}scg_t;
|
||||||
typedef struct {
|
typedef struct {
|
||||||
kvec_t(uint64_t) avoid;
|
kvec_t(uint64_t) avoid;
|
||||||
// kvec_t(uint64_t) occ;
|
// kvec_t(uint64_t) occ;
|
||||||
// kvec_t(uint8_t) hf;
|
// kvec_t(uint8_t) hf;
|
||||||
kvec_pe_hit r_hits, u_hits;
|
kvec_pe_hit r_hits, u_hits;
|
||||||
ma_ug_t *ug;
|
ma_ug_t *ug;
|
||||||
asg_t *r_g;
|
asg_t *r_g;
|
||||||
scg_t sg;
|
scg_t sg;
|
||||||
}horder_t;
|
}horder_t;
|
||||||
|
|
||||||
horder_t *init_horder_t(kvec_pe_hit *i_hits, uint64_t i_hits_uid_bits, uint64_t i_hits_pos_mode,
|
horder_t *init_horder_t(kvec_pe_hit *i_hits, uint64_t i_hits_uid_bits, uint64_t i_hits_pos_mode,
|
||||||
asg_t *i_rg, ma_ug_t* i_ug, bubble_type* bub, kv_u_trans_t *ref, ug_opt_t *opt, uint32_t round);
|
asg_t *i_rg, ma_ug_t* i_ug, bubble_type* bub, kv_u_trans_t *ref, ug_opt_t *opt, uint32_t round);
|
||||||
void destory_horder_t(horder_t **h);
|
void destory_horder_t(horder_t **h);
|
||||||
void horder_clean_sg_by_utg(asg_t *sg, ma_ug_t *ug);
|
void horder_clean_sg_by_utg(asg_t *sg, ma_ug_t *ug);
|
||||||
kvec_pe_hit *get_r_hits_for_trio(kvec_pe_hit *u_hits, asg_t* r_g, ma_ug_t* ug, bubble_type* bub, uint64_t uID_bits, uint64_t pos_mode);
|
kvec_pe_hit *get_r_hits_for_trio(kvec_pe_hit *u_hits, asg_t* r_g, ma_ug_t* ug, bubble_type* bub, uint64_t uID_bits, uint64_t pos_mode);
|
||||||
void update_switch_unitig(ma_ug_t *ug, asg_t *rg, kvec_pe_hit *hits, kv_u_trans_t *k_trans, uint64_t cutoff_s, uint64_t cutoff_e,
|
void update_switch_unitig(ma_ug_t *ug, asg_t *rg, kvec_pe_hit *hits, kv_u_trans_t *k_trans, uint64_t cutoff_s, uint64_t cutoff_e,
|
||||||
uint64_t min_ulen, double boundaryRate);
|
uint64_t min_ulen, double boundaryRate);
|
||||||
kvec_pe_hit *get_r_hits_order(kvec_pe_hit *uhits, uint64_t hits_uid_bits, uint64_t hits_pos_mode,
|
kvec_pe_hit *get_r_hits_order(kvec_pe_hit *uhits, uint64_t hits_uid_bits, uint64_t hits_pos_mode,
|
||||||
asg_t *rg, ma_ug_t* ug, bubble_type* bub);
|
asg_t *rg, ma_ug_t* ug, bubble_type* bub);
|
||||||
void ha_aware_order(kvec_pe_hit *r_hits, asg_t *rg, ma_ug_t *ug_fa, ma_ug_t *ug_mo, kv_u_trans_t *ref,
|
void ha_aware_order(kvec_pe_hit *r_hits, asg_t *rg, ma_ug_t *ug_fa, ma_ug_t *ug_mo, kv_u_trans_t *ref,
|
||||||
ug_opt_t *opt, uint32_t round);
|
ug_opt_t *opt, uint32_t round);
|
||||||
spg_t *horder_utg(kvec_pe_hit *i_hits, uint64_t i_hits_uid_bits, uint64_t i_hits_pos_mode,
|
spg_t *horder_utg(kvec_pe_hit *i_hits, uint64_t i_hits_uid_bits, uint64_t i_hits_pos_mode,
|
||||||
asg_t *i_rg, ma_ug_t* i_ug, bubble_type* bub, ug_opt_t *opt);
|
asg_t *i_rg, ma_ug_t* i_ug, bubble_type* bub, ug_opt_t *opt);
|
||||||
void layout_mc_clus_t(const mc_match_t *ma, uint32_t *a, uint32_t an, scg_t *sg, uint32_t *buf, uint64_t *idx, ma_ug_t* ug,
|
void layout_mc_clus_t(const mc_match_t *ma, uint32_t *a, uint32_t an, scg_t *sg, uint32_t *buf, uint64_t *idx, ma_ug_t* ug,
|
||||||
double min_cut, double max_cut, uint64_t cut_round);
|
double min_cut, double max_cut, uint64_t cut_round);
|
||||||
void osg_destroy(osg_t *g);
|
void osg_destroy(osg_t *g);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,190 +1,193 @@
|
|||||||
#ifndef __HA_HTAB_H__
|
#ifndef __HA_HTAB_H__
|
||||||
#define __HA_HTAB_H__
|
#define __HA_HTAB_H__
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include "Process_Read.h"
|
#include "Process_Read.h"
|
||||||
#include "CommandLines.h"
|
#include "CommandLines.h"
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
size_t n, m;
|
size_t n, m;
|
||||||
uint64_t *a;
|
uint64_t *a;
|
||||||
} st_mt_t;
|
} st_mt_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint64_t x; ///x is the hash key
|
uint64_t x; ///x is the hash key
|
||||||
///rid is the read id, pos is the end pos of this minimizer, rev is the direction
|
///rid is the read id, pos is the end pos of this minimizer, rev is the direction
|
||||||
///span is the length of this k-mer. For non-HPC k-mer, span may not be equal to k
|
///span is the length of this k-mer. For non-HPC k-mer, span may not be equal to k
|
||||||
uint64_t rid:28, pos:27, rev:1, span:8;
|
uint64_t rid:28, pos:27, rev:1, span:8;
|
||||||
} ha_mz1_t;
|
} ha_mz1_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint64_t rid:28, pos:27, rev:1, span:8; // actually it is not necessary to keep span in the index
|
uint64_t rid:28, pos:27, rev:1, span:8; // actually it is not necessary to keep span in the index
|
||||||
} ha_idxpos_t;
|
} ha_idxpos_t;
|
||||||
|
|
||||||
typedef struct { uint32_t n, m; ha_mz1_t *a; } ha_mz1_v;
|
typedef struct { uint32_t n, m; ha_mz1_t *a; } ha_mz1_v;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint64_t x; ///x is the hash key
|
uint64_t x; ///x is the hash key
|
||||||
uint64_t rid:31, rev:1, pos:32;
|
uint64_t rid:31, rev:1, pos:32;
|
||||||
uint8_t span;
|
uint8_t span;
|
||||||
} ha_mzl_t;
|
} ha_mzl_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint64_t rid:31, rev:1, pos:32;
|
uint64_t rid:31, rev:1, pos:32;
|
||||||
uint8_t span;
|
uint8_t span;
|
||||||
} ha_idxposl_t;
|
} ha_idxposl_t;
|
||||||
|
|
||||||
typedef struct { uint32_t n, m; ha_mzl_t *a; } ha_mzl_v;
|
typedef struct { uint32_t n, m; ha_mzl_t *a; } ha_mzl_v;
|
||||||
|
|
||||||
typedef struct { // a simplified version of kdq
|
typedef struct { // a simplified version of kdq
|
||||||
int front, count;
|
int front, count;
|
||||||
int a[64];
|
int a[64];
|
||||||
} tiny_queue_t;
|
} tiny_queue_t;
|
||||||
|
|
||||||
static inline void tq_push(tiny_queue_t *q, int x)
|
static inline void tq_push(tiny_queue_t *q, int x)
|
||||||
{
|
{
|
||||||
q->a[((q->count++) + q->front) & 0x3f] = x;
|
q->a[((q->count++) + q->front) & 0x3f] = x;
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline int tq_shift(tiny_queue_t *q)
|
static inline int tq_shift(tiny_queue_t *q)
|
||||||
{
|
{
|
||||||
int x;
|
int x;
|
||||||
if (q->count == 0) return -1;
|
if (q->count == 0) return -1;
|
||||||
x = q->a[q->front++];
|
x = q->a[q->front++];
|
||||||
q->front &= 0x3f;
|
q->front &= 0x3f;
|
||||||
--q->count;
|
--q->count;
|
||||||
return x;
|
return x;
|
||||||
}
|
}
|
||||||
|
|
||||||
struct ha_pt_s;
|
struct ha_pt_s;
|
||||||
typedef struct ha_pt_s ha_pt_t;
|
typedef struct ha_pt_s ha_pt_t;
|
||||||
|
|
||||||
struct ha_abuf_s;
|
struct ha_abuf_s;
|
||||||
typedef struct ha_abuf_s ha_abuf_t;
|
typedef struct ha_abuf_s ha_abuf_t;
|
||||||
|
|
||||||
struct ha_abufl_s;
|
struct ha_abufl_s;
|
||||||
typedef struct ha_abufl_s ha_abufl_t;
|
typedef struct ha_abufl_s ha_abufl_t;
|
||||||
|
|
||||||
extern const unsigned char seq_nt4_table[256];
|
extern const unsigned char seq_nt4_table[256];
|
||||||
extern void *ha_flt_tab;
|
extern void *ha_flt_tab;
|
||||||
extern ha_pt_t *ha_idx;
|
extern ha_pt_t *ha_idx;
|
||||||
extern void *ha_flt_tab_hp;
|
extern void *ha_flt_tab_hp;
|
||||||
extern ha_pt_t *ha_idx_hp;
|
extern ha_pt_t *ha_idx_hp;
|
||||||
extern void *ha_ct_table;
|
extern void *ha_ct_table;
|
||||||
|
|
||||||
void *ha_ft_ul_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int k, int w, int cutoff);
|
void *ha_ft_ul_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int k, int w, int cutoff);
|
||||||
void *ha_ft_ug_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int is_HPC, int k, int w, int min_freq, int max_freq);
|
void *ha_ft_ug_gen(const hifiasm_opt_t *asm_opt, ma_utg_v *us, int is_HPC, int k, int w, int min_freq, int max_freq);
|
||||||
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode, int read_from_store);
|
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov, int is_hp_mode, int read_from_store);
|
||||||
int32_t ha_ft_cnt(const void *hh, uint64_t y);
|
int32_t ha_ft_cnt(const void *hh, uint64_t y);
|
||||||
void ha_ft_destroy(void *h);
|
void ha_ft_destroy(void *h);
|
||||||
|
|
||||||
ha_pt_t *ha_pt_ul_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, ma_utg_v *us, int k, int w, int cutoff);
|
ha_pt_t *ha_pt_ul_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, ma_utg_v *us, int k, int w, int cutoff);
|
||||||
ha_pt_t *ha_pt_ug_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, ma_utg_v *us, int is_HPC, int k, int w, int min_freq);
|
ha_pt_t *ha_pt_ug_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, ma_utg_v *us, int is_HPC, int k, int w, int min_freq);
|
||||||
ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_from_store, int is_hp_mode, All_reads *rs, int *hom_cov, int *het_cov);
|
ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_from_store, int is_hp_mode, All_reads *rs, int *hom_cov, int *het_cov);
|
||||||
void ha_pt_destroy(ha_pt_t *h);
|
void ha_pt_destroy(ha_pt_t *h);
|
||||||
const ha_idxpos_t *ha_pt_get(const ha_pt_t *h, uint64_t hash, int *n);
|
const ha_idxpos_t *ha_pt_get(const ha_pt_t *h, uint64_t hash, int *n);
|
||||||
const ha_idxposl_t *ha_ptl_get(const ha_pt_t *h, uint64_t hash, int *n);
|
const ha_idxposl_t *ha_ptl_get(const ha_pt_t *h, uint64_t hash, int *n);
|
||||||
const int ha_pt_cnt(const ha_pt_t *h, uint64_t hash);
|
const int ha_pt_cnt(const ha_pt_t *h, uint64_t hash);
|
||||||
|
|
||||||
int write_pt_index(void *flt_tab, ha_pt_t *ha_idx, All_reads* r, hifiasm_opt_t* opt, char* file_name);
|
int write_pt_index(void *flt_tab, ha_pt_t *ha_idx, All_reads* r, hifiasm_opt_t* opt, char* file_name);
|
||||||
int load_pt_index(void **r_flt_tab, ha_pt_t **r_ha_idx, All_reads* r, hifiasm_opt_t* opt, char* file_name);
|
int load_pt_index(void **r_flt_tab, ha_pt_t **r_ha_idx, All_reads* r, hifiasm_opt_t* opt, char* file_name);
|
||||||
int uidx_write(void *flt_tab, ha_pt_t *ha_idx, char* file_name, ma_ug_t *ug);
|
void refresh_pt_idx(void **flt_tab, ha_pt_t **ha_idx, All_reads *r, hifiasm_opt_t *opt, char *file_name, uint8_t is_w);
|
||||||
int uidx_load(void **r_flt_tab, ha_pt_t **r_ha_idx, char* file_name, ma_ug_t *ug);
|
int uidx_write(void *flt_tab, ha_pt_t *ha_idx, char* file_name, ma_ug_t *ug);
|
||||||
int write_ct_index(void *ct_idx, char* file_name);
|
int uidx_load(void **r_flt_tab, ha_pt_t **r_ha_idx, char* file_name, ma_ug_t *ug);
|
||||||
int load_ct_index(void **ct_idx, char* file_name);
|
int write_ct_index(void *ct_idx, char* file_name);
|
||||||
int query_ct_index(void* ct_idx, uint64_t hash);
|
int load_ct_index(void **ct_idx, char* file_name);
|
||||||
|
int query_ct_index(void* ct_idx, uint64_t hash);
|
||||||
ha_abuf_t *ha_abuf_init_buf(void *km);
|
uint64_t tmp_pt_pro(void **r_flt_tab, ha_pt_t **r_ha_idx, All_reads *r, hifiasm_opt_t *opt, char *file_name, uint64_t rr, uint64_t tot_rr, uint64_t is_load);
|
||||||
ha_abufl_t *ha_abufl_init_buf(void *km);
|
|
||||||
void ha_abuf_destroy_buf(void *km, ha_abuf_t *ab);
|
ha_abuf_t *ha_abuf_init_buf(void *km);
|
||||||
void ha_abufl_destroy_buf(void *km, ha_abufl_t *ab);
|
ha_abufl_t *ha_abufl_init_buf(void *km);
|
||||||
void ha_abufl_free_buf(void *km, ha_abufl_t *ab, int is_z);
|
void ha_abuf_destroy_buf(void *km, ha_abuf_t *ab);
|
||||||
ha_abuf_t *ha_abuf_init(void);
|
void ha_abufl_destroy_buf(void *km, ha_abufl_t *ab);
|
||||||
void ha_abuf_destroy(ha_abuf_t *ab);
|
void ha_abufl_free_buf(void *km, ha_abufl_t *ab, int is_z);
|
||||||
uint64_t ha_abuf_mem(const ha_abuf_t *ab);
|
ha_abuf_t *ha_abuf_init(void);
|
||||||
ha_abufl_t *ha_abufl_init(void);
|
void ha_abuf_destroy(ha_abuf_t *ab);
|
||||||
void ha_abufl_destroy(ha_abufl_t *ab);
|
uint64_t ha_abuf_mem(const ha_abuf_t *ab);
|
||||||
uint64_t ha_abufl_mem(const ha_abufl_t *ab);
|
ha_abufl_t *ha_abufl_init(void);
|
||||||
|
void ha_abufl_destroy(ha_abufl_t *ab);
|
||||||
double yak_cputime(void);
|
uint64_t ha_abufl_mem(const ha_abufl_t *ab);
|
||||||
void yak_reset_realtime(void);
|
|
||||||
double yak_realtime_0(void);
|
double yak_cputime(void);
|
||||||
double yak_realtime(void);
|
void yak_reset_realtime(void);
|
||||||
long yak_peakrss(void);
|
double yak_realtime_0(void);
|
||||||
double yak_peakrss_in_gb(void);
|
double yak_realtime(void);
|
||||||
double yak_cpu_usage(void);
|
long yak_peakrss(void);
|
||||||
|
double yak_peakrss_in_gb(void);
|
||||||
void ha_triobin(const hifiasm_opt_t *opt);
|
double yak_cpu_usage(void);
|
||||||
uint32_t test_yak_binning(char* fn, char *cmd);
|
|
||||||
uint32_t *ha_polybin_list(const hifiasm_opt_t *opt);
|
void ha_triobin(const hifiasm_opt_t *opt);
|
||||||
|
uint32_t test_yak_binning(char* fn, char *cmd);
|
||||||
void mz1_ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mz1_v *p, const void *hf, int sample_dist, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct, ha_pt_t *pt, int min_freq, int32_t dp_min_len, float dp_e, st_mt_t *mt, int32_t ws, int32_t is_unique, void *km);
|
uint32_t *ha_polybin_list(const hifiasm_opt_t *opt);
|
||||||
void mz2_ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mzl_v *p, const void *hf, int sample_dist, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct, ha_pt_t *pt, int min_freq, int32_t dp_min_len, float dp_e, st_mt_t *mt, int32_t ws, int32_t is_unique, void *km);
|
uint32_t *ha_charbin_list(const hifiasm_opt_t *opt, uint8_t **idx, uint32_t *idx_n);
|
||||||
int ha_analyze_count(int n_cnt, int start_cnt, int m_peak_hom, const int64_t *cnt, int *peak_het);
|
|
||||||
int adj_m_peak_hom(int m_peak_hom, int max_i, int max2_i, int max3_i, int *peak_het);
|
void mz1_ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mz1_v *p, const void *hf, int sample_dist, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct, ha_pt_t *pt, int min_freq, int32_t dp_min_len, float dp_e, st_mt_t *mt, int32_t ws, int32_t is_unique, void *km);
|
||||||
void print_hist_lines(int n_cnt, int start_cnt, const int64_t *cnt);
|
void mz2_ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mzl_v *p, const void *hf, int sample_dist, kvec_t_u8_warp* k_flag, kvec_t_u64_warp* dbg_ct, ha_pt_t *pt, int min_freq, int32_t dp_min_len, float dp_e, st_mt_t *mt, int32_t ws, int32_t is_unique, void *km);
|
||||||
void debug_adapter(const hifiasm_opt_t *asm_opt, All_reads *rs);
|
int ha_analyze_count(int n_cnt, int start_cnt, int m_peak_hom, const int64_t *cnt, int *peak_het);
|
||||||
|
int adj_m_peak_hom(int m_peak_hom, int max_i, int max2_i, int max3_i, int *peak_het);
|
||||||
|
void print_hist_lines(int n_cnt, int start_cnt, const int64_t *cnt);
|
||||||
inline int mz_low_b(int peak_hom, int peak_het)
|
void debug_adapter(const hifiasm_opt_t *asm_opt, All_reads *rs);
|
||||||
{
|
|
||||||
int low_freq = 2;
|
|
||||||
if(peak_het > 0) low_freq = peak_het/2;
|
inline int mz_low_b(int peak_hom, int peak_het)
|
||||||
else if(peak_hom > 0) low_freq = peak_hom/4;
|
{
|
||||||
if(low_freq < 2) low_freq = 2;
|
int low_freq = 2;
|
||||||
return low_freq;
|
if(peak_het > 0) low_freq = peak_het/2;
|
||||||
}
|
else if(peak_hom > 0) low_freq = peak_hom/4;
|
||||||
|
if(low_freq < 2) low_freq = 2;
|
||||||
static inline uint64_t yak_hash64(uint64_t key, uint64_t mask) // invertible integer hash function
|
return low_freq;
|
||||||
{
|
}
|
||||||
key = (~key + (key << 21)) & mask; // key = (key << 21) - key - 1;
|
|
||||||
key = key ^ key >> 24;
|
static inline uint64_t yak_hash64(uint64_t key, uint64_t mask) // invertible integer hash function
|
||||||
key = ((key + (key << 3)) + (key << 8)) & mask; // key * 265
|
{
|
||||||
key = key ^ key >> 14;
|
key = (~key + (key << 21)) & mask; // key = (key << 21) - key - 1;
|
||||||
key = ((key + (key << 2)) + (key << 4)) & mask; // key * 21
|
key = key ^ key >> 24;
|
||||||
key = key ^ key >> 28;
|
key = ((key + (key << 3)) + (key << 8)) & mask; // key * 265
|
||||||
key = (key + (key << 31)) & mask;
|
key = key ^ key >> 14;
|
||||||
return key;
|
key = ((key + (key << 2)) + (key << 4)) & mask; // key * 21
|
||||||
}
|
key = key ^ key >> 28;
|
||||||
|
key = (key + (key << 31)) & mask;
|
||||||
static inline uint64_t yak_hash64_64(uint64_t key)
|
return key;
|
||||||
{
|
}
|
||||||
key = ~key + (key << 21);
|
|
||||||
key = key ^ key >> 24;
|
static inline uint64_t yak_hash64_64(uint64_t key)
|
||||||
key = (key + (key << 3)) + (key << 8);
|
{
|
||||||
key = key ^ key >> 14;
|
key = ~key + (key << 21);
|
||||||
key = (key + (key << 2)) + (key << 4);
|
key = key ^ key >> 24;
|
||||||
key = key ^ key >> 28;
|
key = (key + (key << 3)) + (key << 8);
|
||||||
key = key + (key << 31);
|
key = key ^ key >> 14;
|
||||||
return key;
|
key = (key + (key << 2)) + (key << 4);
|
||||||
}
|
key = key ^ key >> 28;
|
||||||
|
key = key + (key << 31);
|
||||||
static inline uint64_t yak_hash_long(uint64_t x[4])
|
return key;
|
||||||
{
|
}
|
||||||
///compare forward k-mer and reverse complementary strand
|
|
||||||
int j = x[1] < x[3]? 0 : 1;
|
static inline uint64_t yak_hash_long(uint64_t x[4])
|
||||||
return yak_hash64_64(x[j<<1|0]) + yak_hash64_64(x[j<<1|1]);
|
{
|
||||||
}
|
///compare forward k-mer and reverse complementary strand
|
||||||
|
int j = x[1] < x[3]? 0 : 1;
|
||||||
#define CALLOC(ptr, len) ((ptr) = (__typeof__(ptr))calloc((len), sizeof(*(ptr))))
|
return yak_hash64_64(x[j<<1|0]) + yak_hash64_64(x[j<<1|1]);
|
||||||
#define MALLOC(ptr, len) ((ptr) = (__typeof__(ptr))malloc((len) * sizeof(*(ptr))))
|
}
|
||||||
#define REALLOC(ptr, len) ((ptr) = (__typeof__(ptr))realloc((ptr), (len) * sizeof(*(ptr))))
|
|
||||||
#define MEMCPY(dest, src, len) (memcpy((dest), (src), (len) * sizeof(*(src))))
|
#define CALLOC(ptr, len) ((ptr) = ((((len)*sizeof(*(ptr))) <= 9223372036854775807)?((__typeof__(ptr))calloc((len), sizeof(*(ptr)))):(NULL)))
|
||||||
|
#define MALLOC(ptr, len) ((ptr) = (__typeof__(ptr))malloc((len) * sizeof(*(ptr))))
|
||||||
#ifndef kroundup32
|
#define REALLOC(ptr, len) ((ptr) = (__typeof__(ptr))realloc((ptr), (len) * sizeof(*(ptr))))
|
||||||
#define kroundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
#define MEMCPY(dest, src, len) (memcpy((dest), (src), (len) * sizeof(*(src))))
|
||||||
#endif
|
|
||||||
|
#ifndef kroundup32
|
||||||
#ifndef kroundup64
|
#define kroundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
||||||
#define kroundup64(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, x|=(x)>>32, ++(x))
|
#endif
|
||||||
#endif
|
|
||||||
|
#ifndef kroundup64
|
||||||
#ifndef klib_unused
|
#define kroundup64(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, x|=(x)>>32, ++(x))
|
||||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
#endif
|
||||||
#define klib_unused __attribute__ ((__unused__))
|
|
||||||
#else
|
#ifndef klib_unused
|
||||||
#define klib_unused
|
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
||||||
#endif
|
#define klib_unused __attribute__ ((__unused__))
|
||||||
#endif /* klib_unused */
|
#else
|
||||||
|
#define klib_unused
|
||||||
#endif // __YAK_H__
|
#endif
|
||||||
|
#endif /* klib_unused */
|
||||||
|
|
||||||
|
#endif // __YAK_H__
|
||||||
|
|||||||
@@ -1,136 +1,136 @@
|
|||||||
#ifndef __INTER__
|
#ifndef __INTER__
|
||||||
#define __INTER__
|
#define __INTER__
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
#include "Process_Read.h"
|
#include "Process_Read.h"
|
||||||
#include "hic.h"
|
#include "hic.h"
|
||||||
|
|
||||||
#define G_CHAIN_BW 16//128
|
#define G_CHAIN_BW 16//128
|
||||||
#define FLANK_M (0x7fffU)
|
#define FLANK_M (0x7fffU)
|
||||||
#define P_CHAIN_COV 0.985
|
#define P_CHAIN_COV 0.985
|
||||||
#define P_FRAGEMENT_CHAIN_COV 0.20
|
#define P_FRAGEMENT_CHAIN_COV 0.20
|
||||||
#define P_FRAGEMENT_PRIMARY_CHAIN_COV 0.70
|
#define P_FRAGEMENT_PRIMARY_CHAIN_COV 0.70
|
||||||
#define P_FRAGEMENT_PRIMARY_SECOND_COV 0.25
|
#define P_FRAGEMENT_PRIMARY_SECOND_COV 0.25
|
||||||
#define P_CHAIN_SCORE 0.6
|
#define P_CHAIN_SCORE 0.6
|
||||||
#define G_CHAIN_GAP 0.1
|
#define G_CHAIN_GAP 0.1
|
||||||
#define UG_SKIP 5
|
#define UG_SKIP 5
|
||||||
#define RG_SKIP 25
|
#define RG_SKIP 25
|
||||||
#define UG_SKIP_GRAPH_N 72
|
#define UG_SKIP_GRAPH_N 72
|
||||||
#define UG_SKIP_N 100
|
#define UG_SKIP_N 100
|
||||||
#define UG_ITER_N 5000
|
#define UG_ITER_N 5000
|
||||||
#define UG_DIS_N 50000
|
#define UG_DIS_N 50000
|
||||||
// #define UG_TRANS_W 2
|
// #define UG_TRANS_W 2
|
||||||
#define UG_TRANS_W 2
|
#define UG_TRANS_W 2
|
||||||
// #define UG_TRANS_ERR_W 512
|
// #define UG_TRANS_ERR_W 512
|
||||||
#define UG_TRANS_ERR_W 64
|
#define UG_TRANS_ERR_W 64
|
||||||
#define G_CHAIN_TRANS_RATE 0.25
|
#define G_CHAIN_TRANS_RATE 0.25
|
||||||
#define G_CHAIN_TRANS_WEIGHT -1
|
#define G_CHAIN_TRANS_WEIGHT -1
|
||||||
#define G_CHAIN_INDEL 128
|
#define G_CHAIN_INDEL 128
|
||||||
#define W_CHN_PEN_GAP 0.1
|
#define W_CHN_PEN_GAP 0.1
|
||||||
#define N_GCHAIN_RATE 0.04
|
#define N_GCHAIN_RATE 0.04
|
||||||
#define PRIMARY_UL_CHAIN_MIN 75000
|
#define PRIMARY_UL_CHAIN_MIN 75000
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int w, k, bw, max_gap, is_HPC, hap_n, occ_weight, max_gap_pre, max_gc_seq_ext, seed;
|
int w, k, bw, max_gap, is_HPC, hap_n, occ_weight, max_gap_pre, max_gc_seq_ext, seed;
|
||||||
int max_lc_skip, max_lc_iter, min_lc_cnt, min_lc_score, max_gc_skip, ref_bonus;
|
int max_lc_skip, max_lc_iter, min_lc_cnt, min_lc_score, max_gc_skip, ref_bonus;
|
||||||
int min_gc_cnt, min_gc_score, sub_diff, best_n;
|
int min_gc_cnt, min_gc_score, sub_diff, best_n;
|
||||||
float chn_pen_gap, mask_level, pri_ratio;
|
float chn_pen_gap, mask_level, pri_ratio;
|
||||||
///base-alignment
|
///base-alignment
|
||||||
double bw_thres, diff_ec_ul, diff_ec_ul_low, diff_ec_ul_hpc; int max_n_chain, ec_ul_round;
|
double bw_thres, diff_ec_ul, diff_ec_ul_low, diff_ec_ul_hpc; int max_n_chain, ec_ul_round;
|
||||||
} mg_idxopt_t;
|
} mg_idxopt_t;
|
||||||
|
|
||||||
struct mg_tbuf_s {
|
struct mg_tbuf_s {
|
||||||
void *km;
|
void *km;
|
||||||
int frag_gap;
|
int frag_gap;
|
||||||
};
|
};
|
||||||
typedef struct mg_tbuf_s mg_tbuf_t;
|
typedef struct mg_tbuf_s mg_tbuf_t;
|
||||||
|
|
||||||
|
|
||||||
mg_tbuf_t *mg_tbuf_init(void);
|
mg_tbuf_t *mg_tbuf_init(void);
|
||||||
|
|
||||||
void mg_tbuf_destroy(mg_tbuf_t *b);
|
void mg_tbuf_destroy(mg_tbuf_t *b);
|
||||||
|
|
||||||
void *mg_tbuf_get_km(mg_tbuf_t *b);
|
void *mg_tbuf_get_km(mg_tbuf_t *b);
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
FILE *fp;
|
FILE *fp;
|
||||||
ul_vec_t u;
|
ul_vec_t u;
|
||||||
uint64_t flag;
|
uint64_t flag;
|
||||||
} ucr_file_t;
|
} ucr_file_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int32_t off, cnt;
|
int32_t off, cnt;
|
||||||
uint32_t v;
|
uint32_t v;
|
||||||
int32_t score;
|
int32_t score;
|
||||||
} mg_llchain_t;
|
} mg_llchain_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int32_t id, parent;
|
int32_t id, parent;
|
||||||
int32_t off, cnt;
|
int32_t off, cnt;
|
||||||
int32_t n_anchor, score;
|
int32_t n_anchor, score;
|
||||||
int32_t qs, qe;
|
int32_t qs, qe;
|
||||||
int32_t plen, ps, pe;
|
int32_t plen, ps, pe;
|
||||||
int32_t blen, mlen;
|
int32_t blen, mlen;
|
||||||
float div;
|
float div;
|
||||||
uint32_t hash;
|
uint32_t hash;
|
||||||
int32_t subsc, n_sub;
|
int32_t subsc, n_sub;
|
||||||
uint32_t mapq:8, flt:1, dummy:23;
|
uint32_t mapq:8, flt:1, dummy:23;
|
||||||
} mg_gchain_t;
|
} mg_gchain_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
size_t n,m;
|
size_t n,m;
|
||||||
uint64_t *a, tl;
|
uint64_t *a, tl;
|
||||||
kvec_t(char) cc;
|
kvec_t(char) cc;
|
||||||
} mg_dbn_t;
|
} mg_dbn_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int32_t cnt;
|
int32_t cnt;
|
||||||
uint32_t v;
|
uint32_t v;
|
||||||
int32_t score;
|
int32_t score;
|
||||||
uint32_t qs, qe, ts, te;
|
uint32_t qs, qe, ts, te;
|
||||||
} mg_lres_t;
|
} mg_lres_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int32_t n_gc, n_lc;
|
int32_t n_gc, n_lc;
|
||||||
mg_gchain_t *gc;///g_chain; idx in l_chains
|
mg_gchain_t *gc;///g_chain; idx in l_chains
|
||||||
mg_lres_t *lc;///l_chain
|
mg_lres_t *lc;///l_chain
|
||||||
uint64_t qid, qlen;
|
uint64_t qid, qlen;
|
||||||
} mg_gres_t;
|
} mg_gres_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
size_t n,m;
|
size_t n,m;
|
||||||
mg_gres_t *a;
|
mg_gres_t *a;
|
||||||
uint64_t total_base;
|
uint64_t total_base;
|
||||||
uint64_t total_pair;
|
uint64_t total_pair;
|
||||||
} mg_gres_a;
|
} mg_gres_a;
|
||||||
|
|
||||||
void push_uc_block_t(const ug_opt_t *uopt, kv_ul_ov_t *z, char **seq, uint64_t *len, uint64_t b_id);
|
void push_uc_block_t(const ug_opt_t *uopt, kv_ul_ov_t *z, char **seq, uint64_t *len, uint64_t b_id);
|
||||||
void ul_resolve(ma_ug_t *ug, const asg_t *rg, const ug_opt_t *uopt, int hap_n);
|
void ul_resolve(ma_ug_t *ug, const asg_t *rg, const ug_opt_t *uopt, int hap_n);
|
||||||
void ul_load(const ug_opt_t *uopt);
|
void ul_load(const ug_opt_t *uopt);
|
||||||
uint64_t* get_hifi2ul_list(all_ul_t *x, uint64_t hid, uint64_t* a_n);
|
uint64_t* get_hifi2ul_list(all_ul_t *x, uint64_t hid, uint64_t* a_n);
|
||||||
uint64_t ul_refine_alignment(const ug_opt_t *uopt, asg_t *sg);
|
uint64_t ul_refine_alignment(const ug_opt_t *uopt, asg_t *sg);
|
||||||
ma_ug_t *ul_realignment(const ug_opt_t *uopt, asg_t *sg, uint32_t double_check_cache, const char *bin_file);
|
ma_ug_t *ul_realignment(const ug_opt_t *uopt, asg_t *sg, uint32_t double_check_cache, const char *bin_file);
|
||||||
int32_t write_all_ul_t(all_ul_t *x, char* file_name, ma_ug_t *ug);
|
int32_t write_all_ul_t(all_ul_t *x, char* file_name, ma_ug_t *ug);
|
||||||
int32_t load_all_ul_t(all_ul_t *x, char* file_name, All_reads *hR, ma_ug_t *ug);
|
int32_t load_all_ul_t(all_ul_t *x, char* file_name, All_reads *hR, ma_ug_t *ug);
|
||||||
uint32_t ugl_cover_check(uint64_t is, uint64_t ie, ma_utg_t *u);
|
uint32_t ugl_cover_check(uint64_t is, uint64_t ie, ma_utg_t *u);
|
||||||
void filter_ul_ug(ma_ug_t *ug);
|
void filter_ul_ug(ma_ug_t *ug);
|
||||||
void gen_ul_vec_rid_t(all_ul_t *x, All_reads *rdb, ma_ug_t *ug);
|
void gen_ul_vec_rid_t(all_ul_t *x, All_reads *rdb, ma_ug_t *ug);
|
||||||
void update_ug_arch_ul_mul(ma_ug_t *ug);
|
void update_ug_arch_ul_mul(ma_ug_t *ug);
|
||||||
void print_ul_alignment(ma_ug_t *ug, all_ul_t *aln, uint32_t id, const char* cmd);
|
void print_ul_alignment(ma_ug_t *ug, all_ul_t *aln, uint32_t id, const char* cmd);
|
||||||
void clear_all_ul_t(all_ul_t *x);
|
void clear_all_ul_t(all_ul_t *x);
|
||||||
void trans_base_infer(ma_ug_t *ug, asg_t *sg, ug_opt_t *uopt, kv_u_trans_t *res, bubble_type *bub);
|
void trans_base_infer(ma_ug_t *ug, asg_t *sg, ug_opt_t *uopt, kv_u_trans_t *res, bubble_type *bub);
|
||||||
hpc_re_t *gen_hpc_re_t(ma_ug_t *ug);
|
hpc_re_t *gen_hpc_re_t(ma_ug_t *ug);
|
||||||
idx_emask_t* graph_ovlp_binning(ma_ug_t *ug, asg_t *sg, const ug_opt_t *uopt);
|
idx_emask_t* graph_ovlp_binning(ma_ug_t *ug, asg_t *sg, const ug_opt_t *uopt);
|
||||||
uint32_t gen_src_shared_interval_simple(uint32_t src, ma_ug_t *ug, uint64_t *flt, uint64_t flt_n, kv_ul_ov_t *res);
|
uint32_t gen_src_shared_interval_simple(uint32_t src, ma_ug_t *ug, uint64_t *flt, uint64_t flt_n, kv_ul_ov_t *res);
|
||||||
uint64_t check_ul_ov_t_consist(ul_ov_t *x, ul_ov_t *y, int64_t ql, int64_t tl, double diff);
|
uint64_t check_ul_ov_t_consist(ul_ov_t *x, ul_ov_t *y, int64_t ql, int64_t tl, double diff);
|
||||||
uint32_t infer_se(uint32_t qs, uint32_t qe, uint32_t ts, uint32_t te, uint32_t rev,
|
uint32_t infer_se(uint32_t qs, uint32_t qe, uint32_t ts, uint32_t te, uint32_t rev,
|
||||||
uint32_t rqs, uint32_t rqe, uint32_t *rts, uint32_t *rte);
|
uint32_t rqs, uint32_t rqe, uint32_t *rts, uint32_t *rte);
|
||||||
uint32_t clean_contain_g(const ug_opt_t *uopt, asg_t *sg, uint32_t push_trans);
|
uint32_t clean_contain_g(const ug_opt_t *uopt, asg_t *sg, uint32_t push_trans);
|
||||||
void dedup_contain_g(const ug_opt_t *uopt, asg_t *sg);
|
void dedup_contain_g(const ug_opt_t *uopt, asg_t *sg);
|
||||||
void trans_base_mmhap_infer(ma_ug_t *ug, asg_t *sg, ug_opt_t *uopt, kv_u_trans_t *res);
|
void trans_base_mmhap_infer(ma_ug_t *ug, asg_t *sg, ug_opt_t *uopt, kv_u_trans_t *res);
|
||||||
scaf_res_t *gen_contig_path(const ug_opt_t *uopt, asg_t *sg, ma_ug_t *ctg, ma_ug_t *ref);
|
scaf_res_t *gen_contig_path(const ug_opt_t *uopt, asg_t *sg, ma_ug_t *ctg, ma_ug_t *ref);
|
||||||
void gen_contig_trans(const ug_opt_t *uopt, asg_t *sg, ma_ug_t *qry, scaf_res_t *qry_sc, ma_ug_t *ref, scaf_res_t *ref_sc, ma_ug_t *gfa, kv_u_trans_t *ta, uint32_t qoff, uint32_t toff, bubble_type *bu, kv_u_trans_t *res);
|
void gen_contig_trans(const ug_opt_t *uopt, asg_t *sg, ma_ug_t *qry, scaf_res_t *qry_sc, ma_ug_t *ref, scaf_res_t *ref_sc, ma_ug_t *gfa, kv_u_trans_t *ta, uint32_t qoff, uint32_t toff, bubble_type *bu, kv_u_trans_t *res);
|
||||||
void gen_contig_self(const ug_opt_t *uopt, asg_t *sg, ma_ug_t *db, scaf_res_t *db_sc, ma_ug_t *gfa, kv_u_trans_t *ta, uint64_t soff, bubble_type *bu, kv_u_trans_t *res, uint32_t is_exact);
|
void gen_contig_self(const ug_opt_t *uopt, asg_t *sg, ma_ug_t *db, scaf_res_t *db_sc, ma_ug_t *gfa, kv_u_trans_t *ta, uint64_t soff, bubble_type *bu, kv_u_trans_t *res, uint32_t is_exact);
|
||||||
void order_contig_trans(kv_u_trans_t *in);
|
void order_contig_trans(kv_u_trans_t *in);
|
||||||
void sort_uc_block_qe(uc_block_t* a, uint64_t a_n);
|
void sort_uc_block_qe(uc_block_t* a, uint64_t a_n);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
+205
-205
@@ -1,205 +1,205 @@
|
|||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include "kalloc.h"
|
#include "kalloc.h"
|
||||||
|
|
||||||
/* In kalloc, a *core* is a large chunk of contiguous memory. Each core is
|
/* In kalloc, a *core* is a large chunk of contiguous memory. Each core is
|
||||||
* associated with a master header, which keeps the size of the current core
|
* associated with a master header, which keeps the size of the current core
|
||||||
* and the pointer to next core. Kalloc allocates small *blocks* of memory from
|
* and the pointer to next core. Kalloc allocates small *blocks* of memory from
|
||||||
* the cores and organizes free memory blocks in a circular single-linked list.
|
* the cores and organizes free memory blocks in a circular single-linked list.
|
||||||
*
|
*
|
||||||
* In the following diagram, "@" stands for the header of a free block (of type
|
* In the following diagram, "@" stands for the header of a free block (of type
|
||||||
* header_t), "#" for the header of an allocated block (of type size_t), "-"
|
* header_t), "#" for the header of an allocated block (of type size_t), "-"
|
||||||
* for free memory, and "+" for allocated memory.
|
* for free memory, and "+" for allocated memory.
|
||||||
*
|
*
|
||||||
* master This region is core 1. master This region is core 2.
|
* master This region is core 1. master This region is core 2.
|
||||||
* | |
|
* | |
|
||||||
* *@-------#++++++#++++++++++++@-------- *@----------#++++++++++++#+++++++@------------
|
* *@-------#++++++#++++++++++++@-------- *@----------#++++++++++++#+++++++@------------
|
||||||
* | | | |
|
* | | | |
|
||||||
* p=p->ptr->ptr->ptr->ptr p->ptr p->ptr->ptr p->ptr->ptr->ptr
|
* p=p->ptr->ptr->ptr->ptr p->ptr p->ptr->ptr p->ptr->ptr->ptr
|
||||||
*/
|
*/
|
||||||
typedef struct header_t {
|
typedef struct header_t {
|
||||||
size_t size;
|
size_t size;
|
||||||
struct header_t *ptr;
|
struct header_t *ptr;
|
||||||
} header_t;
|
} header_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
void *par;
|
void *par;
|
||||||
size_t min_core_size;
|
size_t min_core_size;
|
||||||
header_t base, *loop_head, *core_head; /* base is a zero-sized block always kept in the loop */
|
header_t base, *loop_head, *core_head; /* base is a zero-sized block always kept in the loop */
|
||||||
} kmem_t;
|
} kmem_t;
|
||||||
|
|
||||||
static void panic(const char *s)
|
static void panic(const char *s)
|
||||||
{
|
{
|
||||||
fprintf(stderr, "%s\n", s);
|
fprintf(stderr, "%s\n", s);
|
||||||
abort();
|
abort();
|
||||||
}
|
}
|
||||||
|
|
||||||
void *km_init2(void *km_par, size_t min_core_size)
|
void *km_init2(void *km_par, size_t min_core_size)
|
||||||
{
|
{
|
||||||
kmem_t *km;
|
kmem_t *km;
|
||||||
km = (kmem_t*)kcalloc(km_par, 1, sizeof(kmem_t));
|
km = (kmem_t*)kcalloc(km_par, 1, sizeof(kmem_t));
|
||||||
km->par = km_par;
|
km->par = km_par;
|
||||||
km->min_core_size = min_core_size > 0? min_core_size : 0x80000;
|
km->min_core_size = min_core_size > 0? min_core_size : 0x80000;
|
||||||
return (void*)km;
|
return (void*)km;
|
||||||
}
|
}
|
||||||
|
|
||||||
void *km_init(void) { return km_init2(0, 0); }
|
void *km_init(void) { return km_init2(0, 0); }
|
||||||
|
|
||||||
void km_destroy(void *_km)
|
void km_destroy(void *_km)
|
||||||
{
|
{
|
||||||
kmem_t *km = (kmem_t*)_km;
|
kmem_t *km = (kmem_t*)_km;
|
||||||
void *km_par;
|
void *km_par;
|
||||||
header_t *p, *q;
|
header_t *p, *q;
|
||||||
if (km == NULL) return;
|
if (km == NULL) return;
|
||||||
km_par = km->par;
|
km_par = km->par;
|
||||||
for (p = km->core_head; p != NULL;) {
|
for (p = km->core_head; p != NULL;) {
|
||||||
q = p->ptr;
|
q = p->ptr;
|
||||||
kfree(km_par, p);
|
kfree(km_par, p);
|
||||||
p = q;
|
p = q;
|
||||||
}
|
}
|
||||||
kfree(km_par, km);
|
kfree(km_par, km);
|
||||||
}
|
}
|
||||||
|
|
||||||
static header_t *morecore(kmem_t *km, size_t nu)
|
static header_t *morecore(kmem_t *km, size_t nu)
|
||||||
{
|
{
|
||||||
header_t *q;
|
header_t *q;
|
||||||
size_t bytes, *p;
|
size_t bytes, *p;
|
||||||
nu = (nu + 1 + (km->min_core_size - 1)) / km->min_core_size * km->min_core_size; /* the first +1 for core header */
|
nu = (nu + 1 + (km->min_core_size - 1)) / km->min_core_size * km->min_core_size; /* the first +1 for core header */
|
||||||
bytes = nu * sizeof(header_t);
|
bytes = nu * sizeof(header_t);
|
||||||
q = (header_t*)kmalloc(km->par, bytes);
|
q = (header_t*)kmalloc(km->par, bytes);
|
||||||
if (!q) panic("[morecore] insufficient memory");
|
if (!q) panic("[morecore] insufficient memory");
|
||||||
q->ptr = km->core_head, q->size = nu, km->core_head = q;
|
q->ptr = km->core_head, q->size = nu, km->core_head = q;
|
||||||
p = (size_t*)(q + 1);
|
p = (size_t*)(q + 1);
|
||||||
*p = nu - 1; /* the size of the free block; -1 because the first unit is used for the core header */
|
*p = nu - 1; /* the size of the free block; -1 because the first unit is used for the core header */
|
||||||
kfree(km, p + 1); /* initialize the new "core"; NB: the core header is not looped. */
|
kfree(km, p + 1); /* initialize the new "core"; NB: the core header is not looped. */
|
||||||
return km->loop_head;
|
return km->loop_head;
|
||||||
}
|
}
|
||||||
|
|
||||||
void kfree(void *_km, void *ap) /* kfree() also adds a new core to the circular list */
|
void kfree(void *_km, void *ap) /* kfree() also adds a new core to the circular list */
|
||||||
{
|
{
|
||||||
header_t *p, *q;
|
header_t *p, *q;
|
||||||
kmem_t *km = (kmem_t*)_km;
|
kmem_t *km = (kmem_t*)_km;
|
||||||
|
|
||||||
if (!ap) return;
|
if (!ap) return;
|
||||||
if (km == NULL) {
|
if (km == NULL) {
|
||||||
free(ap);
|
free(ap);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
p = (header_t*)((size_t*)ap - 1);
|
p = (header_t*)((size_t*)ap - 1);
|
||||||
p->size = *((size_t*)ap - 1);
|
p->size = *((size_t*)ap - 1);
|
||||||
/* Find the pointer that points to the block to be freed. The following loop can stop on two conditions:
|
/* Find the pointer that points to the block to be freed. The following loop can stop on two conditions:
|
||||||
*
|
*
|
||||||
* a) "p>q && p<q->ptr": @------#++++++++#+++++++@------- @---------------#+++++++@-------
|
* a) "p>q && p<q->ptr": @------#++++++++#+++++++@------- @---------------#+++++++@-------
|
||||||
* (can also be in | | | -> | |
|
* (can also be in | | | -> | |
|
||||||
* two cores) q p q->ptr q q->ptr
|
* two cores) q p q->ptr q q->ptr
|
||||||
*
|
*
|
||||||
* @-------- #+++++++++@-------- @-------- @------------------
|
* @-------- #+++++++++@-------- @-------- @------------------
|
||||||
* | | | -> | |
|
* | | | -> | |
|
||||||
* q p q->ptr q q->ptr
|
* q p q->ptr q q->ptr
|
||||||
*
|
*
|
||||||
* b) "q>=q->ptr && (p>q || p<q->ptr)": @-------#+++++ @--------#+++++++ @-------#+++++ @----------------
|
* b) "q>=q->ptr && (p>q || p<q->ptr)": @-------#+++++ @--------#+++++++ @-------#+++++ @----------------
|
||||||
* | | | -> | |
|
* | | | -> | |
|
||||||
* q->ptr q p q->ptr q
|
* q->ptr q p q->ptr q
|
||||||
*
|
*
|
||||||
* #+++++++@----- #++++++++@------- @------------- #++++++++@-------
|
* #+++++++@----- #++++++++@------- @------------- #++++++++@-------
|
||||||
* | | | -> | |
|
* | | | -> | |
|
||||||
* p q->ptr q q->ptr q
|
* p q->ptr q q->ptr q
|
||||||
*/
|
*/
|
||||||
for (q = km->loop_head; !(p > q && p < q->ptr); q = q->ptr)
|
for (q = km->loop_head; !(p > q && p < q->ptr); q = q->ptr)
|
||||||
if (q >= q->ptr && (p > q || p < q->ptr)) break;
|
if (q >= q->ptr && (p > q || p < q->ptr)) break;
|
||||||
if (p + p->size == q->ptr) { /* two adjacent blocks, merge p and q->ptr (the 2nd and 4th cases) */
|
if (p + p->size == q->ptr) { /* two adjacent blocks, merge p and q->ptr (the 2nd and 4th cases) */
|
||||||
p->size += q->ptr->size;
|
p->size += q->ptr->size;
|
||||||
p->ptr = q->ptr->ptr;
|
p->ptr = q->ptr->ptr;
|
||||||
} else if (p + p->size > q->ptr && q->ptr >= p) {
|
} else if (p + p->size > q->ptr && q->ptr >= p) {
|
||||||
panic("[kfree] The end of the allocated block enters a free block.");
|
panic("[kfree] The end of the allocated block enters a free block.");
|
||||||
} else p->ptr = q->ptr; /* backup q->ptr */
|
} else p->ptr = q->ptr; /* backup q->ptr */
|
||||||
|
|
||||||
if (q + q->size == p) { /* two adjacent blocks, merge q and p (the other two cases) */
|
if (q + q->size == p) { /* two adjacent blocks, merge q and p (the other two cases) */
|
||||||
q->size += p->size;
|
q->size += p->size;
|
||||||
q->ptr = p->ptr;
|
q->ptr = p->ptr;
|
||||||
km->loop_head = q;
|
km->loop_head = q;
|
||||||
} else if (q + q->size > p && p >= q) {
|
} else if (q + q->size > p && p >= q) {
|
||||||
panic("[kfree] The end of a free block enters the allocated block.");
|
panic("[kfree] The end of a free block enters the allocated block.");
|
||||||
} else km->loop_head = p, q->ptr = p; /* in two cores, cannot be merged; create a new block in the list */
|
} else km->loop_head = p, q->ptr = p; /* in two cores, cannot be merged; create a new block in the list */
|
||||||
}
|
}
|
||||||
|
|
||||||
void *kmalloc(void *_km, size_t n_bytes)
|
void *kmalloc(void *_km, size_t n_bytes)
|
||||||
{
|
{
|
||||||
kmem_t *km = (kmem_t*)_km;
|
kmem_t *km = (kmem_t*)_km;
|
||||||
size_t n_units;
|
size_t n_units;
|
||||||
header_t *p, *q;
|
header_t *p, *q;
|
||||||
|
|
||||||
if (n_bytes == 0) return 0;
|
if (n_bytes == 0) return 0;
|
||||||
if (km == NULL) return malloc(n_bytes);
|
if (km == NULL) return malloc(n_bytes);
|
||||||
n_units = (n_bytes + sizeof(size_t) + sizeof(header_t) - 1) / sizeof(header_t); /* header+n_bytes requires at least this number of units */
|
n_units = (n_bytes + sizeof(size_t) + sizeof(header_t) - 1) / sizeof(header_t); /* header+n_bytes requires at least this number of units */
|
||||||
|
|
||||||
if (!(q = km->loop_head)) /* the first time when kmalloc() is called, intialize it */
|
if (!(q = km->loop_head)) /* the first time when kmalloc() is called, intialize it */
|
||||||
q = km->loop_head = km->base.ptr = &km->base;
|
q = km->loop_head = km->base.ptr = &km->base;
|
||||||
for (p = q->ptr;; q = p, p = p->ptr) { /* search for a suitable block */
|
for (p = q->ptr;; q = p, p = p->ptr) { /* search for a suitable block */
|
||||||
if (p->size >= n_units) { /* p->size if the size of current block. This line means the current block is large enough. */
|
if (p->size >= n_units) { /* p->size if the size of current block. This line means the current block is large enough. */
|
||||||
if (p->size == n_units) q->ptr = p->ptr; /* no need to split the block */
|
if (p->size == n_units) q->ptr = p->ptr; /* no need to split the block */
|
||||||
else { /* split the block. NB: memory is allocated at the end of the block! */
|
else { /* split the block. NB: memory is allocated at the end of the block! */
|
||||||
p->size -= n_units; /* reduce the size of the free block */
|
p->size -= n_units; /* reduce the size of the free block */
|
||||||
p += p->size; /* p points to the allocated block */
|
p += p->size; /* p points to the allocated block */
|
||||||
*(size_t*)p = n_units; /* set the size */
|
*(size_t*)p = n_units; /* set the size */
|
||||||
}
|
}
|
||||||
km->loop_head = q; /* set the end of chain */
|
km->loop_head = q; /* set the end of chain */
|
||||||
return (size_t*)p + 1;
|
return (size_t*)p + 1;
|
||||||
}
|
}
|
||||||
if (p == km->loop_head) { /* then ask for more "cores" */
|
if (p == km->loop_head) { /* then ask for more "cores" */
|
||||||
if ((p = morecore(km, n_units)) == 0) return 0;
|
if ((p = morecore(km, n_units)) == 0) return 0;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
void *kcalloc(void *_km, size_t count, size_t size)
|
void *kcalloc(void *_km, size_t count, size_t size)
|
||||||
{
|
{
|
||||||
kmem_t *km = (kmem_t*)_km;
|
kmem_t *km = (kmem_t*)_km;
|
||||||
void *p;
|
void *p;
|
||||||
if (size == 0 || count == 0) return 0;
|
if (size == 0 || count == 0) return 0;
|
||||||
if (km == NULL) return calloc(count, size);
|
if (km == NULL) return calloc(count, size);
|
||||||
p = kmalloc(km, count * size);
|
p = kmalloc(km, count * size);
|
||||||
memset(p, 0, count * size);
|
memset(p, 0, count * size);
|
||||||
return p;
|
return p;
|
||||||
}
|
}
|
||||||
|
|
||||||
void *krealloc(void *_km, void *ap, size_t n_bytes) // TODO: this can be made more efficient in principle
|
void *krealloc(void *_km, void *ap, size_t n_bytes) // TODO: this can be made more efficient in principle
|
||||||
{
|
{
|
||||||
kmem_t *km = (kmem_t*)_km;
|
kmem_t *km = (kmem_t*)_km;
|
||||||
size_t cap, *p, *q;
|
size_t cap, *p, *q;
|
||||||
|
|
||||||
if (n_bytes == 0) {
|
if (n_bytes == 0) {
|
||||||
kfree(km, ap); return 0;
|
kfree(km, ap); return 0;
|
||||||
}
|
}
|
||||||
if (km == NULL) return realloc(ap, n_bytes);
|
if (km == NULL) return realloc(ap, n_bytes);
|
||||||
if (ap == NULL) return kmalloc(km, n_bytes);
|
if (ap == NULL) return kmalloc(km, n_bytes);
|
||||||
p = (size_t*)ap - 1;
|
p = (size_t*)ap - 1;
|
||||||
cap = (*p) * sizeof(header_t) - sizeof(size_t);
|
cap = (*p) * sizeof(header_t) - sizeof(size_t);
|
||||||
if (cap >= n_bytes) return ap; /* TODO: this prevents shrinking */
|
if (cap >= n_bytes) return ap; /* TODO: this prevents shrinking */
|
||||||
q = (size_t*)kmalloc(km, n_bytes);
|
q = (size_t*)kmalloc(km, n_bytes);
|
||||||
memcpy(q, ap, cap);
|
memcpy(q, ap, cap);
|
||||||
kfree(km, ap);
|
kfree(km, ap);
|
||||||
return q;
|
return q;
|
||||||
}
|
}
|
||||||
|
|
||||||
void km_stat(const void *_km, km_stat_t *s)
|
void km_stat(const void *_km, km_stat_t *s)
|
||||||
{
|
{
|
||||||
kmem_t *km = (kmem_t*)_km;
|
kmem_t *km = (kmem_t*)_km;
|
||||||
header_t *p;
|
header_t *p;
|
||||||
memset(s, 0, sizeof(km_stat_t));
|
memset(s, 0, sizeof(km_stat_t));
|
||||||
if (km == NULL || km->loop_head == NULL) return;
|
if (km == NULL || km->loop_head == NULL) return;
|
||||||
for (p = km->loop_head;; p = p->ptr) {
|
for (p = km->loop_head;; p = p->ptr) {
|
||||||
s->available += p->size * sizeof(header_t);
|
s->available += p->size * sizeof(header_t);
|
||||||
if (p->size != 0) ++s->n_blocks; /* &kmem_t::base is always one of the cores. It is zero-sized. */
|
if (p->size != 0) ++s->n_blocks; /* &kmem_t::base is always one of the cores. It is zero-sized. */
|
||||||
if (p->ptr > p && p + p->size > p->ptr)
|
if (p->ptr > p && p + p->size > p->ptr)
|
||||||
panic("[km_stat] The end of a free block enters another free block.");
|
panic("[km_stat] The end of a free block enters another free block.");
|
||||||
if (p->ptr == km->loop_head) break;
|
if (p->ptr == km->loop_head) break;
|
||||||
}
|
}
|
||||||
for (p = km->core_head; p != NULL; p = p->ptr) {
|
for (p = km->core_head; p != NULL; p = p->ptr) {
|
||||||
size_t size = p->size * sizeof(header_t);
|
size_t size = p->size * sizeof(header_t);
|
||||||
++s->n_cores;
|
++s->n_cores;
|
||||||
s->capacity += size;
|
s->capacity += size;
|
||||||
s->largest = s->largest > size? s->largest : size;
|
s->largest = s->largest > size? s->largest : size;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,107 +1,107 @@
|
|||||||
#ifndef _KALLOC_H_
|
#ifndef _KALLOC_H_
|
||||||
#define _KALLOC_H_
|
#define _KALLOC_H_
|
||||||
|
|
||||||
#include <stddef.h> /* for size_t */
|
#include <stddef.h> /* for size_t */
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
extern "C" {
|
extern "C" {
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
size_t capacity, available, n_blocks, n_cores, largest;
|
size_t capacity, available, n_blocks, n_cores, largest;
|
||||||
} km_stat_t;
|
} km_stat_t;
|
||||||
|
|
||||||
void *kmalloc(void *km, size_t size);
|
void *kmalloc(void *km, size_t size);
|
||||||
void *krealloc(void *km, void *ptr, size_t size);
|
void *krealloc(void *km, void *ptr, size_t size);
|
||||||
void *kcalloc(void *km, size_t count, size_t size);
|
void *kcalloc(void *km, size_t count, size_t size);
|
||||||
void kfree(void *km, void *ptr);
|
void kfree(void *km, void *ptr);
|
||||||
|
|
||||||
void *km_init(void);
|
void *km_init(void);
|
||||||
void *km_init2(void *km_par, size_t min_core_size);
|
void *km_init2(void *km_par, size_t min_core_size);
|
||||||
void km_destroy(void *km);
|
void km_destroy(void *km);
|
||||||
void km_stat(const void *_km, km_stat_t *s);
|
void km_stat(const void *_km, km_stat_t *s);
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#define KMALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kmalloc((km), (len) * sizeof(*(ptr))))
|
#define KMALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kmalloc((km), (len) * sizeof(*(ptr))))
|
||||||
#define KCALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kcalloc((km), (len), sizeof(*(ptr))))
|
#define KCALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))kcalloc((km), (len), sizeof(*(ptr))))
|
||||||
#define KREALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))krealloc((km), (ptr), (len) * sizeof(*(ptr))))
|
#define KREALLOC(km, ptr, len) ((ptr) = (__typeof__(ptr))krealloc((km), (ptr), (len) * sizeof(*(ptr))))
|
||||||
|
|
||||||
#define KEXPAND(km, a, m) do { \
|
#define KEXPAND(km, a, m) do { \
|
||||||
(m) = (m) >= 4? (m) + ((m)>>1) : 16; \
|
(m) = (m) >= 4? (m) + ((m)>>1) : 16; \
|
||||||
KREALLOC((km), (a), (m)); \
|
KREALLOC((km), (a), (m)); \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#define kv_resize_km(km, type, v, s) do { \
|
#define kv_resize_km(km, type, v, s) do { \
|
||||||
if ((v).m < (s)) { \
|
if ((v).m < (s)) { \
|
||||||
(v).m = (s); \
|
(v).m = (s); \
|
||||||
kv_roundup32((v).m); \
|
kv_roundup32((v).m); \
|
||||||
KREALLOC((km), (v).a, (v).m); \
|
KREALLOC((km), (v).a, (v).m); \
|
||||||
} \
|
} \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#define kv_copy_km(km, type, v1, v0) do { \
|
#define kv_copy_km(km, type, v1, v0) do { \
|
||||||
if ((v1).m < (v0).n) kv_resize_km((km), type, v1, (v0).n); \
|
if ((v1).m < (v0).n) kv_resize_km((km), type, v1, (v0).n); \
|
||||||
(v1).n = (v0).n; \
|
(v1).n = (v0).n; \
|
||||||
memcpy((v1).a, (v0).a, sizeof(type) * (v0).n); \
|
memcpy((v1).a, (v0).a, sizeof(type) * (v0).n); \
|
||||||
} while (0) \
|
} while (0) \
|
||||||
|
|
||||||
#define kv_push_km(km, type, v, x) do { \
|
#define kv_push_km(km, type, v, x) do { \
|
||||||
if ((v).n == (v).m) { \
|
if ((v).n == (v).m) { \
|
||||||
(v).m = (v).m? (v).m<<1 : 2; \
|
(v).m = (v).m? (v).m<<1 : 2; \
|
||||||
KREALLOC((km), (v).a, (v).m); \
|
KREALLOC((km), (v).a, (v).m); \
|
||||||
} \
|
} \
|
||||||
(v).a[(v).n++] = (x); \
|
(v).a[(v).n++] = (x); \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#define kv_pushp_km(km, type, v, p) do { \
|
#define kv_pushp_km(km, type, v, p) do { \
|
||||||
if ((v).n == (v).m) { \
|
if ((v).n == (v).m) { \
|
||||||
(v).m = (v).m? (v).m<<1 : 2; \
|
(v).m = (v).m? (v).m<<1 : 2; \
|
||||||
KREALLOC((km), (v).a, (v).m); \
|
KREALLOC((km), (v).a, (v).m); \
|
||||||
} \
|
} \
|
||||||
*(p) = &(v).a[(v).n++]; \
|
*(p) = &(v).a[(v).n++]; \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#ifndef klib_unused
|
#ifndef klib_unused
|
||||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
||||||
#define klib_unused __attribute__ ((__unused__))
|
#define klib_unused __attribute__ ((__unused__))
|
||||||
#else
|
#else
|
||||||
#define klib_unused
|
#define klib_unused
|
||||||
#endif
|
#endif
|
||||||
#endif /* klib_unused */
|
#endif /* klib_unused */
|
||||||
|
|
||||||
// adapted from klist.h
|
// adapted from klist.h
|
||||||
#define KALLOC_POOL_INIT2(SCOPE, name, kmptype_t) \
|
#define KALLOC_POOL_INIT2(SCOPE, name, kmptype_t) \
|
||||||
typedef struct { \
|
typedef struct { \
|
||||||
size_t cnt, n, max; \
|
size_t cnt, n, max; \
|
||||||
kmptype_t **buf; \
|
kmptype_t **buf; \
|
||||||
void *km; \
|
void *km; \
|
||||||
} kmp_##name##_t; \
|
} kmp_##name##_t; \
|
||||||
SCOPE kmp_##name##_t *kmp_init_##name(void *km) { \
|
SCOPE kmp_##name##_t *kmp_init_##name(void *km) { \
|
||||||
kmp_##name##_t *mp; \
|
kmp_##name##_t *mp; \
|
||||||
KCALLOC(km, mp, 1); \
|
KCALLOC(km, mp, 1); \
|
||||||
mp->km = km; \
|
mp->km = km; \
|
||||||
return mp; \
|
return mp; \
|
||||||
} \
|
} \
|
||||||
SCOPE void kmp_destroy_##name(kmp_##name##_t *mp) { \
|
SCOPE void kmp_destroy_##name(kmp_##name##_t *mp) { \
|
||||||
size_t k; \
|
size_t k; \
|
||||||
for (k = 0; k < mp->n; ++k) kfree(mp->km, mp->buf[k]); \
|
for (k = 0; k < mp->n; ++k) kfree(mp->km, mp->buf[k]); \
|
||||||
kfree(mp->km, mp->buf); kfree(mp->km, mp); \
|
kfree(mp->km, mp->buf); kfree(mp->km, mp); \
|
||||||
} \
|
} \
|
||||||
SCOPE kmptype_t *kmp_alloc_##name(kmp_##name##_t *mp) { \
|
SCOPE kmptype_t *kmp_alloc_##name(kmp_##name##_t *mp) { \
|
||||||
++mp->cnt; \
|
++mp->cnt; \
|
||||||
if (mp->n == 0) return (kmptype_t*)kcalloc(mp->km, 1, sizeof(kmptype_t)); \
|
if (mp->n == 0) return (kmptype_t*)kcalloc(mp->km, 1, sizeof(kmptype_t)); \
|
||||||
return mp->buf[--mp->n]; \
|
return mp->buf[--mp->n]; \
|
||||||
} \
|
} \
|
||||||
SCOPE void kmp_free_##name(kmp_##name##_t *mp, kmptype_t *p) { \
|
SCOPE void kmp_free_##name(kmp_##name##_t *mp, kmptype_t *p) { \
|
||||||
--mp->cnt; \
|
--mp->cnt; \
|
||||||
if (mp->n == mp->max) KEXPAND(mp->km, mp->buf, mp->max); \
|
if (mp->n == mp->max) KEXPAND(mp->km, mp->buf, mp->max); \
|
||||||
mp->buf[mp->n++] = p; \
|
mp->buf[mp->n++] = p; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define KALLOC_POOL_INIT(name, kmptype_t) \
|
#define KALLOC_POOL_INIT(name, kmptype_t) \
|
||||||
KALLOC_POOL_INIT2(static inline klib_unused, name, kmptype_t)
|
KALLOC_POOL_INIT2(static inline klib_unused, name, kmptype_t)
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,414 +1,414 @@
|
|||||||
/* The MIT License
|
/* The MIT License
|
||||||
|
|
||||||
Copyright (c) 2018 by Attractive Chaos <attractor@live.co.uk>
|
Copyright (c) 2018 by Attractive Chaos <attractor@live.co.uk>
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
a copy of this software and associated documentation files (the
|
a copy of this software and associated documentation files (the
|
||||||
"Software"), to deal in the Software without restriction, including
|
"Software"), to deal in the Software without restriction, including
|
||||||
without limitation the rights to use, copy, modify, merge, publish,
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
distribute, sublicense, and/or sell copies of the Software, and to
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
permit persons to whom the Software is furnished to do so, subject to
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
the following conditions:
|
the following conditions:
|
||||||
|
|
||||||
The above copyright notice and this permission notice shall be
|
The above copyright notice and this permission notice shall be
|
||||||
included in all copies or substantial portions of the Software.
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
SOFTWARE.
|
SOFTWARE.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/* An example:
|
/* An example:
|
||||||
|
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include "kavl.h"
|
#include "kavl.h"
|
||||||
|
|
||||||
struct my_node {
|
struct my_node {
|
||||||
char key;
|
char key;
|
||||||
KAVL_HEAD(struct my_node) head;
|
KAVL_HEAD(struct my_node) head;
|
||||||
};
|
};
|
||||||
#define my_cmp(p, q) (((q)->key < (p)->key) - ((p)->key < (q)->key))
|
#define my_cmp(p, q) (((q)->key < (p)->key) - ((p)->key < (q)->key))
|
||||||
KAVL_INIT(my, struct my_node, head, my_cmp)
|
KAVL_INIT(my, struct my_node, head, my_cmp)
|
||||||
|
|
||||||
int main(void) {
|
int main(void) {
|
||||||
const char *str = "MNOLKQOPHIA"; // from wiki, except a duplicate
|
const char *str = "MNOLKQOPHIA"; // from wiki, except a duplicate
|
||||||
struct my_node *root = 0;
|
struct my_node *root = 0;
|
||||||
int i, l = strlen(str);
|
int i, l = strlen(str);
|
||||||
for (i = 0; i < l; ++i) { // insert in the input order
|
for (i = 0; i < l; ++i) { // insert in the input order
|
||||||
struct my_node *q, *p = malloc(sizeof(*p));
|
struct my_node *q, *p = malloc(sizeof(*p));
|
||||||
p->key = str[i];
|
p->key = str[i];
|
||||||
q = kavl_insert(my, &root, p, 0);
|
q = kavl_insert(my, &root, p, 0);
|
||||||
if (p != q) free(p); // if already present, free
|
if (p != q) free(p); // if already present, free
|
||||||
}
|
}
|
||||||
kavl_itr_t(my) itr;
|
kavl_itr_t(my) itr;
|
||||||
kavl_itr_first(my, root, &itr); // place at first
|
kavl_itr_first(my, root, &itr); // place at first
|
||||||
do { // traverse
|
do { // traverse
|
||||||
const struct my_node *p = kavl_at(&itr);
|
const struct my_node *p = kavl_at(&itr);
|
||||||
putchar(p->key);
|
putchar(p->key);
|
||||||
free((void*)p); // free node
|
free((void*)p); // free node
|
||||||
} while (kavl_itr_next(my, &itr));
|
} while (kavl_itr_next(my, &itr));
|
||||||
putchar('\n');
|
putchar('\n');
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#ifndef KAVL_H
|
#ifndef KAVL_H
|
||||||
#define KAVL_H
|
#define KAVL_H
|
||||||
|
|
||||||
#ifdef __STRICT_ANSI__
|
#ifdef __STRICT_ANSI__
|
||||||
#define inline __inline__
|
#define inline __inline__
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#define KAVL_MAX_DEPTH 64
|
#define KAVL_MAX_DEPTH 64
|
||||||
|
|
||||||
#define kavl_size(head, p) ((p)? (p)->head.size : 0)
|
#define kavl_size(head, p) ((p)? (p)->head.size : 0)
|
||||||
#define kavl_size_child(head, q, i) ((q)->head.p[(i)]? (q)->head.p[(i)]->head.size : 0)
|
#define kavl_size_child(head, q, i) ((q)->head.p[(i)]? (q)->head.p[(i)]->head.size : 0)
|
||||||
|
|
||||||
#define KAVL_HEAD(__type) \
|
#define KAVL_HEAD(__type) \
|
||||||
struct { \
|
struct { \
|
||||||
__type *p[2]; \
|
__type *p[2]; \
|
||||||
signed char balance; /* balance factor */ \
|
signed char balance; /* balance factor */ \
|
||||||
unsigned size; /* #elements in subtree */ \
|
unsigned size; /* #elements in subtree */ \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define __KAVL_FIND(suf, __scope, __type, __head, __cmp) \
|
#define __KAVL_FIND(suf, __scope, __type, __head, __cmp) \
|
||||||
__scope __type *kavl_find_##suf(const __type *root, const __type *x, unsigned *cnt_) { \
|
__scope __type *kavl_find_##suf(const __type *root, const __type *x, unsigned *cnt_) { \
|
||||||
const __type *p = root; \
|
const __type *p = root; \
|
||||||
unsigned cnt = 0; \
|
unsigned cnt = 0; \
|
||||||
while (p != 0) { \
|
while (p != 0) { \
|
||||||
int cmp; \
|
int cmp; \
|
||||||
cmp = __cmp(x, p); \
|
cmp = __cmp(x, p); \
|
||||||
if (cmp >= 0) cnt += kavl_size_child(__head, p, 0) + 1; \
|
if (cmp >= 0) cnt += kavl_size_child(__head, p, 0) + 1; \
|
||||||
if (cmp < 0) p = p->__head.p[0]; \
|
if (cmp < 0) p = p->__head.p[0]; \
|
||||||
else if (cmp > 0) p = p->__head.p[1]; \
|
else if (cmp > 0) p = p->__head.p[1]; \
|
||||||
else break; \
|
else break; \
|
||||||
} \
|
} \
|
||||||
if (cnt_) *cnt_ = cnt; \
|
if (cnt_) *cnt_ = cnt; \
|
||||||
return (__type*)p; \
|
return (__type*)p; \
|
||||||
} \
|
} \
|
||||||
__scope __type *kavl_interval_##suf(const __type *root, const __type *x, __type **lower, __type **upper) { \
|
__scope __type *kavl_interval_##suf(const __type *root, const __type *x, __type **lower, __type **upper) { \
|
||||||
const __type *p = root, *l = 0, *u = 0; \
|
const __type *p = root, *l = 0, *u = 0; \
|
||||||
while (p != 0) { \
|
while (p != 0) { \
|
||||||
int cmp; \
|
int cmp; \
|
||||||
cmp = __cmp(x, p); \
|
cmp = __cmp(x, p); \
|
||||||
if (cmp < 0) u = p, p = p->__head.p[0]; \
|
if (cmp < 0) u = p, p = p->__head.p[0]; \
|
||||||
else if (cmp > 0) l = p, p = p->__head.p[1]; \
|
else if (cmp > 0) l = p, p = p->__head.p[1]; \
|
||||||
else { l = u = p; break; } \
|
else { l = u = p; break; } \
|
||||||
} \
|
} \
|
||||||
if (lower) *lower = (__type*)l; \
|
if (lower) *lower = (__type*)l; \
|
||||||
if (upper) *upper = (__type*)u; \
|
if (upper) *upper = (__type*)u; \
|
||||||
return (__type*)p; \
|
return (__type*)p; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define __KAVL_ROTATE(suf, __type, __head) \
|
#define __KAVL_ROTATE(suf, __type, __head) \
|
||||||
/* one rotation: (a,(b,c)q)p => ((a,b)p,c)q */ \
|
/* one rotation: (a,(b,c)q)p => ((a,b)p,c)q */ \
|
||||||
static inline __type *kavl_rotate1_##suf(__type *p, int dir) { /* dir=0 to left; dir=1 to right */ \
|
static inline __type *kavl_rotate1_##suf(__type *p, int dir) { /* dir=0 to left; dir=1 to right */ \
|
||||||
int opp = 1 - dir; /* opposite direction */ \
|
int opp = 1 - dir; /* opposite direction */ \
|
||||||
__type *q = p->__head.p[opp]; \
|
__type *q = p->__head.p[opp]; \
|
||||||
unsigned size_p = p->__head.size; \
|
unsigned size_p = p->__head.size; \
|
||||||
p->__head.size -= q->__head.size - kavl_size_child(__head, q, dir); \
|
p->__head.size -= q->__head.size - kavl_size_child(__head, q, dir); \
|
||||||
q->__head.size = size_p; \
|
q->__head.size = size_p; \
|
||||||
p->__head.p[opp] = q->__head.p[dir]; \
|
p->__head.p[opp] = q->__head.p[dir]; \
|
||||||
q->__head.p[dir] = p; \
|
q->__head.p[dir] = p; \
|
||||||
return q; \
|
return q; \
|
||||||
} \
|
} \
|
||||||
/* two consecutive rotations: (a,((b,c)r,d)q)p => ((a,b)p,(c,d)q)r */ \
|
/* two consecutive rotations: (a,((b,c)r,d)q)p => ((a,b)p,(c,d)q)r */ \
|
||||||
static inline __type *kavl_rotate2_##suf(__type *p, int dir) { \
|
static inline __type *kavl_rotate2_##suf(__type *p, int dir) { \
|
||||||
int b1, opp = 1 - dir; \
|
int b1, opp = 1 - dir; \
|
||||||
__type *q = p->__head.p[opp], *r = q->__head.p[dir]; \
|
__type *q = p->__head.p[opp], *r = q->__head.p[dir]; \
|
||||||
unsigned size_x_dir = kavl_size_child(__head, r, dir); \
|
unsigned size_x_dir = kavl_size_child(__head, r, dir); \
|
||||||
r->__head.size = p->__head.size; \
|
r->__head.size = p->__head.size; \
|
||||||
p->__head.size -= q->__head.size - size_x_dir; \
|
p->__head.size -= q->__head.size - size_x_dir; \
|
||||||
q->__head.size -= size_x_dir + 1; \
|
q->__head.size -= size_x_dir + 1; \
|
||||||
p->__head.p[opp] = r->__head.p[dir]; \
|
p->__head.p[opp] = r->__head.p[dir]; \
|
||||||
r->__head.p[dir] = p; \
|
r->__head.p[dir] = p; \
|
||||||
q->__head.p[dir] = r->__head.p[opp]; \
|
q->__head.p[dir] = r->__head.p[opp]; \
|
||||||
r->__head.p[opp] = q; \
|
r->__head.p[opp] = q; \
|
||||||
b1 = dir == 0? +1 : -1; \
|
b1 = dir == 0? +1 : -1; \
|
||||||
if (r->__head.balance == b1) q->__head.balance = 0, p->__head.balance = -b1; \
|
if (r->__head.balance == b1) q->__head.balance = 0, p->__head.balance = -b1; \
|
||||||
else if (r->__head.balance == 0) q->__head.balance = p->__head.balance = 0; \
|
else if (r->__head.balance == 0) q->__head.balance = p->__head.balance = 0; \
|
||||||
else q->__head.balance = b1, p->__head.balance = 0; \
|
else q->__head.balance = b1, p->__head.balance = 0; \
|
||||||
r->__head.balance = 0; \
|
r->__head.balance = 0; \
|
||||||
return r; \
|
return r; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define __KAVL_INSERT(suf, __scope, __type, __head, __cmp) \
|
#define __KAVL_INSERT(suf, __scope, __type, __head, __cmp) \
|
||||||
__scope __type *kavl_insert_##suf(__type **root_, __type *x, unsigned *cnt_) { \
|
__scope __type *kavl_insert_##suf(__type **root_, __type *x, unsigned *cnt_) { \
|
||||||
unsigned char stack[KAVL_MAX_DEPTH]; \
|
unsigned char stack[KAVL_MAX_DEPTH]; \
|
||||||
__type *path[KAVL_MAX_DEPTH]; \
|
__type *path[KAVL_MAX_DEPTH]; \
|
||||||
__type *bp, *bq; \
|
__type *bp, *bq; \
|
||||||
__type *p, *q, *r = 0; /* _r_ is potentially the new root */ \
|
__type *p, *q, *r = 0; /* _r_ is potentially the new root */ \
|
||||||
int i, which = 0, top, b1, path_len; \
|
int i, which = 0, top, b1, path_len; \
|
||||||
unsigned cnt = 0; \
|
unsigned cnt = 0; \
|
||||||
bp = *root_, bq = 0; \
|
bp = *root_, bq = 0; \
|
||||||
/* find the insertion location */ \
|
/* find the insertion location */ \
|
||||||
for (p = bp, q = bq, top = path_len = 0; p; q = p, p = p->__head.p[which]) { \
|
for (p = bp, q = bq, top = path_len = 0; p; q = p, p = p->__head.p[which]) { \
|
||||||
int cmp; \
|
int cmp; \
|
||||||
cmp = __cmp(x, p); \
|
cmp = __cmp(x, p); \
|
||||||
if (cmp >= 0) cnt += kavl_size_child(__head, p, 0) + 1; \
|
if (cmp >= 0) cnt += kavl_size_child(__head, p, 0) + 1; \
|
||||||
if (cmp == 0) { \
|
if (cmp == 0) { \
|
||||||
if (cnt_) *cnt_ = cnt; \
|
if (cnt_) *cnt_ = cnt; \
|
||||||
return p; \
|
return p; \
|
||||||
} \
|
} \
|
||||||
if (p->__head.balance != 0) \
|
if (p->__head.balance != 0) \
|
||||||
bq = q, bp = p, top = 0; \
|
bq = q, bp = p, top = 0; \
|
||||||
stack[top++] = which = (cmp > 0); \
|
stack[top++] = which = (cmp > 0); \
|
||||||
path[path_len++] = p; \
|
path[path_len++] = p; \
|
||||||
} \
|
} \
|
||||||
if (cnt_) *cnt_ = cnt; \
|
if (cnt_) *cnt_ = cnt; \
|
||||||
x->__head.balance = 0, x->__head.size = 1, x->__head.p[0] = x->__head.p[1] = 0; \
|
x->__head.balance = 0, x->__head.size = 1, x->__head.p[0] = x->__head.p[1] = 0; \
|
||||||
if (q == 0) *root_ = x; \
|
if (q == 0) *root_ = x; \
|
||||||
else q->__head.p[which] = x; \
|
else q->__head.p[which] = x; \
|
||||||
if (bp == 0) return x; \
|
if (bp == 0) return x; \
|
||||||
for (i = 0; i < path_len; ++i) ++path[i]->__head.size; \
|
for (i = 0; i < path_len; ++i) ++path[i]->__head.size; \
|
||||||
for (p = bp, top = 0; p != x; p = p->__head.p[stack[top]], ++top) /* update balance factors */ \
|
for (p = bp, top = 0; p != x; p = p->__head.p[stack[top]], ++top) /* update balance factors */ \
|
||||||
if (stack[top] == 0) --p->__head.balance; \
|
if (stack[top] == 0) --p->__head.balance; \
|
||||||
else ++p->__head.balance; \
|
else ++p->__head.balance; \
|
||||||
if (bp->__head.balance > -2 && bp->__head.balance < 2) return x; /* no re-balance needed */ \
|
if (bp->__head.balance > -2 && bp->__head.balance < 2) return x; /* no re-balance needed */ \
|
||||||
/* re-balance */ \
|
/* re-balance */ \
|
||||||
which = (bp->__head.balance < 0); \
|
which = (bp->__head.balance < 0); \
|
||||||
b1 = which == 0? +1 : -1; \
|
b1 = which == 0? +1 : -1; \
|
||||||
q = bp->__head.p[1 - which]; \
|
q = bp->__head.p[1 - which]; \
|
||||||
if (q->__head.balance == b1) { \
|
if (q->__head.balance == b1) { \
|
||||||
r = kavl_rotate1_##suf(bp, which); \
|
r = kavl_rotate1_##suf(bp, which); \
|
||||||
q->__head.balance = bp->__head.balance = 0; \
|
q->__head.balance = bp->__head.balance = 0; \
|
||||||
} else r = kavl_rotate2_##suf(bp, which); \
|
} else r = kavl_rotate2_##suf(bp, which); \
|
||||||
if (bq == 0) *root_ = r; \
|
if (bq == 0) *root_ = r; \
|
||||||
else bq->__head.p[bp != bq->__head.p[0]] = r; \
|
else bq->__head.p[bp != bq->__head.p[0]] = r; \
|
||||||
return x; \
|
return x; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define __KAVL_ERASE(suf, __scope, __type, __head, __cmp) \
|
#define __KAVL_ERASE(suf, __scope, __type, __head, __cmp) \
|
||||||
__scope __type *kavl_erase_##suf(__type **root_, const __type *x, unsigned *cnt_) { \
|
__scope __type *kavl_erase_##suf(__type **root_, const __type *x, unsigned *cnt_) { \
|
||||||
__type *p, *path[KAVL_MAX_DEPTH], fake; \
|
__type *p, *path[KAVL_MAX_DEPTH], fake; \
|
||||||
unsigned char dir[KAVL_MAX_DEPTH]; \
|
unsigned char dir[KAVL_MAX_DEPTH]; \
|
||||||
int i, d = 0, cmp; \
|
int i, d = 0, cmp; \
|
||||||
unsigned cnt = 0; \
|
unsigned cnt = 0; \
|
||||||
fake.__head.p[0] = *root_, fake.__head.p[1] = 0; \
|
fake.__head.p[0] = *root_, fake.__head.p[1] = 0; \
|
||||||
if (cnt_) *cnt_ = 0; \
|
if (cnt_) *cnt_ = 0; \
|
||||||
if (x) { \
|
if (x) { \
|
||||||
for (cmp = -1, p = &fake; cmp; cmp = __cmp(x, p)) { \
|
for (cmp = -1, p = &fake; cmp; cmp = __cmp(x, p)) { \
|
||||||
int which = (cmp > 0); \
|
int which = (cmp > 0); \
|
||||||
if (cmp > 0) cnt += kavl_size_child(__head, p, 0) + 1; \
|
if (cmp > 0) cnt += kavl_size_child(__head, p, 0) + 1; \
|
||||||
dir[d] = which; \
|
dir[d] = which; \
|
||||||
path[d++] = p; \
|
path[d++] = p; \
|
||||||
p = p->__head.p[which]; \
|
p = p->__head.p[which]; \
|
||||||
if (p == 0) { \
|
if (p == 0) { \
|
||||||
if (cnt_) *cnt_ = 0; \
|
if (cnt_) *cnt_ = 0; \
|
||||||
return 0; \
|
return 0; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
cnt += kavl_size_child(__head, p, 0) + 1; /* because p==x is not counted */ \
|
cnt += kavl_size_child(__head, p, 0) + 1; /* because p==x is not counted */ \
|
||||||
} else { \
|
} else { \
|
||||||
for (p = &fake, cnt = 1; p; p = p->__head.p[0]) \
|
for (p = &fake, cnt = 1; p; p = p->__head.p[0]) \
|
||||||
dir[d] = 0, path[d++] = p; \
|
dir[d] = 0, path[d++] = p; \
|
||||||
p = path[--d]; \
|
p = path[--d]; \
|
||||||
} \
|
} \
|
||||||
if (cnt_) *cnt_ = cnt; \
|
if (cnt_) *cnt_ = cnt; \
|
||||||
for (i = 1; i < d; ++i) --path[i]->__head.size; \
|
for (i = 1; i < d; ++i) --path[i]->__head.size; \
|
||||||
if (p->__head.p[1] == 0) { /* ((1,.)2,3)4 => (1,3)4; p=2 */ \
|
if (p->__head.p[1] == 0) { /* ((1,.)2,3)4 => (1,3)4; p=2 */ \
|
||||||
path[d-1]->__head.p[dir[d-1]] = p->__head.p[0]; \
|
path[d-1]->__head.p[dir[d-1]] = p->__head.p[0]; \
|
||||||
} else { \
|
} else { \
|
||||||
__type *q = p->__head.p[1]; \
|
__type *q = p->__head.p[1]; \
|
||||||
if (q->__head.p[0] == 0) { /* ((1,2)3,4)5 => ((1)2,4)5; p=3 */ \
|
if (q->__head.p[0] == 0) { /* ((1,2)3,4)5 => ((1)2,4)5; p=3 */ \
|
||||||
q->__head.p[0] = p->__head.p[0]; \
|
q->__head.p[0] = p->__head.p[0]; \
|
||||||
q->__head.balance = p->__head.balance; \
|
q->__head.balance = p->__head.balance; \
|
||||||
path[d-1]->__head.p[dir[d-1]] = q; \
|
path[d-1]->__head.p[dir[d-1]] = q; \
|
||||||
path[d] = q, dir[d++] = 1; \
|
path[d] = q, dir[d++] = 1; \
|
||||||
q->__head.size = p->__head.size - 1; \
|
q->__head.size = p->__head.size - 1; \
|
||||||
} else { /* ((1,((.,2)3,4)5)6,7)8 => ((1,(2,4)5)3,7)8; p=6 */ \
|
} else { /* ((1,((.,2)3,4)5)6,7)8 => ((1,(2,4)5)3,7)8; p=6 */ \
|
||||||
__type *r; \
|
__type *r; \
|
||||||
int e = d++; /* backup _d_ */\
|
int e = d++; /* backup _d_ */\
|
||||||
for (;;) { \
|
for (;;) { \
|
||||||
dir[d] = 0; \
|
dir[d] = 0; \
|
||||||
path[d++] = q; \
|
path[d++] = q; \
|
||||||
r = q->__head.p[0]; \
|
r = q->__head.p[0]; \
|
||||||
if (r->__head.p[0] == 0) break; \
|
if (r->__head.p[0] == 0) break; \
|
||||||
q = r; \
|
q = r; \
|
||||||
} \
|
} \
|
||||||
r->__head.p[0] = p->__head.p[0]; \
|
r->__head.p[0] = p->__head.p[0]; \
|
||||||
q->__head.p[0] = r->__head.p[1]; \
|
q->__head.p[0] = r->__head.p[1]; \
|
||||||
r->__head.p[1] = p->__head.p[1]; \
|
r->__head.p[1] = p->__head.p[1]; \
|
||||||
r->__head.balance = p->__head.balance; \
|
r->__head.balance = p->__head.balance; \
|
||||||
path[e-1]->__head.p[dir[e-1]] = r; \
|
path[e-1]->__head.p[dir[e-1]] = r; \
|
||||||
path[e] = r, dir[e] = 1; \
|
path[e] = r, dir[e] = 1; \
|
||||||
for (i = e + 1; i < d; ++i) --path[i]->__head.size; \
|
for (i = e + 1; i < d; ++i) --path[i]->__head.size; \
|
||||||
r->__head.size = p->__head.size - 1; \
|
r->__head.size = p->__head.size - 1; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
while (--d > 0) { \
|
while (--d > 0) { \
|
||||||
__type *q = path[d]; \
|
__type *q = path[d]; \
|
||||||
int which, other, b1 = 1, b2 = 2; \
|
int which, other, b1 = 1, b2 = 2; \
|
||||||
which = dir[d], other = 1 - which; \
|
which = dir[d], other = 1 - which; \
|
||||||
if (which) b1 = -b1, b2 = -b2; \
|
if (which) b1 = -b1, b2 = -b2; \
|
||||||
q->__head.balance += b1; \
|
q->__head.balance += b1; \
|
||||||
if (q->__head.balance == b1) break; \
|
if (q->__head.balance == b1) break; \
|
||||||
else if (q->__head.balance == b2) { \
|
else if (q->__head.balance == b2) { \
|
||||||
__type *r = q->__head.p[other]; \
|
__type *r = q->__head.p[other]; \
|
||||||
if (r->__head.balance == -b1) { \
|
if (r->__head.balance == -b1) { \
|
||||||
path[d-1]->__head.p[dir[d-1]] = kavl_rotate2_##suf(q, which); \
|
path[d-1]->__head.p[dir[d-1]] = kavl_rotate2_##suf(q, which); \
|
||||||
} else { \
|
} else { \
|
||||||
path[d-1]->__head.p[dir[d-1]] = kavl_rotate1_##suf(q, which); \
|
path[d-1]->__head.p[dir[d-1]] = kavl_rotate1_##suf(q, which); \
|
||||||
if (r->__head.balance == 0) { \
|
if (r->__head.balance == 0) { \
|
||||||
r->__head.balance = -b1; \
|
r->__head.balance = -b1; \
|
||||||
q->__head.balance = b1; \
|
q->__head.balance = b1; \
|
||||||
break; \
|
break; \
|
||||||
} else r->__head.balance = q->__head.balance = 0; \
|
} else r->__head.balance = q->__head.balance = 0; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
*root_ = fake.__head.p[0]; \
|
*root_ = fake.__head.p[0]; \
|
||||||
return p; \
|
return p; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define kavl_free(__type, __head, __root, __free) do { \
|
#define kavl_free(__type, __head, __root, __free) do { \
|
||||||
__type *_p, *_q; \
|
__type *_p, *_q; \
|
||||||
for (_p = __root; _p; _p = _q) { \
|
for (_p = __root; _p; _p = _q) { \
|
||||||
if (_p->__head.p[0] == 0) { \
|
if (_p->__head.p[0] == 0) { \
|
||||||
_q = _p->__head.p[1]; \
|
_q = _p->__head.p[1]; \
|
||||||
__free(_p); \
|
__free(_p); \
|
||||||
} else { \
|
} else { \
|
||||||
_q = _p->__head.p[0]; \
|
_q = _p->__head.p[0]; \
|
||||||
_p->__head.p[0] = _q->__head.p[1]; \
|
_p->__head.p[0] = _q->__head.p[1]; \
|
||||||
_q->__head.p[1] = _p; \
|
_q->__head.p[1] = _p; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#define __KAVL_ITR(suf, __scope, __type, __head, __cmp) \
|
#define __KAVL_ITR(suf, __scope, __type, __head, __cmp) \
|
||||||
struct kavl_itr_##suf { \
|
struct kavl_itr_##suf { \
|
||||||
const __type *stack[KAVL_MAX_DEPTH], **top; \
|
const __type *stack[KAVL_MAX_DEPTH], **top; \
|
||||||
}; \
|
}; \
|
||||||
__scope void kavl_itr_first_##suf(const __type *root, struct kavl_itr_##suf *itr) { \
|
__scope void kavl_itr_first_##suf(const __type *root, struct kavl_itr_##suf *itr) { \
|
||||||
const __type *p; \
|
const __type *p; \
|
||||||
for (itr->top = itr->stack - 1, p = root; p; p = p->__head.p[0]) \
|
for (itr->top = itr->stack - 1, p = root; p; p = p->__head.p[0]) \
|
||||||
*++itr->top = p; \
|
*++itr->top = p; \
|
||||||
} \
|
} \
|
||||||
__scope int kavl_itr_find_##suf(const __type *root, const __type *x, struct kavl_itr_##suf *itr) { \
|
__scope int kavl_itr_find_##suf(const __type *root, const __type *x, struct kavl_itr_##suf *itr) { \
|
||||||
const __type *p = root; \
|
const __type *p = root; \
|
||||||
itr->top = itr->stack - 1; \
|
itr->top = itr->stack - 1; \
|
||||||
while (p != 0) { \
|
while (p != 0) { \
|
||||||
int cmp; \
|
int cmp; \
|
||||||
*++itr->top = p; \
|
*++itr->top = p; \
|
||||||
cmp = __cmp(x, p); \
|
cmp = __cmp(x, p); \
|
||||||
if (cmp < 0) p = p->__head.p[0]; \
|
if (cmp < 0) p = p->__head.p[0]; \
|
||||||
else if (cmp > 0) p = p->__head.p[1]; \
|
else if (cmp > 0) p = p->__head.p[1]; \
|
||||||
else break; \
|
else break; \
|
||||||
} \
|
} \
|
||||||
return p? 1 : 0; \
|
return p? 1 : 0; \
|
||||||
} \
|
} \
|
||||||
__scope int kavl_itr_next_bidir_##suf(struct kavl_itr_##suf *itr, int dir) { \
|
__scope int kavl_itr_next_bidir_##suf(struct kavl_itr_##suf *itr, int dir) { \
|
||||||
const __type *p; \
|
const __type *p; \
|
||||||
if (itr->top < itr->stack) return 0; \
|
if (itr->top < itr->stack) return 0; \
|
||||||
dir = !!dir; \
|
dir = !!dir; \
|
||||||
p = (*itr->top)->__head.p[dir]; \
|
p = (*itr->top)->__head.p[dir]; \
|
||||||
if (p) { /* go down */ \
|
if (p) { /* go down */ \
|
||||||
for (; p; p = p->__head.p[!dir]) \
|
for (; p; p = p->__head.p[!dir]) \
|
||||||
*++itr->top = p; \
|
*++itr->top = p; \
|
||||||
return 1; \
|
return 1; \
|
||||||
} else { /* go up */ \
|
} else { /* go up */ \
|
||||||
do { \
|
do { \
|
||||||
p = *itr->top--; \
|
p = *itr->top--; \
|
||||||
} while (itr->top >= itr->stack && p == (*itr->top)->__head.p[dir]); \
|
} while (itr->top >= itr->stack && p == (*itr->top)->__head.p[dir]); \
|
||||||
return itr->top < itr->stack? 0 : 1; \
|
return itr->top < itr->stack? 0 : 1; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Insert a node to the tree
|
* Insert a node to the tree
|
||||||
*
|
*
|
||||||
* @param suf name suffix used in KAVL_INIT()
|
* @param suf name suffix used in KAVL_INIT()
|
||||||
* @param proot pointer to the root of the tree (in/out: root may change)
|
* @param proot pointer to the root of the tree (in/out: root may change)
|
||||||
* @param x node to insert (in)
|
* @param x node to insert (in)
|
||||||
* @param cnt number of nodes smaller than or equal to _x_; can be NULL (out)
|
* @param cnt number of nodes smaller than or equal to _x_; can be NULL (out)
|
||||||
*
|
*
|
||||||
* @return _x_ if not present in the tree, or the node equal to x.
|
* @return _x_ if not present in the tree, or the node equal to x.
|
||||||
*/
|
*/
|
||||||
#define kavl_insert(suf, proot, x, cnt) kavl_insert_##suf(proot, x, cnt)
|
#define kavl_insert(suf, proot, x, cnt) kavl_insert_##suf(proot, x, cnt)
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Find a node in the tree
|
* Find a node in the tree
|
||||||
*
|
*
|
||||||
* @param suf name suffix used in KAVL_INIT()
|
* @param suf name suffix used in KAVL_INIT()
|
||||||
* @param root root of the tree
|
* @param root root of the tree
|
||||||
* @param x node value to find (in)
|
* @param x node value to find (in)
|
||||||
* @param cnt number of nodes smaller than or equal to _x_; can be NULL (out)
|
* @param cnt number of nodes smaller than or equal to _x_; can be NULL (out)
|
||||||
*
|
*
|
||||||
* @return node equal to _x_ if present, or NULL if absent
|
* @return node equal to _x_ if present, or NULL if absent
|
||||||
*/
|
*/
|
||||||
#define kavl_find(suf, root, x, cnt) kavl_find_##suf(root, x, cnt)
|
#define kavl_find(suf, root, x, cnt) kavl_find_##suf(root, x, cnt)
|
||||||
#define kavl_interval(suf, root, x, lower, upper) kavl_interval_##suf(root, x, lower, upper)
|
#define kavl_interval(suf, root, x, lower, upper) kavl_interval_##suf(root, x, lower, upper)
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Delete a node from the tree
|
* Delete a node from the tree
|
||||||
*
|
*
|
||||||
* @param suf name suffix used in KAVL_INIT()
|
* @param suf name suffix used in KAVL_INIT()
|
||||||
* @param proot pointer to the root of the tree (in/out: root may change)
|
* @param proot pointer to the root of the tree (in/out: root may change)
|
||||||
* @param x node value to delete; if NULL, delete the first node (in)
|
* @param x node value to delete; if NULL, delete the first node (in)
|
||||||
*
|
*
|
||||||
* @return node removed from the tree if present, or NULL if absent
|
* @return node removed from the tree if present, or NULL if absent
|
||||||
*/
|
*/
|
||||||
#define kavl_erase(suf, proot, x, cnt) kavl_erase_##suf(proot, x, cnt)
|
#define kavl_erase(suf, proot, x, cnt) kavl_erase_##suf(proot, x, cnt)
|
||||||
#define kavl_erase_first(suf, proot) kavl_erase_##suf(proot, 0, 0)
|
#define kavl_erase_first(suf, proot) kavl_erase_##suf(proot, 0, 0)
|
||||||
|
|
||||||
#define kavl_itr_t(suf) struct kavl_itr_##suf
|
#define kavl_itr_t(suf) struct kavl_itr_##suf
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Place the iterator at the smallest object
|
* Place the iterator at the smallest object
|
||||||
*
|
*
|
||||||
* @param suf name suffix used in KAVL_INIT()
|
* @param suf name suffix used in KAVL_INIT()
|
||||||
* @param root root of the tree
|
* @param root root of the tree
|
||||||
* @param itr iterator
|
* @param itr iterator
|
||||||
*/
|
*/
|
||||||
#define kavl_itr_first(suf, root, itr) kavl_itr_first_##suf(root, itr)
|
#define kavl_itr_first(suf, root, itr) kavl_itr_first_##suf(root, itr)
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Place the iterator at the object equal to or greater than the query
|
* Place the iterator at the object equal to or greater than the query
|
||||||
*
|
*
|
||||||
* @param suf name suffix used in KAVL_INIT()
|
* @param suf name suffix used in KAVL_INIT()
|
||||||
* @param root root of the tree
|
* @param root root of the tree
|
||||||
* @param x query (in)
|
* @param x query (in)
|
||||||
* @param itr iterator (out)
|
* @param itr iterator (out)
|
||||||
*
|
*
|
||||||
* @return 1 if find; 0 otherwise. kavl_at(itr) is NULL if and only if query is
|
* @return 1 if find; 0 otherwise. kavl_at(itr) is NULL if and only if query is
|
||||||
* larger than all objects in the tree
|
* larger than all objects in the tree
|
||||||
*/
|
*/
|
||||||
#define kavl_itr_find(suf, root, x, itr) kavl_itr_find_##suf(root, x, itr)
|
#define kavl_itr_find(suf, root, x, itr) kavl_itr_find_##suf(root, x, itr)
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Move to the next object in order
|
* Move to the next object in order
|
||||||
*
|
*
|
||||||
* @param itr iterator (modified)
|
* @param itr iterator (modified)
|
||||||
*
|
*
|
||||||
* @return 1 if there is a next object; 0 otherwise
|
* @return 1 if there is a next object; 0 otherwise
|
||||||
*/
|
*/
|
||||||
#define kavl_itr_next(suf, itr) kavl_itr_next_bidir_##suf(itr, 1)
|
#define kavl_itr_next(suf, itr) kavl_itr_next_bidir_##suf(itr, 1)
|
||||||
#define kavl_itr_prev(suf, itr) kavl_itr_next_bidir_##suf(itr, 0)
|
#define kavl_itr_prev(suf, itr) kavl_itr_next_bidir_##suf(itr, 0)
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Return the pointer at the iterator
|
* Return the pointer at the iterator
|
||||||
*
|
*
|
||||||
* @param itr iterator
|
* @param itr iterator
|
||||||
*
|
*
|
||||||
* @return pointer if present; NULL otherwise
|
* @return pointer if present; NULL otherwise
|
||||||
*/
|
*/
|
||||||
#define kavl_at(itr) ((itr)->top < (itr)->stack? 0 : *(itr)->top)
|
#define kavl_at(itr) ((itr)->top < (itr)->stack? 0 : *(itr)->top)
|
||||||
|
|
||||||
#define KAVL_INIT2(suf, __scope, __type, __head, __cmp) \
|
#define KAVL_INIT2(suf, __scope, __type, __head, __cmp) \
|
||||||
__KAVL_FIND(suf, __scope, __type, __head, __cmp) \
|
__KAVL_FIND(suf, __scope, __type, __head, __cmp) \
|
||||||
__KAVL_ROTATE(suf, __type, __head) \
|
__KAVL_ROTATE(suf, __type, __head) \
|
||||||
__KAVL_INSERT(suf, __scope, __type, __head, __cmp) \
|
__KAVL_INSERT(suf, __scope, __type, __head, __cmp) \
|
||||||
__KAVL_ERASE(suf, __scope, __type, __head, __cmp) \
|
__KAVL_ERASE(suf, __scope, __type, __head, __cmp) \
|
||||||
__KAVL_ITR(suf, __scope, __type, __head, __cmp)
|
__KAVL_ITR(suf, __scope, __type, __head, __cmp)
|
||||||
|
|
||||||
#define KAVL_INIT(suf, __type, __head, __cmp) \
|
#define KAVL_INIT(suf, __type, __head, __cmp) \
|
||||||
KAVL_INIT2(suf,, __type, __head, __cmp)
|
KAVL_INIT2(suf,, __type, __head, __cmp)
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,128 +1,128 @@
|
|||||||
#ifndef __AC_KDQ_H
|
#ifndef __AC_KDQ_H
|
||||||
#define __AC_KDQ_H
|
#define __AC_KDQ_H
|
||||||
|
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
|
|
||||||
#define __KDQ_TYPE(type) \
|
#define __KDQ_TYPE(type) \
|
||||||
typedef struct { \
|
typedef struct { \
|
||||||
size_t front:58, bits:6, count, mask; \
|
size_t front:58, bits:6, count, mask; \
|
||||||
type *a; \
|
type *a; \
|
||||||
} kdq_##type##_t;
|
} kdq_##type##_t;
|
||||||
|
|
||||||
#define kdq_t(type) kdq_##type##_t
|
#define kdq_t(type) kdq_##type##_t
|
||||||
#define kdq_size(q) ((q)->count)
|
#define kdq_size(q) ((q)->count)
|
||||||
#define kdq_first(q) ((q)->a[(q)->front])
|
#define kdq_first(q) ((q)->a[(q)->front])
|
||||||
#define kdq_last(q) ((q)->a[((q)->front + (q)->count - 1) & (q)->mask])
|
#define kdq_last(q) ((q)->a[((q)->front + (q)->count - 1) & (q)->mask])
|
||||||
#define kdq_at(q, i) ((q)->a[((q)->front + (i)) & (q)->mask])
|
#define kdq_at(q, i) ((q)->a[((q)->front + (i)) & (q)->mask])
|
||||||
|
|
||||||
#define __KDQ_IMPL(type, SCOPE) \
|
#define __KDQ_IMPL(type, SCOPE) \
|
||||||
SCOPE kdq_##type##_t *kdq_init_##type() \
|
SCOPE kdq_##type##_t *kdq_init_##type() \
|
||||||
{ \
|
{ \
|
||||||
kdq_##type##_t *q; \
|
kdq_##type##_t *q; \
|
||||||
q = (kdq_##type##_t*)calloc(1, sizeof(kdq_##type##_t)); \
|
q = (kdq_##type##_t*)calloc(1, sizeof(kdq_##type##_t)); \
|
||||||
q->bits = 2, q->mask = (1ULL<<q->bits) - 1; \
|
q->bits = 2, q->mask = (1ULL<<q->bits) - 1; \
|
||||||
q->a = (type*)malloc((1<<q->bits) * sizeof(type)); \
|
q->a = (type*)malloc((1<<q->bits) * sizeof(type)); \
|
||||||
return q; \
|
return q; \
|
||||||
} \
|
} \
|
||||||
SCOPE void kdq_destroy_##type(kdq_##type##_t *q) \
|
SCOPE void kdq_destroy_##type(kdq_##type##_t *q) \
|
||||||
{ \
|
{ \
|
||||||
if (q == 0) return; \
|
if (q == 0) return; \
|
||||||
free(q->a); free(q); \
|
free(q->a); free(q); \
|
||||||
} \
|
} \
|
||||||
SCOPE int kdq_resize_##type(kdq_##type##_t *q, int new_bits) \
|
SCOPE int kdq_resize_##type(kdq_##type##_t *q, int new_bits) \
|
||||||
{ \
|
{ \
|
||||||
size_t new_size = 1ULL<<new_bits, old_size = 1ULL<<q->bits; \
|
size_t new_size = 1ULL<<new_bits, old_size = 1ULL<<q->bits; \
|
||||||
if (new_size < q->count) { /* not big enough */ \
|
if (new_size < q->count) { /* not big enough */ \
|
||||||
int i; \
|
int i; \
|
||||||
for (i = 0; i < 64; ++i) \
|
for (i = 0; i < 64; ++i) \
|
||||||
if (1ULL<<i > q->count) break; \
|
if (1ULL<<i > q->count) break; \
|
||||||
new_bits = i, new_size = 1ULL<<new_bits; \
|
new_bits = i, new_size = 1ULL<<new_bits; \
|
||||||
} \
|
} \
|
||||||
if (new_bits == q->bits) return q->bits; /* unchanged */ \
|
if (new_bits == q->bits) return q->bits; /* unchanged */ \
|
||||||
if (new_bits > q->bits) q->a = (type*)realloc(q->a, (1ULL<<new_bits) * sizeof(type)); \
|
if (new_bits > q->bits) q->a = (type*)realloc(q->a, (1ULL<<new_bits) * sizeof(type)); \
|
||||||
if (q->front + q->count <= old_size) { /* unwrapped */ \
|
if (q->front + q->count <= old_size) { /* unwrapped */ \
|
||||||
if (q->front + q->count > new_size) /* only happens for shrinking */ \
|
if (q->front + q->count > new_size) /* only happens for shrinking */ \
|
||||||
memmove(q->a, q->a + new_size, (q->front + q->count - new_size) * sizeof(type)); \
|
memmove(q->a, q->a + new_size, (q->front + q->count - new_size) * sizeof(type)); \
|
||||||
} else { /* wrapped */ \
|
} else { /* wrapped */ \
|
||||||
memmove(q->a + (new_size - (old_size - q->front)), q->a + q->front, (old_size - q->front) * sizeof(type)); \
|
memmove(q->a + (new_size - (old_size - q->front)), q->a + q->front, (old_size - q->front) * sizeof(type)); \
|
||||||
q->front = new_size - (old_size - q->front); \
|
q->front = new_size - (old_size - q->front); \
|
||||||
} \
|
} \
|
||||||
q->bits = new_bits, q->mask = (1ULL<<q->bits) - 1; \
|
q->bits = new_bits, q->mask = (1ULL<<q->bits) - 1; \
|
||||||
if (new_bits < q->bits) q->a = (type*)realloc(q->a, (1ULL<<new_bits) * sizeof(type)); \
|
if (new_bits < q->bits) q->a = (type*)realloc(q->a, (1ULL<<new_bits) * sizeof(type)); \
|
||||||
return q->bits; \
|
return q->bits; \
|
||||||
} \
|
} \
|
||||||
SCOPE type *kdq_pushp_##type(kdq_##type##_t *q) \
|
SCOPE type *kdq_pushp_##type(kdq_##type##_t *q) \
|
||||||
{ \
|
{ \
|
||||||
if (q->count == 1ULL<<q->bits) kdq_resize_##type(q, q->bits + 1); \
|
if (q->count == 1ULL<<q->bits) kdq_resize_##type(q, q->bits + 1); \
|
||||||
return &q->a[((q->count++) + q->front) & (q)->mask]; \
|
return &q->a[((q->count++) + q->front) & (q)->mask]; \
|
||||||
} \
|
} \
|
||||||
SCOPE void kdq_push_##type(kdq_##type##_t *q, type v) \
|
SCOPE void kdq_push_##type(kdq_##type##_t *q, type v) \
|
||||||
{ \
|
{ \
|
||||||
if (q->count == 1ULL<<q->bits) kdq_resize_##type(q, q->bits + 1); \
|
if (q->count == 1ULL<<q->bits) kdq_resize_##type(q, q->bits + 1); \
|
||||||
q->a[((q->count++) + q->front) & (q)->mask] = v; \
|
q->a[((q->count++) + q->front) & (q)->mask] = v; \
|
||||||
} \
|
} \
|
||||||
SCOPE type *kdq_unshiftp_##type(kdq_##type##_t *q) \
|
SCOPE type *kdq_unshiftp_##type(kdq_##type##_t *q) \
|
||||||
{ \
|
{ \
|
||||||
if (q->count == 1ULL<<q->bits) kdq_resize_##type(q, q->bits + 1); \
|
if (q->count == 1ULL<<q->bits) kdq_resize_##type(q, q->bits + 1); \
|
||||||
++q->count; \
|
++q->count; \
|
||||||
q->front = q->front? q->front - 1 : (1ULL<<q->bits) - 1; \
|
q->front = q->front? q->front - 1 : (1ULL<<q->bits) - 1; \
|
||||||
return &q->a[q->front]; \
|
return &q->a[q->front]; \
|
||||||
} \
|
} \
|
||||||
SCOPE void kdq_unshift_##type(kdq_##type##_t *q, type v) \
|
SCOPE void kdq_unshift_##type(kdq_##type##_t *q, type v) \
|
||||||
{ \
|
{ \
|
||||||
type *p; \
|
type *p; \
|
||||||
p = kdq_unshiftp_##type(q); \
|
p = kdq_unshiftp_##type(q); \
|
||||||
*p = v; \
|
*p = v; \
|
||||||
} \
|
} \
|
||||||
SCOPE type *kdq_pop_##type(kdq_##type##_t *q) \
|
SCOPE type *kdq_pop_##type(kdq_##type##_t *q) \
|
||||||
{ \
|
{ \
|
||||||
return q->count? &q->a[((--q->count) + q->front) & q->mask] : 0; \
|
return q->count? &q->a[((--q->count) + q->front) & q->mask] : 0; \
|
||||||
} \
|
} \
|
||||||
SCOPE type *kdq_shift_##type(kdq_##type##_t *q) \
|
SCOPE type *kdq_shift_##type(kdq_##type##_t *q) \
|
||||||
{ \
|
{ \
|
||||||
type *d = 0; \
|
type *d = 0; \
|
||||||
if (q->count == 0) return 0; \
|
if (q->count == 0) return 0; \
|
||||||
d = &q->a[q->front++]; \
|
d = &q->a[q->front++]; \
|
||||||
q->front &= q->mask; \
|
q->front &= q->mask; \
|
||||||
--q->count; \
|
--q->count; \
|
||||||
return d; \
|
return d; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define KDQ_INIT2(type, SCOPE) \
|
#define KDQ_INIT2(type, SCOPE) \
|
||||||
__KDQ_TYPE(type) \
|
__KDQ_TYPE(type) \
|
||||||
__KDQ_IMPL(type, SCOPE)
|
__KDQ_IMPL(type, SCOPE)
|
||||||
|
|
||||||
#ifndef klib_unused
|
#ifndef klib_unused
|
||||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
||||||
#define klib_unused __attribute__ ((__unused__))
|
#define klib_unused __attribute__ ((__unused__))
|
||||||
#else
|
#else
|
||||||
#define klib_unused
|
#define klib_unused
|
||||||
#endif
|
#endif
|
||||||
#endif /* klib_unused */
|
#endif /* klib_unused */
|
||||||
|
|
||||||
#define KDQ_INIT(type) KDQ_INIT2(type, static inline klib_unused)
|
#define KDQ_INIT(type) KDQ_INIT2(type, static inline klib_unused)
|
||||||
|
|
||||||
#define KDQ_DECLARE(type) \
|
#define KDQ_DECLARE(type) \
|
||||||
__KDQ_TYPE(type) \
|
__KDQ_TYPE(type) \
|
||||||
kdq_##type##_t *kdq_init_##type(); \
|
kdq_##type##_t *kdq_init_##type(); \
|
||||||
void kdq_destroy_##type(kdq_##type##_t *q); \
|
void kdq_destroy_##type(kdq_##type##_t *q); \
|
||||||
int kdq_resize_##type(kdq_##type##_t *q, int new_bits); \
|
int kdq_resize_##type(kdq_##type##_t *q, int new_bits); \
|
||||||
type *kdq_pushp_##type(kdq_##type##_t *q); \
|
type *kdq_pushp_##type(kdq_##type##_t *q); \
|
||||||
void kdq_push_##type(kdq_##type##_t *q, type v); \
|
void kdq_push_##type(kdq_##type##_t *q, type v); \
|
||||||
type *kdq_unshiftp_##type(kdq_##type##_t *q); \
|
type *kdq_unshiftp_##type(kdq_##type##_t *q); \
|
||||||
void kdq_unshift_##type(kdq_##type##_t *q, type v); \
|
void kdq_unshift_##type(kdq_##type##_t *q, type v); \
|
||||||
type *kdq_pop_##type(kdq_##type##_t *q); \
|
type *kdq_pop_##type(kdq_##type##_t *q); \
|
||||||
type *kdq_shift_##type(kdq_##type##_t *q);
|
type *kdq_shift_##type(kdq_##type##_t *q);
|
||||||
|
|
||||||
#define kdq_init(type) kdq_init_##type()
|
#define kdq_init(type) kdq_init_##type()
|
||||||
#define kdq_destroy(type, q) kdq_destroy_##type(q)
|
#define kdq_destroy(type, q) kdq_destroy_##type(q)
|
||||||
#define kdq_resize(type, q, new_bits) kdq_resize_##type(q, new_bits)
|
#define kdq_resize(type, q, new_bits) kdq_resize_##type(q, new_bits)
|
||||||
#define kdq_pushp(type, q) kdq_pushp_##type(q)
|
#define kdq_pushp(type, q) kdq_pushp_##type(q)
|
||||||
#define kdq_push(type, q, v) kdq_push_##type(q, v)
|
#define kdq_push(type, q, v) kdq_push_##type(q, v)
|
||||||
#define kdq_pop(type, q) kdq_pop_##type(q)
|
#define kdq_pop(type, q) kdq_pop_##type(q)
|
||||||
#define kdq_unshiftp(type, q) kdq_unshiftp_##type(q)
|
#define kdq_unshiftp(type, q) kdq_unshiftp_##type(q)
|
||||||
#define kdq_unshift(type, q, v) kdq_unshift_##type(q, v)
|
#define kdq_unshift(type, q, v) kdq_unshift_##type(q, v)
|
||||||
#define kdq_shift(type, q) kdq_shift_##type(q)
|
#define kdq_shift(type, q) kdq_shift_##type(q)
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,120 +1,120 @@
|
|||||||
#ifndef KETOPT_H
|
#ifndef KETOPT_H
|
||||||
#define KETOPT_H
|
#define KETOPT_H
|
||||||
|
|
||||||
#include <string.h> /* for strchr() and strncmp() */
|
#include <string.h> /* for strchr() and strncmp() */
|
||||||
|
|
||||||
#define ko_no_argument 0
|
#define ko_no_argument 0
|
||||||
#define ko_required_argument 1
|
#define ko_required_argument 1
|
||||||
#define ko_optional_argument 2
|
#define ko_optional_argument 2
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
int ind; /* equivalent to optind */
|
int ind; /* equivalent to optind */
|
||||||
int opt; /* equivalent to optopt */
|
int opt; /* equivalent to optopt */
|
||||||
char *arg; /* equivalent to optarg */
|
char *arg; /* equivalent to optarg */
|
||||||
int longidx; /* index of a long option; or -1 if short */
|
int longidx; /* index of a long option; or -1 if short */
|
||||||
/* private variables not intended for external uses */
|
/* private variables not intended for external uses */
|
||||||
int i, pos, n_args;
|
int i, pos, n_args;
|
||||||
} ketopt_t;
|
} ketopt_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
const char *name;
|
const char *name;
|
||||||
int has_arg;
|
int has_arg;
|
||||||
int val;
|
int val;
|
||||||
} ko_longopt_t;
|
} ko_longopt_t;
|
||||||
|
|
||||||
static ketopt_t KETOPT_INIT = { 1, 0, 0, -1, 1, 0, 0 };
|
static ketopt_t KETOPT_INIT = { 1, 0, 0, -1, 1, 0, 0 };
|
||||||
|
|
||||||
static void ketopt_permute(char *argv[], int j, int n) /* move argv[j] over n elements to the left */
|
static void ketopt_permute(char *argv[], int j, int n) /* move argv[j] over n elements to the left */
|
||||||
{
|
{
|
||||||
int k;
|
int k;
|
||||||
char *p = argv[j];
|
char *p = argv[j];
|
||||||
for (k = 0; k < n; ++k)
|
for (k = 0; k < n; ++k)
|
||||||
argv[j - k] = argv[j - k - 1];
|
argv[j - k] = argv[j - k - 1];
|
||||||
argv[j - k] = p;
|
argv[j - k] = p;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Parse command-line options and arguments
|
* Parse command-line options and arguments
|
||||||
*
|
*
|
||||||
* This fuction has a similar interface to GNU's getopt_long(). Each call
|
* This fuction has a similar interface to GNU's getopt_long(). Each call
|
||||||
* parses one option and returns the option name. s->arg points to the option
|
* parses one option and returns the option name. s->arg points to the option
|
||||||
* argument if present. The function returns -1 when all command-line arguments
|
* argument if present. The function returns -1 when all command-line arguments
|
||||||
* are parsed. In this case, s->ind is the index of the first non-option
|
* are parsed. In this case, s->ind is the index of the first non-option
|
||||||
* argument.
|
* argument.
|
||||||
*
|
*
|
||||||
* @param s status; shall be initialized to KETOPT_INIT on the first call
|
* @param s status; shall be initialized to KETOPT_INIT on the first call
|
||||||
* @param argc length of argv[]
|
* @param argc length of argv[]
|
||||||
* @param argv list of command-line arguments; argv[0] is ignored
|
* @param argv list of command-line arguments; argv[0] is ignored
|
||||||
* @param permute non-zero to move options ahead of non-option arguments
|
* @param permute non-zero to move options ahead of non-option arguments
|
||||||
* @param ostr option string
|
* @param ostr option string
|
||||||
* @param longopts long options
|
* @param longopts long options
|
||||||
*
|
*
|
||||||
* @return ASCII for a short option; ko_longopt_t::val for a long option; -1 if
|
* @return ASCII for a short option; ko_longopt_t::val for a long option; -1 if
|
||||||
* argv[] is fully processed; '?' for an unknown option or an ambiguous
|
* argv[] is fully processed; '?' for an unknown option or an ambiguous
|
||||||
* long option; ':' if an option argument is missing
|
* long option; ':' if an option argument is missing
|
||||||
*/
|
*/
|
||||||
static int ketopt(ketopt_t *s, int argc, char *argv[], int permute, const char *ostr, const ko_longopt_t *longopts)
|
static int ketopt(ketopt_t *s, int argc, char *argv[], int permute, const char *ostr, const ko_longopt_t *longopts)
|
||||||
{
|
{
|
||||||
int opt = -1, i0, j;
|
int opt = -1, i0, j;
|
||||||
if (permute) {
|
if (permute) {
|
||||||
while (s->i < argc && (argv[s->i][0] != '-' || argv[s->i][1] == '\0'))
|
while (s->i < argc && (argv[s->i][0] != '-' || argv[s->i][1] == '\0'))
|
||||||
++s->i, ++s->n_args;
|
++s->i, ++s->n_args;
|
||||||
}
|
}
|
||||||
s->arg = 0, s->longidx = -1, i0 = s->i;
|
s->arg = 0, s->longidx = -1, i0 = s->i;
|
||||||
if (s->i >= argc || argv[s->i][0] != '-' || argv[s->i][1] == '\0') {
|
if (s->i >= argc || argv[s->i][0] != '-' || argv[s->i][1] == '\0') {
|
||||||
s->ind = s->i - s->n_args;
|
s->ind = s->i - s->n_args;
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
if (argv[s->i][0] == '-' && argv[s->i][1] == '-') { /* "--" or a long option */
|
if (argv[s->i][0] == '-' && argv[s->i][1] == '-') { /* "--" or a long option */
|
||||||
if (argv[s->i][2] == '\0') { /* a bare "--" */
|
if (argv[s->i][2] == '\0') { /* a bare "--" */
|
||||||
ketopt_permute(argv, s->i, s->n_args);
|
ketopt_permute(argv, s->i, s->n_args);
|
||||||
++s->i, s->ind = s->i - s->n_args;
|
++s->i, s->ind = s->i - s->n_args;
|
||||||
return -1;
|
return -1;
|
||||||
}
|
}
|
||||||
s->opt = 0, opt = '?', s->pos = -1;
|
s->opt = 0, opt = '?', s->pos = -1;
|
||||||
if (longopts) { /* parse long options */
|
if (longopts) { /* parse long options */
|
||||||
int k, n_exact = 0, n_partial = 0;
|
int k, n_exact = 0, n_partial = 0;
|
||||||
const ko_longopt_t *o = 0, *o_exact = 0, *o_partial = 0;
|
const ko_longopt_t *o = 0, *o_exact = 0, *o_partial = 0;
|
||||||
for (j = 2; argv[s->i][j] != '\0' && argv[s->i][j] != '='; ++j) {} /* find the end of the option name */
|
for (j = 2; argv[s->i][j] != '\0' && argv[s->i][j] != '='; ++j) {} /* find the end of the option name */
|
||||||
for (k = 0; longopts[k].name != 0; ++k)
|
for (k = 0; longopts[k].name != 0; ++k)
|
||||||
if (strncmp(&argv[s->i][2], longopts[k].name, j - 2) == 0) {
|
if (strncmp(&argv[s->i][2], longopts[k].name, j - 2) == 0) {
|
||||||
if (longopts[k].name[j - 2] == 0) ++n_exact, o_exact = &longopts[k];
|
if (longopts[k].name[j - 2] == 0) ++n_exact, o_exact = &longopts[k];
|
||||||
else ++n_partial, o_partial = &longopts[k];
|
else ++n_partial, o_partial = &longopts[k];
|
||||||
}
|
}
|
||||||
if (n_exact > 1 || (n_exact == 0 && n_partial > 1)) return '?';
|
if (n_exact > 1 || (n_exact == 0 && n_partial > 1)) return '?';
|
||||||
o = n_exact == 1? o_exact : n_partial == 1? o_partial : 0;
|
o = n_exact == 1? o_exact : n_partial == 1? o_partial : 0;
|
||||||
if (o) {
|
if (o) {
|
||||||
s->opt = opt = o->val, s->longidx = o - longopts;
|
s->opt = opt = o->val, s->longidx = o - longopts;
|
||||||
if (argv[s->i][j] == '=') s->arg = &argv[s->i][j + 1];
|
if (argv[s->i][j] == '=') s->arg = &argv[s->i][j + 1];
|
||||||
if (o->has_arg == 1 && argv[s->i][j] == '\0') {
|
if (o->has_arg == 1 && argv[s->i][j] == '\0') {
|
||||||
if (s->i < argc - 1) s->arg = argv[++s->i];
|
if (s->i < argc - 1) s->arg = argv[++s->i];
|
||||||
else opt = ':'; /* missing option argument */
|
else opt = ':'; /* missing option argument */
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else { /* a short option */
|
} else { /* a short option */
|
||||||
const char *p;
|
const char *p;
|
||||||
if (s->pos == 0) s->pos = 1;
|
if (s->pos == 0) s->pos = 1;
|
||||||
opt = s->opt = argv[s->i][s->pos++];
|
opt = s->opt = argv[s->i][s->pos++];
|
||||||
p = strchr((char*)ostr, opt);
|
p = strchr((char*)ostr, opt);
|
||||||
if (p == 0) {
|
if (p == 0) {
|
||||||
opt = '?'; /* unknown option */
|
opt = '?'; /* unknown option */
|
||||||
} else if (p[1] == ':') {
|
} else if (p[1] == ':') {
|
||||||
if (argv[s->i][s->pos] == 0) {
|
if (argv[s->i][s->pos] == 0) {
|
||||||
if (s->i < argc - 1) s->arg = argv[++s->i];
|
if (s->i < argc - 1) s->arg = argv[++s->i];
|
||||||
else opt = ':'; /* missing option argument */
|
else opt = ':'; /* missing option argument */
|
||||||
} else s->arg = &argv[s->i][s->pos];
|
} else s->arg = &argv[s->i][s->pos];
|
||||||
s->pos = -1;
|
s->pos = -1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (s->pos < 0 || argv[s->i][s->pos] == 0) {
|
if (s->pos < 0 || argv[s->i][s->pos] == 0) {
|
||||||
++s->i, s->pos = 0;
|
++s->i, s->pos = 0;
|
||||||
if (s->n_args > 0) /* permute */
|
if (s->n_args > 0) /* permute */
|
||||||
for (j = i0; j < s->i; ++j)
|
for (j = i0; j < s->i; ++j)
|
||||||
ketopt_permute(argv, j, s->n_args);
|
ketopt_permute(argv, j, s->n_args);
|
||||||
}
|
}
|
||||||
s->ind = s->i - s->n_args;
|
s->ind = s->i - s->n_args;
|
||||||
return opt;
|
return opt;
|
||||||
}
|
}
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,393 +1,393 @@
|
|||||||
/* The MIT License
|
/* The MIT License
|
||||||
|
|
||||||
Copyright (c) 2019 by Attractive Chaos <attractor@live.co.uk>
|
Copyright (c) 2019 by Attractive Chaos <attractor@live.co.uk>
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
a copy of this software and associated documentation files (the
|
a copy of this software and associated documentation files (the
|
||||||
"Software"), to deal in the Software without restriction, including
|
"Software"), to deal in the Software without restriction, including
|
||||||
without limitation the rights to use, copy, modify, merge, publish,
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
distribute, sublicense, and/or sell copies of the Software, and to
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
permit persons to whom the Software is furnished to do so, subject to
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
the following conditions:
|
the following conditions:
|
||||||
|
|
||||||
The above copyright notice and this permission notice shall be
|
The above copyright notice and this permission notice shall be
|
||||||
included in all copies or substantial portions of the Software.
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
SOFTWARE.
|
SOFTWARE.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#ifndef __AC_KHASHL_H
|
#ifndef __AC_KHASHL_H
|
||||||
#define __AC_KHASHL_H
|
#define __AC_KHASHL_H
|
||||||
|
|
||||||
#define AC_VERSION_KHASHL_H "0.1"
|
#define AC_VERSION_KHASHL_H "0.1"
|
||||||
|
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include <limits.h>
|
#include <limits.h>
|
||||||
|
|
||||||
/************************************
|
/************************************
|
||||||
* Compiler specific configurations *
|
* Compiler specific configurations *
|
||||||
************************************/
|
************************************/
|
||||||
|
|
||||||
#if UINT_MAX == 0xffffffffu
|
#if UINT_MAX == 0xffffffffu
|
||||||
typedef unsigned int khint32_t;
|
typedef unsigned int khint32_t;
|
||||||
#elif ULONG_MAX == 0xffffffffu
|
#elif ULONG_MAX == 0xffffffffu
|
||||||
typedef unsigned long khint32_t;
|
typedef unsigned long khint32_t;
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#if ULONG_MAX == ULLONG_MAX
|
#if ULONG_MAX == ULLONG_MAX
|
||||||
typedef unsigned long khint64_t;
|
typedef unsigned long khint64_t;
|
||||||
#else
|
#else
|
||||||
typedef unsigned long long khint64_t;
|
typedef unsigned long long khint64_t;
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#ifndef kh_inline
|
#ifndef kh_inline
|
||||||
#ifdef _MSC_VER
|
#ifdef _MSC_VER
|
||||||
#define kh_inline __inline
|
#define kh_inline __inline
|
||||||
#else
|
#else
|
||||||
#define kh_inline inline
|
#define kh_inline inline
|
||||||
#endif
|
#endif
|
||||||
#endif /* kh_inline */
|
#endif /* kh_inline */
|
||||||
|
|
||||||
#ifndef klib_unused
|
#ifndef klib_unused
|
||||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
||||||
#define klib_unused __attribute__ ((__unused__))
|
#define klib_unused __attribute__ ((__unused__))
|
||||||
#else
|
#else
|
||||||
#define klib_unused
|
#define klib_unused
|
||||||
#endif
|
#endif
|
||||||
#endif /* klib_unused */
|
#endif /* klib_unused */
|
||||||
|
|
||||||
#define KH_LOCAL static kh_inline klib_unused
|
#define KH_LOCAL static kh_inline klib_unused
|
||||||
|
|
||||||
typedef khint32_t khint_t;
|
typedef khint32_t khint_t;
|
||||||
|
|
||||||
/******************
|
/******************
|
||||||
* malloc aliases *
|
* malloc aliases *
|
||||||
******************/
|
******************/
|
||||||
|
|
||||||
#ifndef kcalloc
|
#ifndef kcalloc
|
||||||
#define kcalloc(N,Z) calloc(N,Z)
|
#define kcalloc(N,Z) calloc(N,Z)
|
||||||
#endif
|
#endif
|
||||||
#ifndef kmalloc
|
#ifndef kmalloc
|
||||||
#define kmalloc(Z) malloc(Z)
|
#define kmalloc(Z) malloc(Z)
|
||||||
#endif
|
#endif
|
||||||
#ifndef krealloc
|
#ifndef krealloc
|
||||||
#define krealloc(P,Z) realloc(P,Z)
|
#define krealloc(P,Z) realloc(P,Z)
|
||||||
#endif
|
#endif
|
||||||
#ifndef kfree
|
#ifndef kfree
|
||||||
#define kfree(P) free(P)
|
#define kfree(P) free(P)
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/****************************
|
/****************************
|
||||||
* Simple private functions *
|
* Simple private functions *
|
||||||
****************************/
|
****************************/
|
||||||
|
|
||||||
#define __kh_used(flag, i) (flag[i>>5] >> (i&0x1fU) & 1U)
|
#define __kh_used(flag, i) (flag[i>>5] >> (i&0x1fU) & 1U)
|
||||||
#define __kh_set_used(flag, i) (flag[i>>5] |= 1U<<(i&0x1fU))
|
#define __kh_set_used(flag, i) (flag[i>>5] |= 1U<<(i&0x1fU))
|
||||||
#define __kh_set_unused(flag, i) (flag[i>>5] &= ~(1U<<(i&0x1fU)))
|
#define __kh_set_unused(flag, i) (flag[i>>5] &= ~(1U<<(i&0x1fU)))
|
||||||
|
|
||||||
#define __kh_fsize(m) ((m) < 32? 1 : (m)>>5)
|
#define __kh_fsize(m) ((m) < 32? 1 : (m)>>5)
|
||||||
|
|
||||||
static kh_inline khint_t __kh_h2b(khint_t hash, khint_t bits) { return hash * 2654435769U >> (32 - bits); }
|
static kh_inline khint_t __kh_h2b(khint_t hash, khint_t bits) { return hash * 2654435769U >> (32 - bits); }
|
||||||
|
|
||||||
/*******************
|
/*******************
|
||||||
* Hash table base *
|
* Hash table base *
|
||||||
*******************/
|
*******************/
|
||||||
|
|
||||||
#define __KHASHL_TYPE(HType, khkey_t) \
|
#define __KHASHL_TYPE(HType, khkey_t) \
|
||||||
typedef struct HType { \
|
typedef struct HType { \
|
||||||
khint_t bits, count; \
|
khint_t bits, count; \
|
||||||
khint32_t *used; \
|
khint32_t *used; \
|
||||||
khkey_t *keys; \
|
khkey_t *keys; \
|
||||||
} HType;
|
} HType;
|
||||||
|
|
||||||
#define __KHASHL_PROTOTYPES(HType, prefix, khkey_t) \
|
#define __KHASHL_PROTOTYPES(HType, prefix, khkey_t) \
|
||||||
extern HType *prefix##_init(void); \
|
extern HType *prefix##_init(void); \
|
||||||
extern void prefix##_destroy(HType *h); \
|
extern void prefix##_destroy(HType *h); \
|
||||||
extern void prefix##_clear(HType *h); \
|
extern void prefix##_clear(HType *h); \
|
||||||
extern khint_t prefix##_getp(const HType *h, const khkey_t *key); \
|
extern khint_t prefix##_getp(const HType *h, const khkey_t *key); \
|
||||||
extern int prefix##_resize(HType *h, khint_t new_n_buckets); \
|
extern int prefix##_resize(HType *h, khint_t new_n_buckets); \
|
||||||
extern khint_t prefix##_putp(HType *h, const khkey_t *key, int *absent); \
|
extern khint_t prefix##_putp(HType *h, const khkey_t *key, int *absent); \
|
||||||
extern void prefix##_del(HType *h, khint_t k);
|
extern void prefix##_del(HType *h, khint_t k);
|
||||||
|
|
||||||
#define __KHASHL_IMPL_BASIC(SCOPE, HType, prefix) \
|
#define __KHASHL_IMPL_BASIC(SCOPE, HType, prefix) \
|
||||||
SCOPE HType *prefix##_init(void) { \
|
SCOPE HType *prefix##_init(void) { \
|
||||||
return (HType*)kcalloc(1, sizeof(HType)); \
|
return (HType*)kcalloc(1, sizeof(HType)); \
|
||||||
} \
|
} \
|
||||||
SCOPE void prefix##_destroy(HType *h) { \
|
SCOPE void prefix##_destroy(HType *h) { \
|
||||||
if (!h) return; \
|
if (!h) return; \
|
||||||
kfree((void *)h->keys); kfree(h->used); \
|
kfree((void *)h->keys); kfree(h->used); \
|
||||||
kfree(h); \
|
kfree(h); \
|
||||||
} \
|
} \
|
||||||
SCOPE void prefix##_clear(HType *h) { \
|
SCOPE void prefix##_clear(HType *h) { \
|
||||||
if (h && h->used) { \
|
if (h && h->used) { \
|
||||||
uint32_t n_buckets = 1U << h->bits; \
|
uint32_t n_buckets = 1U << h->bits; \
|
||||||
memset(h->used, 0, __kh_fsize(n_buckets) * sizeof(khint32_t)); \
|
memset(h->used, 0, __kh_fsize(n_buckets) * sizeof(khint32_t)); \
|
||||||
h->count = 0; \
|
h->count = 0; \
|
||||||
} \
|
} \
|
||||||
}
|
}
|
||||||
#define __KHASHL_IMPL_S_L(SCOPE, HType, prefix, khkey_t) \
|
#define __KHASHL_IMPL_S_L(SCOPE, HType, prefix, khkey_t) \
|
||||||
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { \
|
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { \
|
||||||
if (!h) return 0; \
|
if (!h) return 0; \
|
||||||
uint8_t ff; \
|
uint8_t ff; \
|
||||||
khint_t n_buckets = (h->keys? 1U<<h->bits : 0U); \
|
khint_t n_buckets = (h->keys? 1U<<h->bits : 0U); \
|
||||||
fwrite(&n_buckets, sizeof(n_buckets), 1, fp); \
|
fwrite(&n_buckets, sizeof(n_buckets), 1, fp); \
|
||||||
fwrite(&h->bits, sizeof(h->bits), 1, fp); \
|
fwrite(&h->bits, sizeof(h->bits), 1, fp); \
|
||||||
fwrite(&h->count, sizeof(h->count), 1, fp); \
|
fwrite(&h->count, sizeof(h->count), 1, fp); \
|
||||||
ff = h->used? 1:0; fwrite(&ff, sizeof(ff), 1, fp); \
|
ff = h->used? 1:0; fwrite(&ff, sizeof(ff), 1, fp); \
|
||||||
if(ff) fwrite(h->used, sizeof(khint32_t), __kh_fsize(n_buckets), fp); \
|
if(ff) fwrite(h->used, sizeof(khint32_t), __kh_fsize(n_buckets), fp); \
|
||||||
ff = h->keys? 1:0; fwrite(&ff, sizeof(ff), 1, fp); \
|
ff = h->keys? 1:0; fwrite(&ff, sizeof(ff), 1, fp); \
|
||||||
if(ff) fwrite(h->keys, sizeof(khkey_t), n_buckets, fp); \
|
if(ff) fwrite(h->keys, sizeof(khkey_t), n_buckets, fp); \
|
||||||
return 1; \
|
return 1; \
|
||||||
} \
|
} \
|
||||||
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { \
|
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { \
|
||||||
(*h) = prefix##_init(); \
|
(*h) = prefix##_init(); \
|
||||||
khint_t n_buckets; \
|
khint_t n_buckets; \
|
||||||
uint64_t flag = 0;\
|
uint64_t flag = 0;\
|
||||||
uint8_t ff; \
|
uint8_t ff; \
|
||||||
flag += fread(&n_buckets, sizeof(n_buckets), 1, fp); \
|
flag += fread(&n_buckets, sizeof(n_buckets), 1, fp); \
|
||||||
flag += fread(&(*h)->bits, sizeof((*h)->bits), 1, fp); \
|
flag += fread(&(*h)->bits, sizeof((*h)->bits), 1, fp); \
|
||||||
flag += fread(&(*h)->count, sizeof((*h)->count), 1, fp); \
|
flag += fread(&(*h)->count, sizeof((*h)->count), 1, fp); \
|
||||||
flag += fread(&ff, sizeof(ff), 1, fp); \
|
flag += fread(&ff, sizeof(ff), 1, fp); \
|
||||||
if(ff) {\
|
if(ff) {\
|
||||||
(*h)->used = (khint32_t*)kmalloc(__kh_fsize(n_buckets) * sizeof(khint32_t)); \
|
(*h)->used = (khint32_t*)kmalloc(__kh_fsize(n_buckets) * sizeof(khint32_t)); \
|
||||||
flag += fread((*h)->used, sizeof(khint32_t), __kh_fsize(n_buckets), fp); }\
|
flag += fread((*h)->used, sizeof(khint32_t), __kh_fsize(n_buckets), fp); }\
|
||||||
flag += fread(&ff, sizeof(ff), 1, fp); \
|
flag += fread(&ff, sizeof(ff), 1, fp); \
|
||||||
if(ff) {\
|
if(ff) {\
|
||||||
(*h)->keys = (khkey_t*)kmalloc(n_buckets * sizeof(khkey_t)); \
|
(*h)->keys = (khkey_t*)kmalloc(n_buckets * sizeof(khkey_t)); \
|
||||||
flag += fread((*h)->keys, sizeof(khkey_t), n_buckets, fp); }\
|
flag += fread((*h)->keys, sizeof(khkey_t), n_buckets, fp); }\
|
||||||
return 1; \
|
return 1; \
|
||||||
} \
|
} \
|
||||||
|
|
||||||
#define __KHASHL_IMPL_GET(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
#define __KHASHL_IMPL_GET(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||||
SCOPE khint_t prefix##_getp(const HType *h, const khkey_t *key) { \
|
SCOPE khint_t prefix##_getp(const HType *h, const khkey_t *key) { \
|
||||||
khint_t i, last, n_buckets, mask; \
|
khint_t i, last, n_buckets, mask; \
|
||||||
if (h->keys == 0) return 0; \
|
if (h->keys == 0) return 0; \
|
||||||
n_buckets = 1U << h->bits; \
|
n_buckets = 1U << h->bits; \
|
||||||
mask = n_buckets - 1U; \
|
mask = n_buckets - 1U; \
|
||||||
i = last = __kh_h2b(__hash_fn(*key), h->bits); \
|
i = last = __kh_h2b(__hash_fn(*key), h->bits); \
|
||||||
while (__kh_used(h->used, i) && !__hash_eq(h->keys[i], *key)) { \
|
while (__kh_used(h->used, i) && !__hash_eq(h->keys[i], *key)) { \
|
||||||
i = (i + 1U) & mask; \
|
i = (i + 1U) & mask; \
|
||||||
if (i == last) return n_buckets; \
|
if (i == last) return n_buckets; \
|
||||||
} \
|
} \
|
||||||
return !__kh_used(h->used, i)? n_buckets : i; \
|
return !__kh_used(h->used, i)? n_buckets : i; \
|
||||||
} \
|
} \
|
||||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { return prefix##_getp(h, &key); }
|
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { return prefix##_getp(h, &key); }
|
||||||
|
|
||||||
#define __KHASHL_IMPL_RESIZE(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
#define __KHASHL_IMPL_RESIZE(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||||
SCOPE int prefix##_resize(HType *h, khint_t new_n_buckets) { \
|
SCOPE int prefix##_resize(HType *h, khint_t new_n_buckets) { \
|
||||||
khint32_t *new_used = 0; \
|
khint32_t *new_used = 0; \
|
||||||
khint_t j = 0, x = new_n_buckets, n_buckets, new_bits, new_mask; \
|
khint_t j = 0, x = new_n_buckets, n_buckets, new_bits, new_mask; \
|
||||||
while ((x >>= 1) != 0) ++j; \
|
while ((x >>= 1) != 0) ++j; \
|
||||||
if (new_n_buckets & (new_n_buckets - 1)) ++j; \
|
if (new_n_buckets & (new_n_buckets - 1)) ++j; \
|
||||||
new_bits = j > 2? j : 2; \
|
new_bits = j > 2? j : 2; \
|
||||||
new_n_buckets = 1U << new_bits; \
|
new_n_buckets = 1U << new_bits; \
|
||||||
if (h->count > (new_n_buckets>>1) + (new_n_buckets>>2)) return 0; /* requested size is too small */ \
|
if (h->count > (new_n_buckets>>1) + (new_n_buckets>>2)) return 0; /* requested size is too small */ \
|
||||||
new_used = (khint32_t*)kmalloc(__kh_fsize(new_n_buckets) * sizeof(khint32_t)); \
|
new_used = (khint32_t*)kmalloc(__kh_fsize(new_n_buckets) * sizeof(khint32_t)); \
|
||||||
memset(new_used, 0, __kh_fsize(new_n_buckets) * sizeof(khint32_t)); \
|
memset(new_used, 0, __kh_fsize(new_n_buckets) * sizeof(khint32_t)); \
|
||||||
if (!new_used) return -1; /* not enough memory */ \
|
if (!new_used) return -1; /* not enough memory */ \
|
||||||
n_buckets = h->keys? 1U<<h->bits : 0U; \
|
n_buckets = h->keys? 1U<<h->bits : 0U; \
|
||||||
if (n_buckets < new_n_buckets) { /* expand */ \
|
if (n_buckets < new_n_buckets) { /* expand */ \
|
||||||
khkey_t *new_keys = (khkey_t*)krealloc((void*)h->keys, new_n_buckets * sizeof(khkey_t)); \
|
khkey_t *new_keys = (khkey_t*)krealloc((void*)h->keys, new_n_buckets * sizeof(khkey_t)); \
|
||||||
if (!new_keys) { kfree(new_used); return -1; } \
|
if (!new_keys) { kfree(new_used); return -1; } \
|
||||||
h->keys = new_keys; \
|
h->keys = new_keys; \
|
||||||
} /* otherwise shrink */ \
|
} /* otherwise shrink */ \
|
||||||
new_mask = new_n_buckets - 1; \
|
new_mask = new_n_buckets - 1; \
|
||||||
for (j = 0; j != n_buckets; ++j) { \
|
for (j = 0; j != n_buckets; ++j) { \
|
||||||
khkey_t key; \
|
khkey_t key; \
|
||||||
if (!__kh_used(h->used, j)) continue; \
|
if (!__kh_used(h->used, j)) continue; \
|
||||||
key = h->keys[j]; \
|
key = h->keys[j]; \
|
||||||
__kh_set_unused(h->used, j); \
|
__kh_set_unused(h->used, j); \
|
||||||
while (1) { /* kick-out process; sort of like in Cuckoo hashing */ \
|
while (1) { /* kick-out process; sort of like in Cuckoo hashing */ \
|
||||||
khint_t i; \
|
khint_t i; \
|
||||||
i = __kh_h2b(__hash_fn(key), new_bits); \
|
i = __kh_h2b(__hash_fn(key), new_bits); \
|
||||||
while (__kh_used(new_used, i)) i = (i + 1) & new_mask; \
|
while (__kh_used(new_used, i)) i = (i + 1) & new_mask; \
|
||||||
__kh_set_used(new_used, i); \
|
__kh_set_used(new_used, i); \
|
||||||
if (i < n_buckets && __kh_used(h->used, i)) { /* kick out the existing element */ \
|
if (i < n_buckets && __kh_used(h->used, i)) { /* kick out the existing element */ \
|
||||||
{ khkey_t tmp = h->keys[i]; h->keys[i] = key; key = tmp; } \
|
{ khkey_t tmp = h->keys[i]; h->keys[i] = key; key = tmp; } \
|
||||||
__kh_set_unused(h->used, i); /* mark it as deleted in the old hash table */ \
|
__kh_set_unused(h->used, i); /* mark it as deleted in the old hash table */ \
|
||||||
} else { /* write the element and jump out of the loop */ \
|
} else { /* write the element and jump out of the loop */ \
|
||||||
h->keys[i] = key; \
|
h->keys[i] = key; \
|
||||||
break; \
|
break; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
if (n_buckets > new_n_buckets) /* shrink the hash table */ \
|
if (n_buckets > new_n_buckets) /* shrink the hash table */ \
|
||||||
h->keys = (khkey_t*)krealloc((void *)h->keys, new_n_buckets * sizeof(khkey_t)); \
|
h->keys = (khkey_t*)krealloc((void *)h->keys, new_n_buckets * sizeof(khkey_t)); \
|
||||||
kfree(h->used); /* free the working space */ \
|
kfree(h->used); /* free the working space */ \
|
||||||
h->used = new_used, h->bits = new_bits; \
|
h->used = new_used, h->bits = new_bits; \
|
||||||
return 0; \
|
return 0; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define __KHASHL_IMPL_PUT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
#define __KHASHL_IMPL_PUT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||||
SCOPE khint_t prefix##_putp(HType *h, const khkey_t *key, int *absent) { \
|
SCOPE khint_t prefix##_putp(HType *h, const khkey_t *key, int *absent) { \
|
||||||
khint_t n_buckets, i, last, mask; \
|
khint_t n_buckets, i, last, mask; \
|
||||||
n_buckets = h->keys? 1U<<h->bits : 0U; \
|
n_buckets = h->keys? 1U<<h->bits : 0U; \
|
||||||
*absent = -1; \
|
*absent = -1; \
|
||||||
if (h->count >= (n_buckets>>1) + (n_buckets>>2)) { /* rehashing */ \
|
if (h->count >= (n_buckets>>1) + (n_buckets>>2)) { /* rehashing */ \
|
||||||
if (prefix##_resize(h, n_buckets + 1U) < 0) \
|
if (prefix##_resize(h, n_buckets + 1U) < 0) \
|
||||||
return n_buckets; \
|
return n_buckets; \
|
||||||
n_buckets = 1U<<h->bits; \
|
n_buckets = 1U<<h->bits; \
|
||||||
} /* TODO: to implement automatically shrinking; resize() already support shrinking */ \
|
} /* TODO: to implement automatically shrinking; resize() already support shrinking */ \
|
||||||
mask = n_buckets - 1; \
|
mask = n_buckets - 1; \
|
||||||
i = last = __kh_h2b(__hash_fn(*key), h->bits); \
|
i = last = __kh_h2b(__hash_fn(*key), h->bits); \
|
||||||
while (__kh_used(h->used, i) && !__hash_eq(h->keys[i], *key)) { \
|
while (__kh_used(h->used, i) && !__hash_eq(h->keys[i], *key)) { \
|
||||||
i = (i + 1U) & mask; \
|
i = (i + 1U) & mask; \
|
||||||
if (i == last) break; \
|
if (i == last) break; \
|
||||||
} \
|
} \
|
||||||
if (!__kh_used(h->used, i)) { /* not present at all */ \
|
if (!__kh_used(h->used, i)) { /* not present at all */ \
|
||||||
h->keys[i] = *key; \
|
h->keys[i] = *key; \
|
||||||
__kh_set_used(h->used, i); \
|
__kh_set_used(h->used, i); \
|
||||||
++h->count; \
|
++h->count; \
|
||||||
*absent = 1; \
|
*absent = 1; \
|
||||||
} else *absent = 0; /* Don't touch h->keys[i] if present */ \
|
} else *absent = 0; /* Don't touch h->keys[i] if present */ \
|
||||||
return i; \
|
return i; \
|
||||||
} \
|
} \
|
||||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { return prefix##_putp(h, &key, absent); }
|
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { return prefix##_putp(h, &key, absent); }
|
||||||
|
|
||||||
#define __KHASHL_IMPL_DEL(SCOPE, HType, prefix, khkey_t, __hash_fn) \
|
#define __KHASHL_IMPL_DEL(SCOPE, HType, prefix, khkey_t, __hash_fn) \
|
||||||
SCOPE int prefix##_del(HType *h, khint_t i) { \
|
SCOPE int prefix##_del(HType *h, khint_t i) { \
|
||||||
khint_t j = i, k, mask, n_buckets; \
|
khint_t j = i, k, mask, n_buckets; \
|
||||||
if (h->keys == 0) return 0; \
|
if (h->keys == 0) return 0; \
|
||||||
n_buckets = 1U<<h->bits; \
|
n_buckets = 1U<<h->bits; \
|
||||||
mask = n_buckets - 1U; \
|
mask = n_buckets - 1U; \
|
||||||
while (1) { \
|
while (1) { \
|
||||||
j = (j + 1U) & mask; \
|
j = (j + 1U) & mask; \
|
||||||
if (j == i || !__kh_used(h->used, j)) break; /* j==i only when the table is completely full */ \
|
if (j == i || !__kh_used(h->used, j)) break; /* j==i only when the table is completely full */ \
|
||||||
k = __kh_h2b(__hash_fn(h->keys[j]), h->bits); \
|
k = __kh_h2b(__hash_fn(h->keys[j]), h->bits); \
|
||||||
if ((j > i && (k <= i || k > j)) || (j < i && (k <= i && k > j))) \
|
if ((j > i && (k <= i || k > j)) || (j < i && (k <= i && k > j))) \
|
||||||
h->keys[i] = h->keys[j], i = j; \
|
h->keys[i] = h->keys[j], i = j; \
|
||||||
} \
|
} \
|
||||||
__kh_set_unused(h->used, i); \
|
__kh_set_unused(h->used, i); \
|
||||||
--h->count; \
|
--h->count; \
|
||||||
return 1; \
|
return 1; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define KHASHL_DECLARE(HType, prefix, khkey_t) \
|
#define KHASHL_DECLARE(HType, prefix, khkey_t) \
|
||||||
__KHASHL_TYPE(HType, khkey_t) \
|
__KHASHL_TYPE(HType, khkey_t) \
|
||||||
__KHASHL_PROTOTYPES(HType, prefix, khkey_t)
|
__KHASHL_PROTOTYPES(HType, prefix, khkey_t)
|
||||||
|
|
||||||
#define KHASHL_INIT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
#define KHASHL_INIT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||||
__KHASHL_TYPE(HType, khkey_t) \
|
__KHASHL_TYPE(HType, khkey_t) \
|
||||||
__KHASHL_IMPL_BASIC(SCOPE, HType, prefix) \
|
__KHASHL_IMPL_BASIC(SCOPE, HType, prefix) \
|
||||||
__KHASHL_IMPL_S_L(SCOPE, HType, prefix, khkey_t) \
|
__KHASHL_IMPL_S_L(SCOPE, HType, prefix, khkey_t) \
|
||||||
__KHASHL_IMPL_GET(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
__KHASHL_IMPL_GET(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||||
__KHASHL_IMPL_RESIZE(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
__KHASHL_IMPL_RESIZE(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||||
__KHASHL_IMPL_PUT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
__KHASHL_IMPL_PUT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||||
__KHASHL_IMPL_DEL(SCOPE, HType, prefix, khkey_t, __hash_fn)
|
__KHASHL_IMPL_DEL(SCOPE, HType, prefix, khkey_t, __hash_fn)
|
||||||
|
|
||||||
/*****************************
|
/*****************************
|
||||||
* More convenient interface *
|
* More convenient interface *
|
||||||
*****************************/
|
*****************************/
|
||||||
|
|
||||||
#define __kh_packed __attribute__ ((__packed__))
|
#define __kh_packed __attribute__ ((__packed__))
|
||||||
#define __kh_cached_hash(x) ((x).hash)
|
#define __kh_cached_hash(x) ((x).hash)
|
||||||
|
|
||||||
#define KHASHL_SET_INIT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
#define KHASHL_SET_INIT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||||
typedef struct { khkey_t key; } __kh_packed HType##_s_bucket_t; \
|
typedef struct { khkey_t key; } __kh_packed HType##_s_bucket_t; \
|
||||||
static kh_inline khint_t prefix##_s_hash(HType##_s_bucket_t x) { return __hash_fn(x.key); } \
|
static kh_inline khint_t prefix##_s_hash(HType##_s_bucket_t x) { return __hash_fn(x.key); } \
|
||||||
static kh_inline int prefix##_s_eq(HType##_s_bucket_t x, HType##_s_bucket_t y) { return __hash_eq(x.key, y.key); } \
|
static kh_inline int prefix##_s_eq(HType##_s_bucket_t x, HType##_s_bucket_t y) { return __hash_eq(x.key, y.key); } \
|
||||||
KHASHL_INIT(KH_LOCAL, HType, prefix##_s, HType##_s_bucket_t, prefix##_s_hash, prefix##_s_eq) \
|
KHASHL_INIT(KH_LOCAL, HType, prefix##_s, HType##_s_bucket_t, prefix##_s_hash, prefix##_s_eq) \
|
||||||
SCOPE HType *prefix##_init(void) { return prefix##_s_init(); } \
|
SCOPE HType *prefix##_init(void) { return prefix##_s_init(); } \
|
||||||
SCOPE void prefix##_destroy(HType *h) { prefix##_s_destroy(h); } \
|
SCOPE void prefix##_destroy(HType *h) { prefix##_s_destroy(h); } \
|
||||||
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { return prefix##_s_save(h, fp); } \
|
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { return prefix##_s_save(h, fp); } \
|
||||||
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { return prefix##_s_load(h, fp); } \
|
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { return prefix##_s_load(h, fp); } \
|
||||||
SCOPE void prefix##_resize(HType *h, khint_t new_n_buckets) { prefix##_s_resize(h, new_n_buckets); } \
|
SCOPE void prefix##_resize(HType *h, khint_t new_n_buckets) { prefix##_s_resize(h, new_n_buckets); } \
|
||||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_s_bucket_t t; t.key = key; return prefix##_s_getp(h, &t); } \
|
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_s_bucket_t t; t.key = key; return prefix##_s_getp(h, &t); } \
|
||||||
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_s_del(h, k); } \
|
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_s_del(h, k); } \
|
||||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_s_bucket_t t; t.key = key; return prefix##_s_putp(h, &t, absent); }
|
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_s_bucket_t t; t.key = key; return prefix##_s_putp(h, &t, absent); }
|
||||||
|
|
||||||
#define KHASHL_MAP_INIT(SCOPE, HType, prefix, khkey_t, kh_val_t, __hash_fn, __hash_eq) \
|
#define KHASHL_MAP_INIT(SCOPE, HType, prefix, khkey_t, kh_val_t, __hash_fn, __hash_eq) \
|
||||||
typedef struct { khkey_t key; kh_val_t val; } __kh_packed HType##_m_bucket_t; \
|
typedef struct { khkey_t key; kh_val_t val; } __kh_packed HType##_m_bucket_t; \
|
||||||
static kh_inline khint_t prefix##_m_hash(HType##_m_bucket_t x) { return __hash_fn(x.key); } \
|
static kh_inline khint_t prefix##_m_hash(HType##_m_bucket_t x) { return __hash_fn(x.key); } \
|
||||||
static kh_inline int prefix##_m_eq(HType##_m_bucket_t x, HType##_m_bucket_t y) { return __hash_eq(x.key, y.key); } \
|
static kh_inline int prefix##_m_eq(HType##_m_bucket_t x, HType##_m_bucket_t y) { return __hash_eq(x.key, y.key); } \
|
||||||
KHASHL_INIT(KH_LOCAL, HType, prefix##_m, HType##_m_bucket_t, prefix##_m_hash, prefix##_m_eq) \
|
KHASHL_INIT(KH_LOCAL, HType, prefix##_m, HType##_m_bucket_t, prefix##_m_hash, prefix##_m_eq) \
|
||||||
SCOPE HType *prefix##_init(void) { return prefix##_m_init(); } \
|
SCOPE HType *prefix##_init(void) { return prefix##_m_init(); } \
|
||||||
SCOPE void prefix##_destroy(HType *h) { prefix##_m_destroy(h); } \
|
SCOPE void prefix##_destroy(HType *h) { prefix##_m_destroy(h); } \
|
||||||
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { return prefix##_m_save(h, fp); } \
|
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { return prefix##_m_save(h, fp); } \
|
||||||
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { return prefix##_m_load(h, fp); } \
|
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { return prefix##_m_load(h, fp); } \
|
||||||
SCOPE void prefix##_resize(HType *h, khint_t new_n_buckets) { prefix##_m_resize(h, new_n_buckets); } \
|
SCOPE void prefix##_resize(HType *h, khint_t new_n_buckets) { prefix##_m_resize(h, new_n_buckets); } \
|
||||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_m_bucket_t t; t.key = key; return prefix##_m_getp(h, &t); } \
|
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_m_bucket_t t; t.key = key; return prefix##_m_getp(h, &t); } \
|
||||||
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_m_del(h, k); } \
|
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_m_del(h, k); } \
|
||||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_m_bucket_t t; t.key = key; return prefix##_m_putp(h, &t, absent); }
|
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_m_bucket_t t; t.key = key; return prefix##_m_putp(h, &t, absent); }
|
||||||
|
|
||||||
#define KHASHL_CSET_INIT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
#define KHASHL_CSET_INIT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||||
typedef struct { khkey_t key; khint_t hash; } __kh_packed HType##_cs_bucket_t; \
|
typedef struct { khkey_t key; khint_t hash; } __kh_packed HType##_cs_bucket_t; \
|
||||||
static kh_inline int prefix##_cs_eq(HType##_cs_bucket_t x, HType##_cs_bucket_t y) { return x.hash == y.hash && __hash_eq(x.key, y.key); } \
|
static kh_inline int prefix##_cs_eq(HType##_cs_bucket_t x, HType##_cs_bucket_t y) { return x.hash == y.hash && __hash_eq(x.key, y.key); } \
|
||||||
KHASHL_INIT(KH_LOCAL, HType, prefix##_cs, HType##_cs_bucket_t, __kh_cached_hash, prefix##_cs_eq) \
|
KHASHL_INIT(KH_LOCAL, HType, prefix##_cs, HType##_cs_bucket_t, __kh_cached_hash, prefix##_cs_eq) \
|
||||||
SCOPE HType *prefix##_init(void) { return prefix##_cs_init(); } \
|
SCOPE HType *prefix##_init(void) { return prefix##_cs_init(); } \
|
||||||
SCOPE void prefix##_destroy(HType *h) { prefix##_cs_destroy(h); } \
|
SCOPE void prefix##_destroy(HType *h) { prefix##_cs_destroy(h); } \
|
||||||
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { return prefix##_cs_save(h, fp); } \
|
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { return prefix##_cs_save(h, fp); } \
|
||||||
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { return prefix##_cs_load(h, fp); } \
|
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { return prefix##_cs_load(h, fp); } \
|
||||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_cs_bucket_t t; t.key = key; t.hash = __hash_fn(key); return prefix##_cs_getp(h, &t); } \
|
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_cs_bucket_t t; t.key = key; t.hash = __hash_fn(key); return prefix##_cs_getp(h, &t); } \
|
||||||
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_cs_del(h, k); } \
|
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_cs_del(h, k); } \
|
||||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_cs_bucket_t t; t.key = key, t.hash = __hash_fn(key); return prefix##_cs_putp(h, &t, absent); }
|
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_cs_bucket_t t; t.key = key, t.hash = __hash_fn(key); return prefix##_cs_putp(h, &t, absent); }
|
||||||
|
|
||||||
#define KHASHL_CMAP_INIT(SCOPE, HType, prefix, khkey_t, kh_val_t, __hash_fn, __hash_eq) \
|
#define KHASHL_CMAP_INIT(SCOPE, HType, prefix, khkey_t, kh_val_t, __hash_fn, __hash_eq) \
|
||||||
typedef struct { khkey_t key; kh_val_t val; khint_t hash; } __kh_packed HType##_cm_bucket_t; \
|
typedef struct { khkey_t key; kh_val_t val; khint_t hash; } __kh_packed HType##_cm_bucket_t; \
|
||||||
static kh_inline int prefix##_cm_eq(HType##_cm_bucket_t x, HType##_cm_bucket_t y) { return x.hash == y.hash && __hash_eq(x.key, y.key); } \
|
static kh_inline int prefix##_cm_eq(HType##_cm_bucket_t x, HType##_cm_bucket_t y) { return x.hash == y.hash && __hash_eq(x.key, y.key); } \
|
||||||
KHASHL_INIT(KH_LOCAL, HType, prefix##_cm, HType##_cm_bucket_t, __kh_cached_hash, prefix##_cm_eq) \
|
KHASHL_INIT(KH_LOCAL, HType, prefix##_cm, HType##_cm_bucket_t, __kh_cached_hash, prefix##_cm_eq) \
|
||||||
SCOPE HType *prefix##_init(void) { return prefix##_cm_init(); } \
|
SCOPE HType *prefix##_init(void) { return prefix##_cm_init(); } \
|
||||||
SCOPE void prefix##_destroy(HType *h) { prefix##_cm_destroy(h); } \
|
SCOPE void prefix##_destroy(HType *h) { prefix##_cm_destroy(h); } \
|
||||||
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { return prefix##_cm_save(h, fp); } \
|
SCOPE khint_t prefix##_save(HType *h, FILE* fp) { return prefix##_cm_save(h, fp); } \
|
||||||
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { return prefix##_cm_load(h, fp); } \
|
SCOPE khint_t prefix##_load(HType **h, FILE* fp) { return prefix##_cm_load(h, fp); } \
|
||||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_cm_bucket_t t; t.key = key; t.hash = __hash_fn(key); return prefix##_cm_getp(h, &t); } \
|
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_cm_bucket_t t; t.key = key; t.hash = __hash_fn(key); return prefix##_cm_getp(h, &t); } \
|
||||||
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_cm_del(h, k); } \
|
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_cm_del(h, k); } \
|
||||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_cm_bucket_t t; t.key = key, t.hash = __hash_fn(key); return prefix##_cm_putp(h, &t, absent); }
|
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_cm_bucket_t t; t.key = key, t.hash = __hash_fn(key); return prefix##_cm_putp(h, &t, absent); }
|
||||||
|
|
||||||
/**************************
|
/**************************
|
||||||
* Public macro functions *
|
* Public macro functions *
|
||||||
**************************/
|
**************************/
|
||||||
|
|
||||||
#define kh_bucket(h, x) ((h)->keys[x])
|
#define kh_bucket(h, x) ((h)->keys[x])
|
||||||
#define kh_size(h) ((h)->count)
|
#define kh_size(h) ((h)->count)
|
||||||
#define kh_capacity(h) ((h)->keys? 1U<<(h)->bits : 0U)
|
#define kh_capacity(h) ((h)->keys? 1U<<(h)->bits : 0U)
|
||||||
#define kh_end(h) kh_capacity(h)
|
#define kh_end(h) kh_capacity(h)
|
||||||
|
|
||||||
#define kh_key(h, x) ((h)->keys[x].key)
|
#define kh_key(h, x) ((h)->keys[x].key)
|
||||||
#define kh_val(h, x) ((h)->keys[x].val)
|
#define kh_val(h, x) ((h)->keys[x].val)
|
||||||
#define kh_exist(h, x) __kh_used((h)->used, (x))
|
#define kh_exist(h, x) __kh_used((h)->used, (x))
|
||||||
|
|
||||||
/**************************************
|
/**************************************
|
||||||
* Common hash and equality functions *
|
* Common hash and equality functions *
|
||||||
**************************************/
|
**************************************/
|
||||||
|
|
||||||
#define kh_eq_generic(a, b) ((a) == (b))
|
#define kh_eq_generic(a, b) ((a) == (b))
|
||||||
#define kh_eq_str(a, b) (strcmp((a), (b)) == 0)
|
#define kh_eq_str(a, b) (strcmp((a), (b)) == 0)
|
||||||
#define kh_hash_dummy(x) ((khint_t)(x))
|
#define kh_hash_dummy(x) ((khint_t)(x))
|
||||||
|
|
||||||
static kh_inline khint_t kh_hash_uint32(khint_t key) {
|
static kh_inline khint_t kh_hash_uint32(khint_t key) {
|
||||||
key += ~(key << 15);
|
key += ~(key << 15);
|
||||||
key ^= (key >> 10);
|
key ^= (key >> 10);
|
||||||
key += (key << 3);
|
key += (key << 3);
|
||||||
key ^= (key >> 6);
|
key ^= (key >> 6);
|
||||||
key += ~(key << 11);
|
key += ~(key << 11);
|
||||||
key ^= (key >> 16);
|
key ^= (key >> 16);
|
||||||
return key;
|
return key;
|
||||||
}
|
}
|
||||||
|
|
||||||
static kh_inline khint_t kh_hash_uint64(khint64_t key) {
|
static kh_inline khint_t kh_hash_uint64(khint64_t key) {
|
||||||
key = ~key + (key << 21);
|
key = ~key + (key << 21);
|
||||||
key = key ^ key >> 24;
|
key = key ^ key >> 24;
|
||||||
key = (key + (key << 3)) + (key << 8);
|
key = (key + (key << 3)) + (key << 8);
|
||||||
key = key ^ key >> 14;
|
key = key ^ key >> 14;
|
||||||
key = (key + (key << 2)) + (key << 4);
|
key = (key + (key << 2)) + (key << 4);
|
||||||
key = key ^ key >> 28;
|
key = key ^ key >> 28;
|
||||||
key = key + (key << 31);
|
key = key + (key << 31);
|
||||||
return (khint_t)key;
|
return (khint_t)key;
|
||||||
}
|
}
|
||||||
|
|
||||||
static kh_inline khint_t kh_hash_str(const char *s) {
|
static kh_inline khint_t kh_hash_str(const char *s) {
|
||||||
khint_t h = (khint_t)*s;
|
khint_t h = (khint_t)*s;
|
||||||
if (h) for (++s ; *s; ++s) h = (h << 5) - h + (khint_t)*s;
|
if (h) for (++s ; *s; ++s) h = (h << 5) - h + (khint_t)*s;
|
||||||
return h;
|
return h;
|
||||||
}
|
}
|
||||||
|
|
||||||
#endif /* __AC_KHASHL_H */
|
#endif /* __AC_KHASHL_H */
|
||||||
|
|||||||
@@ -1,251 +1,251 @@
|
|||||||
/* The MIT License
|
/* The MIT License
|
||||||
|
|
||||||
Copyright (c) 2008, 2009, 2011 Attractive Chaos <attractor@live.co.uk>
|
Copyright (c) 2008, 2009, 2011 Attractive Chaos <attractor@live.co.uk>
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
a copy of this software and associated documentation files (the
|
a copy of this software and associated documentation files (the
|
||||||
"Software"), to deal in the Software without restriction, including
|
"Software"), to deal in the Software without restriction, including
|
||||||
without limitation the rights to use, copy, modify, merge, publish,
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
distribute, sublicense, and/or sell copies of the Software, and to
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
permit persons to whom the Software is furnished to do so, subject to
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
the following conditions:
|
the following conditions:
|
||||||
|
|
||||||
The above copyright notice and this permission notice shall be
|
The above copyright notice and this permission notice shall be
|
||||||
included in all copies or substantial portions of the Software.
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
SOFTWARE.
|
SOFTWARE.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/* Last Modified: 05MAR2012 */
|
/* Last Modified: 05MAR2012 */
|
||||||
|
|
||||||
#ifndef AC_KSEQ_H
|
#ifndef AC_KSEQ_H
|
||||||
#define AC_KSEQ_H
|
#define AC_KSEQ_H
|
||||||
|
|
||||||
#include <ctype.h>
|
#include <ctype.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
|
|
||||||
#define KS_SEP_SPACE 0 // isspace(): \t, \n, \v, \f, \r
|
#define KS_SEP_SPACE 0 // isspace(): \t, \n, \v, \f, \r
|
||||||
#define KS_SEP_TAB 1 // isspace() && !' '
|
#define KS_SEP_TAB 1 // isspace() && !' '
|
||||||
#define KS_SEP_LINE 2 // line separator: "\n" (Unix) or "\r\n" (Windows)
|
#define KS_SEP_LINE 2 // line separator: "\n" (Unix) or "\r\n" (Windows)
|
||||||
#define KS_SEP_MAX 2
|
#define KS_SEP_MAX 2
|
||||||
|
|
||||||
#ifndef klib_unused
|
#ifndef klib_unused
|
||||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
||||||
#define klib_unused __attribute__ ((__unused__))
|
#define klib_unused __attribute__ ((__unused__))
|
||||||
#else
|
#else
|
||||||
#define klib_unused
|
#define klib_unused
|
||||||
#endif
|
#endif
|
||||||
#endif /* klib_unused */
|
#endif /* klib_unused */
|
||||||
|
|
||||||
#define __KS_TYPE(type_t) \
|
#define __KS_TYPE(type_t) \
|
||||||
typedef struct __kstream_t { \
|
typedef struct __kstream_t { \
|
||||||
unsigned char *buf; \
|
unsigned char *buf; \
|
||||||
int begin, end, is_eof; \
|
int begin, end, is_eof; \
|
||||||
type_t f; \
|
type_t f; \
|
||||||
} kstream_t;
|
} kstream_t;
|
||||||
|
|
||||||
#define ks_err(ks) ((ks)->end == -1)
|
#define ks_err(ks) ((ks)->end == -1)
|
||||||
#define ks_eof(ks) ((ks)->is_eof && (ks)->begin >= (ks)->end)
|
#define ks_eof(ks) ((ks)->is_eof && (ks)->begin >= (ks)->end)
|
||||||
#define ks_rewind(ks) ((ks)->is_eof = (ks)->begin = (ks)->end = 0)
|
#define ks_rewind(ks) ((ks)->is_eof = (ks)->begin = (ks)->end = 0)
|
||||||
|
|
||||||
#define __KS_BASIC(type_t, __bufsize) \
|
#define __KS_BASIC(type_t, __bufsize) \
|
||||||
static inline kstream_t *ks_init(type_t f) \
|
static inline kstream_t *ks_init(type_t f) \
|
||||||
{ \
|
{ \
|
||||||
kstream_t *ks = (kstream_t*)calloc(1, sizeof(kstream_t)); \
|
kstream_t *ks = (kstream_t*)calloc(1, sizeof(kstream_t)); \
|
||||||
ks->f = f; \
|
ks->f = f; \
|
||||||
ks->buf = (unsigned char*)malloc(__bufsize); \
|
ks->buf = (unsigned char*)malloc(__bufsize); \
|
||||||
return ks; \
|
return ks; \
|
||||||
} \
|
} \
|
||||||
static inline void ks_destroy(kstream_t *ks) \
|
static inline void ks_destroy(kstream_t *ks) \
|
||||||
{ \
|
{ \
|
||||||
if (ks) { \
|
if (ks) { \
|
||||||
free(ks->buf); \
|
free(ks->buf); \
|
||||||
free(ks); \
|
free(ks); \
|
||||||
} \
|
} \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define __KS_GETC(__read, __bufsize) \
|
#define __KS_GETC(__read, __bufsize) \
|
||||||
static inline klib_unused int ks_getc(kstream_t *ks) \
|
static inline klib_unused int ks_getc(kstream_t *ks) \
|
||||||
{ \
|
{ \
|
||||||
if (ks_err(ks)) return -3; \
|
if (ks_err(ks)) return -3; \
|
||||||
if (ks->is_eof && ks->begin >= ks->end) return -1; \
|
if (ks->is_eof && ks->begin >= ks->end) return -1; \
|
||||||
if (ks->begin >= ks->end) { \
|
if (ks->begin >= ks->end) { \
|
||||||
ks->begin = 0; \
|
ks->begin = 0; \
|
||||||
ks->end = __read(ks->f, ks->buf, __bufsize); \
|
ks->end = __read(ks->f, ks->buf, __bufsize); \
|
||||||
if (ks->end == 0) { ks->is_eof = 1; return -1;} \
|
if (ks->end == 0) { ks->is_eof = 1; return -1;} \
|
||||||
if (ks->end == -1) { ks->is_eof = 1; return -3;}\
|
if (ks->end == -1) { ks->is_eof = 1; return -3;}\
|
||||||
} \
|
} \
|
||||||
return (int)ks->buf[ks->begin++]; \
|
return (int)ks->buf[ks->begin++]; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#ifndef KSTRING_T
|
#ifndef KSTRING_T
|
||||||
#define KSTRING_T kstring_t
|
#define KSTRING_T kstring_t
|
||||||
typedef struct __kstring_t {
|
typedef struct __kstring_t {
|
||||||
size_t l, m;
|
size_t l, m;
|
||||||
char *s;
|
char *s;
|
||||||
} kstring_t;
|
} kstring_t;
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#ifndef kroundup32
|
#ifndef kroundup32
|
||||||
#define kroundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
#define kroundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#define __KS_GETUNTIL(__read, __bufsize) \
|
#define __KS_GETUNTIL(__read, __bufsize) \
|
||||||
static int ks_getuntil2(kstream_t *ks, int delimiter, kstring_t *str, int *dret, int append) \
|
static int ks_getuntil2(kstream_t *ks, int delimiter, kstring_t *str, int *dret, int append) \
|
||||||
{ \
|
{ \
|
||||||
int gotany = 0; \
|
int gotany = 0; \
|
||||||
if (dret) *dret = 0; \
|
if (dret) *dret = 0; \
|
||||||
str->l = append? str->l : 0; \
|
str->l = append? str->l : 0; \
|
||||||
for (;;) { \
|
for (;;) { \
|
||||||
int i; \
|
int i; \
|
||||||
if (ks_err(ks)) return -3; \
|
if (ks_err(ks)) return -3; \
|
||||||
if (ks->begin >= ks->end) { \
|
if (ks->begin >= ks->end) { \
|
||||||
if (!ks->is_eof) { \
|
if (!ks->is_eof) { \
|
||||||
ks->begin = 0; \
|
ks->begin = 0; \
|
||||||
ks->end = __read(ks->f, ks->buf, __bufsize); \
|
ks->end = __read(ks->f, ks->buf, __bufsize); \
|
||||||
if (ks->end == 0) { ks->is_eof = 1; break; } \
|
if (ks->end == 0) { ks->is_eof = 1; break; } \
|
||||||
if (ks->end == -1) { ks->is_eof = 1; return -3; } \
|
if (ks->end == -1) { ks->is_eof = 1; return -3; } \
|
||||||
} else break; \
|
} else break; \
|
||||||
} \
|
} \
|
||||||
if (delimiter == KS_SEP_LINE) { \
|
if (delimiter == KS_SEP_LINE) { \
|
||||||
for (i = ks->begin; i < ks->end; ++i) \
|
for (i = ks->begin; i < ks->end; ++i) \
|
||||||
if (ks->buf[i] == '\n') break; \
|
if (ks->buf[i] == '\n') break; \
|
||||||
} else if (delimiter > KS_SEP_MAX) { \
|
} else if (delimiter > KS_SEP_MAX) { \
|
||||||
for (i = ks->begin; i < ks->end; ++i) \
|
for (i = ks->begin; i < ks->end; ++i) \
|
||||||
if (ks->buf[i] == delimiter) break; \
|
if (ks->buf[i] == delimiter) break; \
|
||||||
} else if (delimiter == KS_SEP_SPACE) { \
|
} else if (delimiter == KS_SEP_SPACE) { \
|
||||||
for (i = ks->begin; i < ks->end; ++i) \
|
for (i = ks->begin; i < ks->end; ++i) \
|
||||||
if (isspace(ks->buf[i])) break; \
|
if (isspace(ks->buf[i])) break; \
|
||||||
} else if (delimiter == KS_SEP_TAB) { \
|
} else if (delimiter == KS_SEP_TAB) { \
|
||||||
for (i = ks->begin; i < ks->end; ++i) \
|
for (i = ks->begin; i < ks->end; ++i) \
|
||||||
if (isspace(ks->buf[i]) && ks->buf[i] != ' ') break; \
|
if (isspace(ks->buf[i]) && ks->buf[i] != ' ') break; \
|
||||||
} else i = 0; /* never come to here! */ \
|
} else i = 0; /* never come to here! */ \
|
||||||
if (str->m - str->l < (size_t)(i - ks->begin + 1)) { \
|
if (str->m - str->l < (size_t)(i - ks->begin + 1)) { \
|
||||||
str->m = str->l + (i - ks->begin) + 1; \
|
str->m = str->l + (i - ks->begin) + 1; \
|
||||||
kroundup32(str->m); \
|
kroundup32(str->m); \
|
||||||
str->s = (char*)realloc(str->s, str->m); \
|
str->s = (char*)realloc(str->s, str->m); \
|
||||||
} \
|
} \
|
||||||
gotany = 1; \
|
gotany = 1; \
|
||||||
memcpy(str->s + str->l, ks->buf + ks->begin, i - ks->begin); \
|
memcpy(str->s + str->l, ks->buf + ks->begin, i - ks->begin); \
|
||||||
str->l = str->l + (i - ks->begin); \
|
str->l = str->l + (i - ks->begin); \
|
||||||
ks->begin = i + 1; \
|
ks->begin = i + 1; \
|
||||||
if (i < ks->end) { \
|
if (i < ks->end) { \
|
||||||
if (dret) *dret = ks->buf[i]; \
|
if (dret) *dret = ks->buf[i]; \
|
||||||
break; \
|
break; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
if (!gotany && ks_eof(ks)) return -1; \
|
if (!gotany && ks_eof(ks)) return -1; \
|
||||||
if (str->s == 0) { \
|
if (str->s == 0) { \
|
||||||
str->m = 1; \
|
str->m = 1; \
|
||||||
str->s = (char*)calloc(1, 1); \
|
str->s = (char*)calloc(1, 1); \
|
||||||
} else if (delimiter == KS_SEP_LINE && str->l > 1 && str->s[str->l-1] == '\r') --str->l; \
|
} else if (delimiter == KS_SEP_LINE && str->l > 1 && str->s[str->l-1] == '\r') --str->l; \
|
||||||
str->s[str->l] = '\0'; \
|
str->s[str->l] = '\0'; \
|
||||||
return str->l; \
|
return str->l; \
|
||||||
} \
|
} \
|
||||||
static inline int ks_getuntil(kstream_t *ks, int delimiter, kstring_t *str, int *dret) \
|
static inline int ks_getuntil(kstream_t *ks, int delimiter, kstring_t *str, int *dret) \
|
||||||
{ return ks_getuntil2(ks, delimiter, str, dret, 0); }
|
{ return ks_getuntil2(ks, delimiter, str, dret, 0); }
|
||||||
|
|
||||||
#define KSTREAM_INIT(type_t, __read, __bufsize) \
|
#define KSTREAM_INIT(type_t, __read, __bufsize) \
|
||||||
__KS_TYPE(type_t) \
|
__KS_TYPE(type_t) \
|
||||||
__KS_BASIC(type_t, __bufsize) \
|
__KS_BASIC(type_t, __bufsize) \
|
||||||
__KS_GETC(__read, __bufsize) \
|
__KS_GETC(__read, __bufsize) \
|
||||||
__KS_GETUNTIL(__read, __bufsize)
|
__KS_GETUNTIL(__read, __bufsize)
|
||||||
|
|
||||||
#define kseq_rewind(ks) ((ks)->last_char = (ks)->f->is_eof = (ks)->f->begin = (ks)->f->end = 0)
|
#define kseq_rewind(ks) ((ks)->last_char = (ks)->f->is_eof = (ks)->f->begin = (ks)->f->end = 0)
|
||||||
|
|
||||||
#define __KSEQ_BASIC(SCOPE, type_t) \
|
#define __KSEQ_BASIC(SCOPE, type_t) \
|
||||||
SCOPE kseq_t *kseq_init(type_t fd) \
|
SCOPE kseq_t *kseq_init(type_t fd) \
|
||||||
{ \
|
{ \
|
||||||
kseq_t *s = (kseq_t*)calloc(1, sizeof(kseq_t)); \
|
kseq_t *s = (kseq_t*)calloc(1, sizeof(kseq_t)); \
|
||||||
s->f = ks_init(fd); \
|
s->f = ks_init(fd); \
|
||||||
return s; \
|
return s; \
|
||||||
} \
|
} \
|
||||||
SCOPE void kseq_destroy(kseq_t *ks) \
|
SCOPE void kseq_destroy(kseq_t *ks) \
|
||||||
{ \
|
{ \
|
||||||
if (!ks) return; \
|
if (!ks) return; \
|
||||||
free(ks->name.s); free(ks->comment.s); free(ks->seq.s); free(ks->qual.s); \
|
free(ks->name.s); free(ks->comment.s); free(ks->seq.s); free(ks->qual.s); \
|
||||||
ks_destroy(ks->f); \
|
ks_destroy(ks->f); \
|
||||||
free(ks); \
|
free(ks); \
|
||||||
}
|
}
|
||||||
|
|
||||||
/* Return value:
|
/* Return value:
|
||||||
>=0 length of the sequence (normal)
|
>=0 length of the sequence (normal)
|
||||||
-1 end-of-file
|
-1 end-of-file
|
||||||
-2 truncated quality string
|
-2 truncated quality string
|
||||||
-3 error reading stream
|
-3 error reading stream
|
||||||
*/
|
*/
|
||||||
#define __KSEQ_READ(SCOPE) \
|
#define __KSEQ_READ(SCOPE) \
|
||||||
SCOPE int kseq_read(kseq_t *seq) \
|
SCOPE int kseq_read(kseq_t *seq) \
|
||||||
{ \
|
{ \
|
||||||
int c,r; \
|
int c,r; \
|
||||||
kstream_t *ks = seq->f; \
|
kstream_t *ks = seq->f; \
|
||||||
if (seq->last_char == 0) { /* then jump to the next header line */ \
|
if (seq->last_char == 0) { /* then jump to the next header line */ \
|
||||||
while ((c = ks_getc(ks)) >= 0 && c != '>' && c != '@'); \
|
while ((c = ks_getc(ks)) >= 0 && c != '>' && c != '@'); \
|
||||||
if (c < 0) return c; /* end of file or error*/ \
|
if (c < 0) return c; /* end of file or error*/ \
|
||||||
seq->last_char = c; \
|
seq->last_char = c; \
|
||||||
} /* else: the first header char has been read in the previous call */ \
|
} /* else: the first header char has been read in the previous call */ \
|
||||||
seq->comment.l = seq->seq.l = seq->qual.l = 0; /* reset all members */ \
|
seq->comment.l = seq->seq.l = seq->qual.l = 0; /* reset all members */ \
|
||||||
if ((r=ks_getuntil(ks, 0, &seq->name, &c)) < 0) return r; /* normal exit: EOF or error */ \
|
if ((r=ks_getuntil(ks, 0, &seq->name, &c)) < 0) return r; /* normal exit: EOF or error */ \
|
||||||
if (c != '\n') ks_getuntil(ks, KS_SEP_LINE, &seq->comment, 0); /* read FASTA/Q comment */ \
|
if (c != '\n') ks_getuntil(ks, KS_SEP_LINE, &seq->comment, 0); /* read FASTA/Q comment */ \
|
||||||
if (seq->seq.s == 0) { /* we can do this in the loop below, but that is slower */ \
|
if (seq->seq.s == 0) { /* we can do this in the loop below, but that is slower */ \
|
||||||
seq->seq.m = 256; \
|
seq->seq.m = 256; \
|
||||||
seq->seq.s = (char*)malloc(seq->seq.m); \
|
seq->seq.s = (char*)malloc(seq->seq.m); \
|
||||||
} \
|
} \
|
||||||
while ((c = ks_getc(ks)) >= 0 && c != '>' && c != '+' && c != '@') { \
|
while ((c = ks_getc(ks)) >= 0 && c != '>' && c != '+' && c != '@') { \
|
||||||
if (c == '\n') continue; /* skip empty lines */ \
|
if (c == '\n') continue; /* skip empty lines */ \
|
||||||
seq->seq.s[seq->seq.l++] = c; /* this is safe: we always have enough space for 1 char */ \
|
seq->seq.s[seq->seq.l++] = c; /* this is safe: we always have enough space for 1 char */ \
|
||||||
ks_getuntil2(ks, KS_SEP_LINE, &seq->seq, 0, 1); /* read the rest of the line */ \
|
ks_getuntil2(ks, KS_SEP_LINE, &seq->seq, 0, 1); /* read the rest of the line */ \
|
||||||
} \
|
} \
|
||||||
if (c == '>' || c == '@') seq->last_char = c; /* the first header char has been read */ \
|
if (c == '>' || c == '@') seq->last_char = c; /* the first header char has been read */ \
|
||||||
if (seq->seq.l + 1 >= seq->seq.m) { /* seq->seq.s[seq->seq.l] below may be out of boundary */ \
|
if (seq->seq.l + 1 >= seq->seq.m) { /* seq->seq.s[seq->seq.l] below may be out of boundary */ \
|
||||||
seq->seq.m = seq->seq.l + 2; \
|
seq->seq.m = seq->seq.l + 2; \
|
||||||
kroundup32(seq->seq.m); /* rounded to the next closest 2^k */ \
|
kroundup32(seq->seq.m); /* rounded to the next closest 2^k */ \
|
||||||
seq->seq.s = (char*)realloc(seq->seq.s, seq->seq.m); \
|
seq->seq.s = (char*)realloc(seq->seq.s, seq->seq.m); \
|
||||||
} \
|
} \
|
||||||
seq->seq.s[seq->seq.l] = 0; /* null terminated string */ \
|
seq->seq.s[seq->seq.l] = 0; /* null terminated string */ \
|
||||||
if (c != '+') return seq->seq.l; /* FASTA */ \
|
if (c != '+') return seq->seq.l; /* FASTA */ \
|
||||||
if (seq->qual.m < seq->seq.m) { /* allocate memory for qual in case insufficient */ \
|
if (seq->qual.m < seq->seq.m) { /* allocate memory for qual in case insufficient */ \
|
||||||
seq->qual.m = seq->seq.m; \
|
seq->qual.m = seq->seq.m; \
|
||||||
seq->qual.s = (char*)realloc(seq->qual.s, seq->qual.m); \
|
seq->qual.s = (char*)realloc(seq->qual.s, seq->qual.m); \
|
||||||
} \
|
} \
|
||||||
while ((c = ks_getc(ks)) >= 0 && c != '\n'); /* skip the rest of '+' line */ \
|
while ((c = ks_getc(ks)) >= 0 && c != '\n'); /* skip the rest of '+' line */ \
|
||||||
if (c == -1) return -2; /* error: no quality string */ \
|
if (c == -1) return -2; /* error: no quality string */ \
|
||||||
while ((c = ks_getuntil2(ks, KS_SEP_LINE, &seq->qual, 0, 1) >= 0 && seq->qual.l < seq->seq.l)); \
|
while ((c = ks_getuntil2(ks, KS_SEP_LINE, &seq->qual, 0, 1) >= 0 && seq->qual.l < seq->seq.l)); \
|
||||||
if (c == -3) return -3; /* stream error */ \
|
if (c == -3) return -3; /* stream error */ \
|
||||||
seq->last_char = 0; /* we have not come to the next header line */ \
|
seq->last_char = 0; /* we have not come to the next header line */ \
|
||||||
if (seq->seq.l != seq->qual.l) return -2; /* error: qual string is of a different length */ \
|
if (seq->seq.l != seq->qual.l) return -2; /* error: qual string is of a different length */ \
|
||||||
return seq->seq.l; \
|
return seq->seq.l; \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define __KSEQ_TYPE(type_t) \
|
#define __KSEQ_TYPE(type_t) \
|
||||||
typedef struct { \
|
typedef struct { \
|
||||||
kstring_t name, comment, seq, qual; \
|
kstring_t name, comment, seq, qual; \
|
||||||
int last_char; \
|
int last_char; \
|
||||||
kstream_t *f; \
|
kstream_t *f; \
|
||||||
uint64_t ID; \
|
uint64_t ID; \
|
||||||
} kseq_t;
|
} kseq_t;
|
||||||
|
|
||||||
#define KSEQ_INIT2(SCOPE, type_t, __read) \
|
#define KSEQ_INIT2(SCOPE, type_t, __read) \
|
||||||
KSTREAM_INIT(type_t, __read, 16384) \
|
KSTREAM_INIT(type_t, __read, 16384) \
|
||||||
__KSEQ_TYPE(type_t) \
|
__KSEQ_TYPE(type_t) \
|
||||||
__KSEQ_BASIC(SCOPE, type_t) \
|
__KSEQ_BASIC(SCOPE, type_t) \
|
||||||
__KSEQ_READ(SCOPE)
|
__KSEQ_READ(SCOPE)
|
||||||
|
|
||||||
#define KSEQ_INIT(type_t, __read) KSEQ_INIT2(static klib_unused, type_t, __read)
|
#define KSEQ_INIT(type_t, __read) KSEQ_INIT2(static klib_unused, type_t, __read)
|
||||||
|
|
||||||
#define KSEQ_DECLARE(type_t) \
|
#define KSEQ_DECLARE(type_t) \
|
||||||
__KS_TYPE(type_t) \
|
__KS_TYPE(type_t) \
|
||||||
__KSEQ_TYPE(type_t) \
|
__KSEQ_TYPE(type_t) \
|
||||||
extern kseq_t *kseq_init(type_t fd); \
|
extern kseq_t *kseq_init(type_t fd); \
|
||||||
void kseq_destroy(kseq_t *ks); \
|
void kseq_destroy(kseq_t *ks); \
|
||||||
int kseq_read(kseq_t *seq);
|
int kseq_read(kseq_t *seq);
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,223 +1,223 @@
|
|||||||
/* The MIT License
|
/* The MIT License
|
||||||
|
|
||||||
Copyright (c) 2008, 2011 Attractive Chaos <attractor@live.co.uk>
|
Copyright (c) 2008, 2011 Attractive Chaos <attractor@live.co.uk>
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
a copy of this software and associated documentation files (the
|
a copy of this software and associated documentation files (the
|
||||||
"Software"), to deal in the Software without restriction, including
|
"Software"), to deal in the Software without restriction, including
|
||||||
without limitation the rights to use, copy, modify, merge, publish,
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
distribute, sublicense, and/or sell copies of the Software, and to
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
permit persons to whom the Software is furnished to do so, subject to
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
the following conditions:
|
the following conditions:
|
||||||
|
|
||||||
The above copyright notice and this permission notice shall be
|
The above copyright notice and this permission notice shall be
|
||||||
included in all copies or substantial portions of the Software.
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
SOFTWARE.
|
SOFTWARE.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
// This is a simplified version of ksort.h
|
// This is a simplified version of ksort.h
|
||||||
|
|
||||||
#ifndef AC_KSORT_H
|
#ifndef AC_KSORT_H
|
||||||
#define AC_KSORT_H
|
#define AC_KSORT_H
|
||||||
|
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <string.h>
|
#include <string.h>
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
void *left, *right;
|
void *left, *right;
|
||||||
int depth;
|
int depth;
|
||||||
} ks_isort_stack_t;
|
} ks_isort_stack_t;
|
||||||
|
|
||||||
#define member_size(type, member) sizeof(((type *)0)->member)
|
#define member_size(type, member) sizeof(((type *)0)->member)
|
||||||
|
|
||||||
#define KSORT_SWAP(type_t, a, b) { register type_t t=(a); (a)=(b); (b)=t; }
|
#define KSORT_SWAP(type_t, a, b) { register type_t t=(a); (a)=(b); (b)=t; }
|
||||||
|
|
||||||
#define KSORT_INIT(name, type_t, __sort_lt) \
|
#define KSORT_INIT(name, type_t, __sort_lt) \
|
||||||
void ks_heapdown_##name(size_t i, size_t n, type_t l[]) \
|
void ks_heapdown_##name(size_t i, size_t n, type_t l[]) \
|
||||||
{ \
|
{ \
|
||||||
size_t k = i; \
|
size_t k = i; \
|
||||||
type_t tmp = l[i]; \
|
type_t tmp = l[i]; \
|
||||||
while ((k = (k << 1) + 1) < n) { \
|
while ((k = (k << 1) + 1) < n) { \
|
||||||
if (k != n - 1 && __sort_lt(l[k], l[k+1])) ++k; \
|
if (k != n - 1 && __sort_lt(l[k], l[k+1])) ++k; \
|
||||||
if (__sort_lt(l[k], tmp)) break; \
|
if (__sort_lt(l[k], tmp)) break; \
|
||||||
l[i] = l[k]; i = k; \
|
l[i] = l[k]; i = k; \
|
||||||
} \
|
} \
|
||||||
l[i] = tmp; \
|
l[i] = tmp; \
|
||||||
} \
|
} \
|
||||||
void ks_heapup_##name(size_t n, type_t l[]) \
|
void ks_heapup_##name(size_t n, type_t l[]) \
|
||||||
{ \
|
{ \
|
||||||
size_t i, k = n - 1; \
|
size_t i, k = n - 1; \
|
||||||
type_t tmp = l[k]; \
|
type_t tmp = l[k]; \
|
||||||
while (k) { \
|
while (k) { \
|
||||||
i = (k - 1) >> 1; \
|
i = (k - 1) >> 1; \
|
||||||
if (__sort_lt(tmp, l[i])) break; \
|
if (__sort_lt(tmp, l[i])) break; \
|
||||||
l[k] = l[i]; k = i; \
|
l[k] = l[i]; k = i; \
|
||||||
} \
|
} \
|
||||||
l[k] = tmp; \
|
l[k] = tmp; \
|
||||||
} \
|
} \
|
||||||
void ks_heapmake_##name(size_t lsize, type_t l[]) \
|
void ks_heapmake_##name(size_t lsize, type_t l[]) \
|
||||||
{ \
|
{ \
|
||||||
size_t i; \
|
size_t i; \
|
||||||
for (i = (lsize >> 1) - 1; i != (size_t)(-1); --i) \
|
for (i = (lsize >> 1) - 1; i != (size_t)(-1); --i) \
|
||||||
ks_heapdown_##name(i, lsize, l); \
|
ks_heapdown_##name(i, lsize, l); \
|
||||||
} \
|
} \
|
||||||
void ks_heapsort_##name(size_t lsize, type_t l[]) \
|
void ks_heapsort_##name(size_t lsize, type_t l[]) \
|
||||||
{ \
|
{ \
|
||||||
size_t i; \
|
size_t i; \
|
||||||
for (i = lsize - 1; i > 0; --i) { \
|
for (i = lsize - 1; i > 0; --i) { \
|
||||||
type_t tmp; \
|
type_t tmp; \
|
||||||
tmp = *l; *l = l[i]; l[i] = tmp; ks_heapdown_##name(0, i, l); \
|
tmp = *l; *l = l[i]; l[i] = tmp; ks_heapdown_##name(0, i, l); \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
static inline void __ks_insertsort_##name(type_t *s, type_t *t) \
|
static inline void __ks_insertsort_##name(type_t *s, type_t *t) \
|
||||||
{ \
|
{ \
|
||||||
type_t *i, *j, swap_tmp; \
|
type_t *i, *j, swap_tmp; \
|
||||||
for (i = s + 1; i < t; ++i) \
|
for (i = s + 1; i < t; ++i) \
|
||||||
for (j = i; j > s && __sort_lt(*j, *(j-1)); --j) { \
|
for (j = i; j > s && __sort_lt(*j, *(j-1)); --j) { \
|
||||||
swap_tmp = *j; *j = *(j-1); *(j-1) = swap_tmp; \
|
swap_tmp = *j; *j = *(j-1); *(j-1) = swap_tmp; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
void ks_combsort_##name(size_t n, type_t a[]) \
|
void ks_combsort_##name(size_t n, type_t a[]) \
|
||||||
{ \
|
{ \
|
||||||
const double shrink_factor = 1.2473309501039786540366528676643; \
|
const double shrink_factor = 1.2473309501039786540366528676643; \
|
||||||
int do_swap; \
|
int do_swap; \
|
||||||
size_t gap = n; \
|
size_t gap = n; \
|
||||||
type_t tmp, *i, *j; \
|
type_t tmp, *i, *j; \
|
||||||
do { \
|
do { \
|
||||||
if (gap > 2) { \
|
if (gap > 2) { \
|
||||||
gap = (size_t)(gap / shrink_factor); \
|
gap = (size_t)(gap / shrink_factor); \
|
||||||
if (gap == 9 || gap == 10) gap = 11; \
|
if (gap == 9 || gap == 10) gap = 11; \
|
||||||
} \
|
} \
|
||||||
do_swap = 0; \
|
do_swap = 0; \
|
||||||
for (i = a; i < a + n - gap; ++i) { \
|
for (i = a; i < a + n - gap; ++i) { \
|
||||||
j = i + gap; \
|
j = i + gap; \
|
||||||
if (__sort_lt(*j, *i)) { \
|
if (__sort_lt(*j, *i)) { \
|
||||||
tmp = *i; *i = *j; *j = tmp; \
|
tmp = *i; *i = *j; *j = tmp; \
|
||||||
do_swap = 1; \
|
do_swap = 1; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
} while (do_swap || gap > 2); \
|
} while (do_swap || gap > 2); \
|
||||||
if (gap != 1) __ks_insertsort_##name(a, a + n); \
|
if (gap != 1) __ks_insertsort_##name(a, a + n); \
|
||||||
} \
|
} \
|
||||||
void ks_introsort_##name(size_t n, type_t a[]) \
|
void ks_introsort_##name(size_t n, type_t a[]) \
|
||||||
{ \
|
{ \
|
||||||
int d; \
|
int d; \
|
||||||
ks_isort_stack_t *top, *stack; \
|
ks_isort_stack_t *top, *stack; \
|
||||||
type_t rp, swap_tmp; \
|
type_t rp, swap_tmp; \
|
||||||
type_t *s, *t, *i, *j, *k; \
|
type_t *s, *t, *i, *j, *k; \
|
||||||
\
|
\
|
||||||
if (n < 1) return; \
|
if (n < 1) return; \
|
||||||
else if (n == 2) { \
|
else if (n == 2) { \
|
||||||
if (__sort_lt(a[1], a[0])) { swap_tmp = a[0]; a[0] = a[1]; a[1] = swap_tmp; } \
|
if (__sort_lt(a[1], a[0])) { swap_tmp = a[0]; a[0] = a[1]; a[1] = swap_tmp; } \
|
||||||
return; \
|
return; \
|
||||||
} \
|
} \
|
||||||
for (d = 2; 1ul<<d < n; ++d); \
|
for (d = 2; 1ul<<d < n; ++d); \
|
||||||
stack = (ks_isort_stack_t*)malloc(sizeof(ks_isort_stack_t) * ((sizeof(size_t)*d)+2)); \
|
stack = (ks_isort_stack_t*)malloc(sizeof(ks_isort_stack_t) * ((sizeof(size_t)*d)+2)); \
|
||||||
top = stack; s = a; t = a + (n-1); d <<= 1; \
|
top = stack; s = a; t = a + (n-1); d <<= 1; \
|
||||||
while (1) { \
|
while (1) { \
|
||||||
if (s < t) { \
|
if (s < t) { \
|
||||||
if (--d == 0) { \
|
if (--d == 0) { \
|
||||||
ks_combsort_##name(t - s + 1, s); \
|
ks_combsort_##name(t - s + 1, s); \
|
||||||
t = s; \
|
t = s; \
|
||||||
continue; \
|
continue; \
|
||||||
} \
|
} \
|
||||||
i = s; j = t; k = i + ((j-i)>>1) + 1; \
|
i = s; j = t; k = i + ((j-i)>>1) + 1; \
|
||||||
if (__sort_lt(*k, *i)) { \
|
if (__sort_lt(*k, *i)) { \
|
||||||
if (__sort_lt(*k, *j)) k = j; \
|
if (__sort_lt(*k, *j)) k = j; \
|
||||||
} else k = __sort_lt(*j, *i)? i : j; \
|
} else k = __sort_lt(*j, *i)? i : j; \
|
||||||
rp = *k; \
|
rp = *k; \
|
||||||
if (k != t) { swap_tmp = *k; *k = *t; *t = swap_tmp; } \
|
if (k != t) { swap_tmp = *k; *k = *t; *t = swap_tmp; } \
|
||||||
for (;;) { \
|
for (;;) { \
|
||||||
do ++i; while (__sort_lt(*i, rp)); \
|
do ++i; while (__sort_lt(*i, rp)); \
|
||||||
do --j; while (i <= j && __sort_lt(rp, *j)); \
|
do --j; while (i <= j && __sort_lt(rp, *j)); \
|
||||||
if (j <= i) break; \
|
if (j <= i) break; \
|
||||||
swap_tmp = *i; *i = *j; *j = swap_tmp; \
|
swap_tmp = *i; *i = *j; *j = swap_tmp; \
|
||||||
} \
|
} \
|
||||||
swap_tmp = *i; *i = *t; *t = swap_tmp; \
|
swap_tmp = *i; *i = *t; *t = swap_tmp; \
|
||||||
if (i-s > t-i) { \
|
if (i-s > t-i) { \
|
||||||
if (i-s > 16) { top->left = s; top->right = i-1; top->depth = d; ++top; } \
|
if (i-s > 16) { top->left = s; top->right = i-1; top->depth = d; ++top; } \
|
||||||
s = t-i > 16? i+1 : t; \
|
s = t-i > 16? i+1 : t; \
|
||||||
} else { \
|
} else { \
|
||||||
if (t-i > 16) { top->left = i+1; top->right = t; top->depth = d; ++top; } \
|
if (t-i > 16) { top->left = i+1; top->right = t; top->depth = d; ++top; } \
|
||||||
t = i-s > 16? i-1 : s; \
|
t = i-s > 16? i-1 : s; \
|
||||||
} \
|
} \
|
||||||
} else { \
|
} else { \
|
||||||
if (top == stack) { \
|
if (top == stack) { \
|
||||||
free(stack); \
|
free(stack); \
|
||||||
__ks_insertsort_##name(a, a+n); \
|
__ks_insertsort_##name(a, a+n); \
|
||||||
return; \
|
return; \
|
||||||
} else { --top; s = (type_t*)top->left; t = (type_t*)top->right; d = top->depth; } \
|
} else { --top; s = (type_t*)top->left; t = (type_t*)top->right; d = top->depth; } \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
}
|
}
|
||||||
|
|
||||||
#define ks_lt_generic(a, b) ((a) < (b))
|
#define ks_lt_generic(a, b) ((a) < (b))
|
||||||
#define ks_lt_str(a, b) (strcmp((a), (b)) < 0)
|
#define ks_lt_str(a, b) (strcmp((a), (b)) < 0)
|
||||||
|
|
||||||
typedef const char *ksstr_t;
|
typedef const char *ksstr_t;
|
||||||
|
|
||||||
#define KSORT_INIT_GENERIC(type_t) KSORT_INIT(type_t, type_t, ks_lt_generic)
|
#define KSORT_INIT_GENERIC(type_t) KSORT_INIT(type_t, type_t, ks_lt_generic)
|
||||||
#define KSORT_INIT_STR KSORT_INIT(str, ksstr_t, ks_lt_str)
|
#define KSORT_INIT_STR KSORT_INIT(str, ksstr_t, ks_lt_str)
|
||||||
|
|
||||||
#define RS_MIN_SIZE 64
|
#define RS_MIN_SIZE 64
|
||||||
|
|
||||||
#define KRADIX_SORT_INIT(name, rstype_t, rskey, sizeof_key) \
|
#define KRADIX_SORT_INIT(name, rstype_t, rskey, sizeof_key) \
|
||||||
typedef struct { \
|
typedef struct { \
|
||||||
rstype_t *b, *e; \
|
rstype_t *b, *e; \
|
||||||
} rsbucket_##name##_t; \
|
} rsbucket_##name##_t; \
|
||||||
void rs_insertsort_##name(rstype_t *beg, rstype_t *end) \
|
void rs_insertsort_##name(rstype_t *beg, rstype_t *end) \
|
||||||
{ \
|
{ \
|
||||||
rstype_t *i; \
|
rstype_t *i; \
|
||||||
for (i = beg + 1; i < end; ++i) \
|
for (i = beg + 1; i < end; ++i) \
|
||||||
if (rskey(*i) < rskey(*(i - 1))) { \
|
if (rskey(*i) < rskey(*(i - 1))) { \
|
||||||
rstype_t *j, tmp = *i; \
|
rstype_t *j, tmp = *i; \
|
||||||
for (j = i; j > beg && rskey(tmp) < rskey(*(j-1)); --j) \
|
for (j = i; j > beg && rskey(tmp) < rskey(*(j-1)); --j) \
|
||||||
*j = *(j - 1); \
|
*j = *(j - 1); \
|
||||||
*j = tmp; \
|
*j = tmp; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
void rs_sort_##name(rstype_t *beg, rstype_t *end, int n_bits, int s) \
|
void rs_sort_##name(rstype_t *beg, rstype_t *end, int n_bits, int s) \
|
||||||
{ \
|
{ \
|
||||||
rstype_t *i; \
|
rstype_t *i; \
|
||||||
int size = 1<<n_bits, m = size - 1; \
|
int size = 1<<n_bits, m = size - 1; \
|
||||||
rsbucket_##name##_t *k, b[size], *be = b + size; \
|
rsbucket_##name##_t *k, b[size], *be = b + size; \
|
||||||
for (k = b; k != be; ++k) k->b = k->e = beg; \
|
for (k = b; k != be; ++k) k->b = k->e = beg; \
|
||||||
for (i = beg; i != end; ++i) ++b[rskey(*i)>>s&m].e; \
|
for (i = beg; i != end; ++i) ++b[rskey(*i)>>s&m].e; \
|
||||||
for (k = b + 1; k != be; ++k) \
|
for (k = b + 1; k != be; ++k) \
|
||||||
k->e += (k-1)->e - beg, k->b = (k-1)->e; \
|
k->e += (k-1)->e - beg, k->b = (k-1)->e; \
|
||||||
for (k = b; k != be;) { \
|
for (k = b; k != be;) { \
|
||||||
if (k->b != k->e) { \
|
if (k->b != k->e) { \
|
||||||
rsbucket_##name##_t *l; \
|
rsbucket_##name##_t *l; \
|
||||||
if ((l = b + (rskey(*k->b)>>s&m)) != k) { \
|
if ((l = b + (rskey(*k->b)>>s&m)) != k) { \
|
||||||
rstype_t tmp = *k->b, swap; \
|
rstype_t tmp = *k->b, swap; \
|
||||||
do { \
|
do { \
|
||||||
swap = tmp; tmp = *l->b; *l->b++ = swap; \
|
swap = tmp; tmp = *l->b; *l->b++ = swap; \
|
||||||
l = b + (rskey(tmp)>>s&m); \
|
l = b + (rskey(tmp)>>s&m); \
|
||||||
} while (l != k); \
|
} while (l != k); \
|
||||||
*k->b++ = tmp; \
|
*k->b++ = tmp; \
|
||||||
} else ++k->b; \
|
} else ++k->b; \
|
||||||
} else ++k; \
|
} else ++k; \
|
||||||
} \
|
} \
|
||||||
for (b->b = beg, k = b + 1; k != be; ++k) k->b = (k-1)->e; \
|
for (b->b = beg, k = b + 1; k != be; ++k) k->b = (k-1)->e; \
|
||||||
if (s) { \
|
if (s) { \
|
||||||
s = s > n_bits? s - n_bits : 0; \
|
s = s > n_bits? s - n_bits : 0; \
|
||||||
for (k = b; k != be; ++k) \
|
for (k = b; k != be; ++k) \
|
||||||
if (k->e - k->b > RS_MIN_SIZE) rs_sort_##name(k->b, k->e, n_bits, s); \
|
if (k->e - k->b > RS_MIN_SIZE) rs_sort_##name(k->b, k->e, n_bits, s); \
|
||||||
else if (k->e - k->b > 1) rs_insertsort_##name(k->b, k->e); \
|
else if (k->e - k->b > 1) rs_insertsort_##name(k->b, k->e); \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
void radix_sort_##name(rstype_t *beg, rstype_t *end) \
|
void radix_sort_##name(rstype_t *beg, rstype_t *end) \
|
||||||
{ \
|
{ \
|
||||||
if (end - beg <= RS_MIN_SIZE) rs_insertsort_##name(beg, end); \
|
if (end - beg <= RS_MIN_SIZE) rs_insertsort_##name(beg, end); \
|
||||||
else rs_sort_##name(beg, end, 8, sizeof_key * 8 - 8); \
|
else rs_sort_##name(beg, end, 8, sizeof_key * 8 - 8); \
|
||||||
}
|
}
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,178 +1,178 @@
|
|||||||
#ifndef KSW2_H_
|
#ifndef KSW2_H_
|
||||||
#define KSW2_H_
|
#define KSW2_H_
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
|
|
||||||
#define KSW_NEG_INF -0x40000000
|
#define KSW_NEG_INF -0x40000000
|
||||||
|
|
||||||
#define KSW_EZ_SCORE_ONLY 0x01 // don't record alignment path/cigar
|
#define KSW_EZ_SCORE_ONLY 0x01 // don't record alignment path/cigar
|
||||||
#define KSW_EZ_RIGHT 0x02 // right-align gaps
|
#define KSW_EZ_RIGHT 0x02 // right-align gaps
|
||||||
#define KSW_EZ_GENERIC_SC 0x04 // without this flag: match/mismatch only; last symbol is a wildcard
|
#define KSW_EZ_GENERIC_SC 0x04 // without this flag: match/mismatch only; last symbol is a wildcard
|
||||||
#define KSW_EZ_APPROX_MAX 0x08 // approximate max; this is faster with sse
|
#define KSW_EZ_APPROX_MAX 0x08 // approximate max; this is faster with sse
|
||||||
#define KSW_EZ_APPROX_DROP 0x10 // approximate Z-drop; faster with sse
|
#define KSW_EZ_APPROX_DROP 0x10 // approximate Z-drop; faster with sse
|
||||||
#define KSW_EZ_EXTZ_ONLY 0x40 // only perform extension
|
#define KSW_EZ_EXTZ_ONLY 0x40 // only perform extension
|
||||||
#define KSW_EZ_REV_CIGAR 0x80 // reverse CIGAR in the output
|
#define KSW_EZ_REV_CIGAR 0x80 // reverse CIGAR in the output
|
||||||
#define KSW_EZ_SPLICE_FOR 0x100
|
#define KSW_EZ_SPLICE_FOR 0x100
|
||||||
#define KSW_EZ_SPLICE_REV 0x200
|
#define KSW_EZ_SPLICE_REV 0x200
|
||||||
#define KSW_EZ_SPLICE_FLANK 0x400
|
#define KSW_EZ_SPLICE_FLANK 0x400
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
extern "C" {
|
extern "C" {
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint32_t max:31, zdropped:1;
|
uint32_t max:31, zdropped:1;
|
||||||
int max_q, max_t; // max extension coordinate
|
int max_q, max_t; // max extension coordinate
|
||||||
int mqe, mqe_t; // max score when reaching the end of query
|
int mqe, mqe_t; // max score when reaching the end of query
|
||||||
int mte, mte_q; // max score when reaching the end of target
|
int mte, mte_q; // max score when reaching the end of target
|
||||||
int score; // max score reaching both ends; may be KSW_NEG_INF
|
int score; // max score reaching both ends; may be KSW_NEG_INF
|
||||||
int m_cigar, n_cigar;
|
int m_cigar, n_cigar;
|
||||||
int reach_end;
|
int reach_end;
|
||||||
uint32_t *cigar;
|
uint32_t *cigar;
|
||||||
} ksw_extz_t;
|
} ksw_extz_t;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* NW-like extension
|
* NW-like extension
|
||||||
*
|
*
|
||||||
* @param km memory pool, when used with kalloc
|
* @param km memory pool, when used with kalloc
|
||||||
* @param qlen query length
|
* @param qlen query length
|
||||||
* @param query query sequence with 0 <= query[i] < m
|
* @param query query sequence with 0 <= query[i] < m
|
||||||
* @param tlen target length
|
* @param tlen target length
|
||||||
* @param target target sequence with 0 <= target[i] < m
|
* @param target target sequence with 0 <= target[i] < m
|
||||||
* @param m number of residue types
|
* @param m number of residue types
|
||||||
* @param mat m*m scoring mattrix in one-dimension array
|
* @param mat m*m scoring mattrix in one-dimension array
|
||||||
* @param gapo gap open penalty; a gap of length l cost "-(gapo+l*gape)"
|
* @param gapo gap open penalty; a gap of length l cost "-(gapo+l*gape)"
|
||||||
* @param gape gap extension penalty
|
* @param gape gap extension penalty
|
||||||
* @param w band width (<0 to disable)
|
* @param w band width (<0 to disable)
|
||||||
* @param zdrop off-diagonal drop-off to stop extension (positive; <0 to disable)
|
* @param zdrop off-diagonal drop-off to stop extension (positive; <0 to disable)
|
||||||
* @param flag flag (see KSW_EZ_* macros)
|
* @param flag flag (see KSW_EZ_* macros)
|
||||||
* @param ez (out) scores and cigar
|
* @param ez (out) scores and cigar
|
||||||
*/
|
*/
|
||||||
void ksw_extz(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_extz(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t q, int8_t e, int w, int zdrop, int flag, ksw_extz_t *ez);
|
int8_t q, int8_t e, int w, int zdrop, int flag, ksw_extz_t *ez);
|
||||||
|
|
||||||
void ksw_extz2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_extz2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||||
|
|
||||||
void ksw_extd(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_extd(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t gape2, int w, int zdrop, int flag, ksw_extz_t *ez);
|
int8_t gapo, int8_t gape, int8_t gapo2, int8_t gape2, int w, int zdrop, int flag, ksw_extz_t *ez);
|
||||||
|
|
||||||
void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t gape2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
int8_t gapo, int8_t gape, int8_t gapo2, int8_t gape2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||||
|
|
||||||
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t noncan, int zdrop, int flag, ksw_extz_t *ez);
|
int8_t gapo, int8_t gape, int8_t gapo2, int8_t noncan, int zdrop, int flag, ksw_extz_t *ez);
|
||||||
|
|
||||||
void ksw_extf2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t mch, int8_t mis, int8_t e, int w, int xdrop, ksw_extz_t *ez);
|
void ksw_extf2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t mch, int8_t mis, int8_t e, int w, int xdrop, ksw_extz_t *ez);
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Global alignment
|
* Global alignment
|
||||||
*
|
*
|
||||||
* (first 10 parameters identical to ksw_extz_sse())
|
* (first 10 parameters identical to ksw_extz_sse())
|
||||||
* @param m_cigar (modified) max CIGAR length; feed 0 if cigar==0
|
* @param m_cigar (modified) max CIGAR length; feed 0 if cigar==0
|
||||||
* @param n_cigar (out) number of CIGAR elements
|
* @param n_cigar (out) number of CIGAR elements
|
||||||
* @param cigar (out) BAM-encoded CIGAR; caller need to deallocate with kfree(km, )
|
* @param cigar (out) BAM-encoded CIGAR; caller need to deallocate with kfree(km, )
|
||||||
*
|
*
|
||||||
* @return score of the alignment
|
* @return score of the alignment
|
||||||
*/
|
*/
|
||||||
int ksw_gg(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t gapo, int8_t gape, int w, int *m_cigar_, int *n_cigar_, uint32_t **cigar_);
|
int ksw_gg(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t gapo, int8_t gape, int w, int *m_cigar_, int *n_cigar_, uint32_t **cigar_);
|
||||||
int ksw_gg2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t gapo, int8_t gape, int w, int *m_cigar_, int *n_cigar_, uint32_t **cigar_);
|
int ksw_gg2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t gapo, int8_t gape, int w, int *m_cigar_, int *n_cigar_, uint32_t **cigar_);
|
||||||
int ksw_gg2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t gapo, int8_t gape, int w, int *m_cigar_, int *n_cigar_, uint32_t **cigar_);
|
int ksw_gg2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t gapo, int8_t gape, int w, int *m_cigar_, int *n_cigar_, uint32_t **cigar_);
|
||||||
|
|
||||||
void *ksw_ll_qinit(void *km, int size, int qlen, const uint8_t *query, int m, const int8_t *mat);
|
void *ksw_ll_qinit(void *km, int size, int qlen, const uint8_t *query, int m, const int8_t *mat);
|
||||||
int ksw_ll_i16(void *q, int tlen, const uint8_t *target, int gapo, int gape, int *qe, int *te);
|
int ksw_ll_i16(void *q, int tlen, const uint8_t *target, int gapo, int gape, int *qe, int *te);
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/************************************
|
/************************************
|
||||||
*** Private macros and functions ***
|
*** Private macros and functions ***
|
||||||
************************************/
|
************************************/
|
||||||
|
|
||||||
#ifdef HAVE_KALLOC
|
#ifdef HAVE_KALLOC
|
||||||
#include "kalloc.h"
|
#include "kalloc.h"
|
||||||
#else
|
#else
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#define kmalloc(km, size) malloc((size))
|
#define kmalloc(km, size) malloc((size))
|
||||||
#define kcalloc(km, count, size) calloc((count), (size))
|
#define kcalloc(km, count, size) calloc((count), (size))
|
||||||
#define krealloc(km, ptr, size) realloc((ptr), (size))
|
#define krealloc(km, ptr, size) realloc((ptr), (size))
|
||||||
#define kfree(km, ptr) free((ptr))
|
#define kfree(km, ptr) free((ptr))
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
static inline uint32_t *ksw_push_cigar(void *km, int *n_cigar, int *m_cigar, uint32_t *cigar, uint32_t op, int len)
|
static inline uint32_t *ksw_push_cigar(void *km, int *n_cigar, int *m_cigar, uint32_t *cigar, uint32_t op, int len)
|
||||||
{
|
{
|
||||||
if (*n_cigar == 0 || op != (cigar[(*n_cigar) - 1]&0xf)) {
|
if (*n_cigar == 0 || op != (cigar[(*n_cigar) - 1]&0xf)) {
|
||||||
if (*n_cigar == *m_cigar) {
|
if (*n_cigar == *m_cigar) {
|
||||||
*m_cigar = *m_cigar? (*m_cigar)<<1 : 4;
|
*m_cigar = *m_cigar? (*m_cigar)<<1 : 4;
|
||||||
cigar = (uint32_t*)krealloc(km, cigar, (*m_cigar) << 2);
|
cigar = (uint32_t*)krealloc(km, cigar, (*m_cigar) << 2);
|
||||||
}
|
}
|
||||||
cigar[(*n_cigar)++] = len<<4 | op;
|
cigar[(*n_cigar)++] = len<<4 | op;
|
||||||
} else cigar[(*n_cigar)-1] += len<<4;
|
} else cigar[(*n_cigar)-1] += len<<4;
|
||||||
return cigar;
|
return cigar;
|
||||||
}
|
}
|
||||||
|
|
||||||
// In the backtrack matrix, value p[] has the following structure:
|
// In the backtrack matrix, value p[] has the following structure:
|
||||||
// bit 0-2: which type gets the max - 0 for H, 1 for E, 2 for F, 3 for \tilde{E} and 4 for \tilde{F}
|
// bit 0-2: which type gets the max - 0 for H, 1 for E, 2 for F, 3 for \tilde{E} and 4 for \tilde{F}
|
||||||
// bit 3/0x08: 1 if a continuation on the E state (bit 5/0x20 for a continuation on \tilde{E})
|
// bit 3/0x08: 1 if a continuation on the E state (bit 5/0x20 for a continuation on \tilde{E})
|
||||||
// bit 4/0x10: 1 if a continuation on the F state (bit 6/0x40 for a continuation on \tilde{F})
|
// bit 4/0x10: 1 if a continuation on the F state (bit 6/0x40 for a continuation on \tilde{F})
|
||||||
static inline void ksw_backtrack(void *km, int is_rot, int is_rev, int min_intron_len, const uint8_t *p, const int *off, const int *off_end, int n_col, int i0, int j0,
|
static inline void ksw_backtrack(void *km, int is_rot, int is_rev, int min_intron_len, const uint8_t *p, const int *off, const int *off_end, int n_col, int i0, int j0,
|
||||||
int *m_cigar_, int *n_cigar_, uint32_t **cigar_)
|
int *m_cigar_, int *n_cigar_, uint32_t **cigar_)
|
||||||
{ // p[] - lower 3 bits: which type gets the max; bit
|
{ // p[] - lower 3 bits: which type gets the max; bit
|
||||||
int n_cigar = 0, m_cigar = *m_cigar_, i = i0, j = j0, r, state = 0;
|
int n_cigar = 0, m_cigar = *m_cigar_, i = i0, j = j0, r, state = 0;
|
||||||
uint32_t *cigar = *cigar_, tmp;
|
uint32_t *cigar = *cigar_, tmp;
|
||||||
while (i >= 0 && j >= 0) { // at the beginning of the loop, _state_ tells us which state to check
|
while (i >= 0 && j >= 0) { // at the beginning of the loop, _state_ tells us which state to check
|
||||||
int force_state = -1;
|
int force_state = -1;
|
||||||
if (is_rot) {
|
if (is_rot) {
|
||||||
r = i + j;
|
r = i + j;
|
||||||
if (i < off[r]) force_state = 2;
|
if (i < off[r]) force_state = 2;
|
||||||
if (off_end && i > off_end[r]) force_state = 1;
|
if (off_end && i > off_end[r]) force_state = 1;
|
||||||
tmp = force_state < 0? p[(size_t)r * n_col + i - off[r]] : 0;
|
tmp = force_state < 0? p[(size_t)r * n_col + i - off[r]] : 0;
|
||||||
} else {
|
} else {
|
||||||
if (j < off[i]) force_state = 2;
|
if (j < off[i]) force_state = 2;
|
||||||
if (off_end && j > off_end[i]) force_state = 1;
|
if (off_end && j > off_end[i]) force_state = 1;
|
||||||
tmp = force_state < 0? p[(size_t)i * n_col + j - off[i]] : 0;
|
tmp = force_state < 0? p[(size_t)i * n_col + j - off[i]] : 0;
|
||||||
}
|
}
|
||||||
if (state == 0) state = tmp & 7; // if requesting the H state, find state one maximizes it.
|
if (state == 0) state = tmp & 7; // if requesting the H state, find state one maximizes it.
|
||||||
else if (!(tmp >> (state + 2) & 1)) state = 0; // if requesting other states, _state_ stays the same if it is a continuation; otherwise, set to H
|
else if (!(tmp >> (state + 2) & 1)) state = 0; // if requesting other states, _state_ stays the same if it is a continuation; otherwise, set to H
|
||||||
if (state == 0) state = tmp & 7; // TODO: probably this line can be merged into the "else if" line right above; not 100% sure
|
if (state == 0) state = tmp & 7; // TODO: probably this line can be merged into the "else if" line right above; not 100% sure
|
||||||
if (force_state >= 0) state = force_state;
|
if (force_state >= 0) state = force_state;
|
||||||
if (state == 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 0, 1), --i, --j; // match
|
if (state == 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 0, 1), --i, --j; // match
|
||||||
else if (state == 1 || (state == 3 && min_intron_len <= 0)) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 2, 1), --i; // deletion
|
else if (state == 1 || (state == 3 && min_intron_len <= 0)) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 2, 1), --i; // deletion
|
||||||
else if (state == 3 && min_intron_len > 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 3, 1), --i; // intron
|
else if (state == 3 && min_intron_len > 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 3, 1), --i; // intron
|
||||||
else cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, 1), --j; // insertion
|
else cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, 1), --j; // insertion
|
||||||
}
|
}
|
||||||
if (i >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, min_intron_len > 0 && i >= min_intron_len? 3 : 2, i + 1); // first deletion
|
if (i >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, min_intron_len > 0 && i >= min_intron_len? 3 : 2, i + 1); // first deletion
|
||||||
if (j >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, j + 1); // first insertion
|
if (j >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, j + 1); // first insertion
|
||||||
if (!is_rev)
|
if (!is_rev)
|
||||||
for (i = 0; i < n_cigar>>1; ++i) // reverse CIGAR
|
for (i = 0; i < n_cigar>>1; ++i) // reverse CIGAR
|
||||||
tmp = cigar[i], cigar[i] = cigar[n_cigar-1-i], cigar[n_cigar-1-i] = tmp;
|
tmp = cigar[i], cigar[i] = cigar[n_cigar-1-i], cigar[n_cigar-1-i] = tmp;
|
||||||
*m_cigar_ = m_cigar, *n_cigar_ = n_cigar, *cigar_ = cigar;
|
*m_cigar_ = m_cigar, *n_cigar_ = n_cigar, *cigar_ = cigar;
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline void ksw_reset_extz(ksw_extz_t *ez)
|
static inline void ksw_reset_extz(ksw_extz_t *ez)
|
||||||
{
|
{
|
||||||
ez->max_q = ez->max_t = ez->mqe_t = ez->mte_q = -1;
|
ez->max_q = ez->max_t = ez->mqe_t = ez->mte_q = -1;
|
||||||
ez->max = 0, ez->score = ez->mqe = ez->mte = KSW_NEG_INF;
|
ez->max = 0, ez->score = ez->mqe = ez->mte = KSW_NEG_INF;
|
||||||
ez->n_cigar = 0, ez->zdropped = 0, ez->reach_end = 0;
|
ez->n_cigar = 0, ez->zdropped = 0, ez->reach_end = 0;
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline int ksw_apply_zdrop(ksw_extz_t *ez, int is_rot, int32_t H, int a, int b, int zdrop, int8_t e)
|
static inline int ksw_apply_zdrop(ksw_extz_t *ez, int is_rot, int32_t H, int a, int b, int zdrop, int8_t e)
|
||||||
{
|
{
|
||||||
int r, t;
|
int r, t;
|
||||||
if (is_rot) r = a, t = b;
|
if (is_rot) r = a, t = b;
|
||||||
else r = a + b, t = a;
|
else r = a + b, t = a;
|
||||||
if (H > (int32_t)ez->max) {
|
if (H > (int32_t)ez->max) {
|
||||||
ez->max = H, ez->max_t = t, ez->max_q = r - t;
|
ez->max = H, ez->max_t = t, ez->max_q = r - t;
|
||||||
} else if (t >= ez->max_t && r - t >= ez->max_q) {
|
} else if (t >= ez->max_t && r - t >= ez->max_q) {
|
||||||
int tl = t - ez->max_t, ql = (r - t) - ez->max_q, l;
|
int tl = t - ez->max_t, ql = (r - t) - ez->max_q, l;
|
||||||
l = tl > ql? tl - ql : ql - tl;
|
l = tl > ql? tl - ql : ql - tl;
|
||||||
if (zdrop >= 0 && ez->max - H > zdrop + l * e) {
|
if (zdrop >= 0 && ez->max - H > zdrop + l * e) {
|
||||||
ez->zdropped = 1;
|
ez->zdropped = 1;
|
||||||
return 1;
|
return 1;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
+305
-305
@@ -1,305 +1,305 @@
|
|||||||
#include <string.h>
|
#include <string.h>
|
||||||
#include <assert.h>
|
#include <assert.h>
|
||||||
#include "ksw2.h"
|
#include "ksw2.h"
|
||||||
|
|
||||||
#ifdef __SSE2__
|
#ifdef __SSE2__
|
||||||
#include <emmintrin.h>
|
#include <emmintrin.h>
|
||||||
|
|
||||||
#ifdef KSW_SSE2_ONLY
|
#ifdef KSW_SSE2_ONLY
|
||||||
#undef __SSE4_1__
|
#undef __SSE4_1__
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#ifdef __SSE4_1__
|
#ifdef __SSE4_1__
|
||||||
#include <smmintrin.h>
|
#include <smmintrin.h>
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#ifdef KSW_CPU_DISPATCH
|
#ifdef KSW_CPU_DISPATCH
|
||||||
#ifdef __SSE4_1__
|
#ifdef __SSE4_1__
|
||||||
void ksw_extz2_sse41(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez)
|
void ksw_extz2_sse41(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez)
|
||||||
#else
|
#else
|
||||||
void ksw_extz2_sse2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez)
|
void ksw_extz2_sse2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez)
|
||||||
#endif
|
#endif
|
||||||
#else
|
#else
|
||||||
void ksw_extz2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez)
|
void ksw_extz2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez)
|
||||||
#endif // ~KSW_CPU_DISPATCH
|
#endif // ~KSW_CPU_DISPATCH
|
||||||
{
|
{
|
||||||
#define __dp_code_block1 \
|
#define __dp_code_block1 \
|
||||||
z = _mm_add_epi8(_mm_load_si128(&s[t]), qe2_); \
|
z = _mm_add_epi8(_mm_load_si128(&s[t]), qe2_); \
|
||||||
xt1 = _mm_load_si128(&x[t]); /* xt1 <- x[r-1][t..t+15] */ \
|
xt1 = _mm_load_si128(&x[t]); /* xt1 <- x[r-1][t..t+15] */ \
|
||||||
tmp = _mm_srli_si128(xt1, 15); /* tmp <- x[r-1][t+15] */ \
|
tmp = _mm_srli_si128(xt1, 15); /* tmp <- x[r-1][t+15] */ \
|
||||||
xt1 = _mm_or_si128(_mm_slli_si128(xt1, 1), x1_); /* xt1 <- x[r-1][t-1..t+14] */ \
|
xt1 = _mm_or_si128(_mm_slli_si128(xt1, 1), x1_); /* xt1 <- x[r-1][t-1..t+14] */ \
|
||||||
x1_ = tmp; \
|
x1_ = tmp; \
|
||||||
vt1 = _mm_load_si128(&v[t]); /* vt1 <- v[r-1][t..t+15] */ \
|
vt1 = _mm_load_si128(&v[t]); /* vt1 <- v[r-1][t..t+15] */ \
|
||||||
tmp = _mm_srli_si128(vt1, 15); /* tmp <- v[r-1][t+15] */ \
|
tmp = _mm_srli_si128(vt1, 15); /* tmp <- v[r-1][t+15] */ \
|
||||||
vt1 = _mm_or_si128(_mm_slli_si128(vt1, 1), v1_); /* vt1 <- v[r-1][t-1..t+14] */ \
|
vt1 = _mm_or_si128(_mm_slli_si128(vt1, 1), v1_); /* vt1 <- v[r-1][t-1..t+14] */ \
|
||||||
v1_ = tmp; \
|
v1_ = tmp; \
|
||||||
a = _mm_add_epi8(xt1, vt1); /* a <- x[r-1][t-1..t+14] + v[r-1][t-1..t+14] */ \
|
a = _mm_add_epi8(xt1, vt1); /* a <- x[r-1][t-1..t+14] + v[r-1][t-1..t+14] */ \
|
||||||
ut = _mm_load_si128(&u[t]); /* ut <- u[t..t+15] */ \
|
ut = _mm_load_si128(&u[t]); /* ut <- u[t..t+15] */ \
|
||||||
b = _mm_add_epi8(_mm_load_si128(&y[t]), ut); /* b <- y[r-1][t..t+15] + u[r-1][t..t+15] */
|
b = _mm_add_epi8(_mm_load_si128(&y[t]), ut); /* b <- y[r-1][t..t+15] + u[r-1][t..t+15] */
|
||||||
|
|
||||||
#define __dp_code_block2 \
|
#define __dp_code_block2 \
|
||||||
z = _mm_max_epu8(z, b); /* z = max(z, b); this works because both are non-negative */ \
|
z = _mm_max_epu8(z, b); /* z = max(z, b); this works because both are non-negative */ \
|
||||||
z = _mm_min_epu8(z, max_sc_); \
|
z = _mm_min_epu8(z, max_sc_); \
|
||||||
_mm_store_si128(&u[t], _mm_sub_epi8(z, vt1)); /* u[r][t..t+15] <- z - v[r-1][t-1..t+14] */ \
|
_mm_store_si128(&u[t], _mm_sub_epi8(z, vt1)); /* u[r][t..t+15] <- z - v[r-1][t-1..t+14] */ \
|
||||||
_mm_store_si128(&v[t], _mm_sub_epi8(z, ut)); /* v[r][t..t+15] <- z - u[r-1][t..t+15] */ \
|
_mm_store_si128(&v[t], _mm_sub_epi8(z, ut)); /* v[r][t..t+15] <- z - u[r-1][t..t+15] */ \
|
||||||
z = _mm_sub_epi8(z, q_); \
|
z = _mm_sub_epi8(z, q_); \
|
||||||
a = _mm_sub_epi8(a, z); \
|
a = _mm_sub_epi8(a, z); \
|
||||||
b = _mm_sub_epi8(b, z);
|
b = _mm_sub_epi8(b, z);
|
||||||
|
|
||||||
int r, t, qe = q + e, n_col_, *off = 0, *off_end = 0, tlen_, qlen_, last_st, last_en, wl, wr, max_sc, min_sc;
|
int r, t, qe = q + e, n_col_, *off = 0, *off_end = 0, tlen_, qlen_, last_st, last_en, wl, wr, max_sc, min_sc;
|
||||||
int with_cigar = !(flag&KSW_EZ_SCORE_ONLY), approx_max = !!(flag&KSW_EZ_APPROX_MAX);
|
int with_cigar = !(flag&KSW_EZ_SCORE_ONLY), approx_max = !!(flag&KSW_EZ_APPROX_MAX);
|
||||||
int32_t *H = 0, H0 = 0, last_H0_t = 0;
|
int32_t *H = 0, H0 = 0, last_H0_t = 0;
|
||||||
uint8_t *qr, *sf, *mem, *mem2 = 0;
|
uint8_t *qr, *sf, *mem, *mem2 = 0;
|
||||||
__m128i q_, qe2_, zero_, flag1_, flag2_, flag8_, flag16_, sc_mch_, sc_mis_, sc_N_, m1_, max_sc_;
|
__m128i q_, qe2_, zero_, flag1_, flag2_, flag8_, flag16_, sc_mch_, sc_mis_, sc_N_, m1_, max_sc_;
|
||||||
__m128i *u, *v, *x, *y, *s, *p = 0;
|
__m128i *u, *v, *x, *y, *s, *p = 0;
|
||||||
|
|
||||||
ksw_reset_extz(ez);
|
ksw_reset_extz(ez);
|
||||||
if (m <= 0 || qlen <= 0 || tlen <= 0) return;
|
if (m <= 0 || qlen <= 0 || tlen <= 0) return;
|
||||||
|
|
||||||
zero_ = _mm_set1_epi8(0);
|
zero_ = _mm_set1_epi8(0);
|
||||||
q_ = _mm_set1_epi8(q);
|
q_ = _mm_set1_epi8(q);
|
||||||
qe2_ = _mm_set1_epi8((q + e) * 2);
|
qe2_ = _mm_set1_epi8((q + e) * 2);
|
||||||
flag1_ = _mm_set1_epi8(1);
|
flag1_ = _mm_set1_epi8(1);
|
||||||
flag2_ = _mm_set1_epi8(2);
|
flag2_ = _mm_set1_epi8(2);
|
||||||
flag8_ = _mm_set1_epi8(0x08);
|
flag8_ = _mm_set1_epi8(0x08);
|
||||||
flag16_ = _mm_set1_epi8(0x10);
|
flag16_ = _mm_set1_epi8(0x10);
|
||||||
sc_mch_ = _mm_set1_epi8(mat[0]);
|
sc_mch_ = _mm_set1_epi8(mat[0]);
|
||||||
sc_mis_ = _mm_set1_epi8(mat[1]);
|
sc_mis_ = _mm_set1_epi8(mat[1]);
|
||||||
sc_N_ = mat[m*m-1] == 0? _mm_set1_epi8(-e) : _mm_set1_epi8(mat[m*m-1]);
|
sc_N_ = mat[m*m-1] == 0? _mm_set1_epi8(-e) : _mm_set1_epi8(mat[m*m-1]);
|
||||||
m1_ = _mm_set1_epi8(m - 1); // wildcard
|
m1_ = _mm_set1_epi8(m - 1); // wildcard
|
||||||
max_sc_ = _mm_set1_epi8(mat[0] + (q + e) * 2);
|
max_sc_ = _mm_set1_epi8(mat[0] + (q + e) * 2);
|
||||||
|
|
||||||
if (w < 0) w = tlen > qlen? tlen : qlen;
|
if (w < 0) w = tlen > qlen? tlen : qlen;
|
||||||
wl = wr = w;
|
wl = wr = w;
|
||||||
tlen_ = (tlen + 15) / 16;
|
tlen_ = (tlen + 15) / 16;
|
||||||
n_col_ = qlen < tlen? qlen : tlen;
|
n_col_ = qlen < tlen? qlen : tlen;
|
||||||
n_col_ = ((n_col_ < w + 1? n_col_ : w + 1) + 15) / 16 + 1;
|
n_col_ = ((n_col_ < w + 1? n_col_ : w + 1) + 15) / 16 + 1;
|
||||||
qlen_ = (qlen + 15) / 16;
|
qlen_ = (qlen + 15) / 16;
|
||||||
for (t = 1, max_sc = mat[0], min_sc = mat[1]; t < m * m; ++t) {
|
for (t = 1, max_sc = mat[0], min_sc = mat[1]; t < m * m; ++t) {
|
||||||
max_sc = max_sc > mat[t]? max_sc : mat[t];
|
max_sc = max_sc > mat[t]? max_sc : mat[t];
|
||||||
min_sc = min_sc < mat[t]? min_sc : mat[t];
|
min_sc = min_sc < mat[t]? min_sc : mat[t];
|
||||||
}
|
}
|
||||||
if (-min_sc > 2 * (q + e)) return; // otherwise, we won't see any mismatches
|
if (-min_sc > 2 * (q + e)) return; // otherwise, we won't see any mismatches
|
||||||
|
|
||||||
mem = (uint8_t*)kcalloc(km, tlen_ * 6 + qlen_ + 1, 16);
|
mem = (uint8_t*)kcalloc(km, tlen_ * 6 + qlen_ + 1, 16);
|
||||||
u = (__m128i*)(((size_t)mem + 15) >> 4 << 4); // 16-byte aligned
|
u = (__m128i*)(((size_t)mem + 15) >> 4 << 4); // 16-byte aligned
|
||||||
v = u + tlen_, x = v + tlen_, y = x + tlen_, s = y + tlen_, sf = (uint8_t*)(s + tlen_), qr = sf + tlen_ * 16;
|
v = u + tlen_, x = v + tlen_, y = x + tlen_, s = y + tlen_, sf = (uint8_t*)(s + tlen_), qr = sf + tlen_ * 16;
|
||||||
if (!approx_max) {
|
if (!approx_max) {
|
||||||
H = (int32_t*)kmalloc(km, tlen_ * 16 * 4);
|
H = (int32_t*)kmalloc(km, tlen_ * 16 * 4);
|
||||||
for (t = 0; t < tlen_ * 16; ++t) H[t] = KSW_NEG_INF;
|
for (t = 0; t < tlen_ * 16; ++t) H[t] = KSW_NEG_INF;
|
||||||
}
|
}
|
||||||
if (with_cigar) {
|
if (with_cigar) {
|
||||||
mem2 = (uint8_t*)kmalloc(km, ((size_t)(qlen + tlen - 1) * n_col_ + 1) * 16);
|
mem2 = (uint8_t*)kmalloc(km, ((size_t)(qlen + tlen - 1) * n_col_ + 1) * 16);
|
||||||
p = (__m128i*)(((size_t)mem2 + 15) >> 4 << 4);
|
p = (__m128i*)(((size_t)mem2 + 15) >> 4 << 4);
|
||||||
off = (int*)kmalloc(km, (qlen + tlen - 1) * sizeof(int) * 2);
|
off = (int*)kmalloc(km, (qlen + tlen - 1) * sizeof(int) * 2);
|
||||||
off_end = off + qlen + tlen - 1;
|
off_end = off + qlen + tlen - 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
for (t = 0; t < qlen; ++t) qr[t] = query[qlen - 1 - t];
|
for (t = 0; t < qlen; ++t) qr[t] = query[qlen - 1 - t];
|
||||||
memcpy(sf, target, tlen);
|
memcpy(sf, target, tlen);
|
||||||
|
|
||||||
for (r = 0, last_st = last_en = -1; r < qlen + tlen - 1; ++r) {
|
for (r = 0, last_st = last_en = -1; r < qlen + tlen - 1; ++r) {
|
||||||
int st = 0, en = tlen - 1, st0, en0, st_, en_;
|
int st = 0, en = tlen - 1, st0, en0, st_, en_;
|
||||||
int8_t x1, v1;
|
int8_t x1, v1;
|
||||||
uint8_t *qrr = qr + (qlen - 1 - r), *u8 = (uint8_t*)u, *v8 = (uint8_t*)v;
|
uint8_t *qrr = qr + (qlen - 1 - r), *u8 = (uint8_t*)u, *v8 = (uint8_t*)v;
|
||||||
__m128i x1_, v1_;
|
__m128i x1_, v1_;
|
||||||
// find the boundaries
|
// find the boundaries
|
||||||
if (st < r - qlen + 1) st = r - qlen + 1;
|
if (st < r - qlen + 1) st = r - qlen + 1;
|
||||||
if (en > r) en = r;
|
if (en > r) en = r;
|
||||||
if (st < (r-wr+1)>>1) st = (r-wr+1)>>1; // take the ceil
|
if (st < (r-wr+1)>>1) st = (r-wr+1)>>1; // take the ceil
|
||||||
if (en > (r+wl)>>1) en = (r+wl)>>1; // take the floor
|
if (en > (r+wl)>>1) en = (r+wl)>>1; // take the floor
|
||||||
if (st > en) {
|
if (st > en) {
|
||||||
ez->zdropped = 1;
|
ez->zdropped = 1;
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
st0 = st, en0 = en;
|
st0 = st, en0 = en;
|
||||||
st = st / 16 * 16, en = (en + 16) / 16 * 16 - 1;
|
st = st / 16 * 16, en = (en + 16) / 16 * 16 - 1;
|
||||||
// set boundary conditions
|
// set boundary conditions
|
||||||
if (st > 0) {
|
if (st > 0) {
|
||||||
if (st - 1 >= last_st && st - 1 <= last_en)
|
if (st - 1 >= last_st && st - 1 <= last_en)
|
||||||
x1 = ((uint8_t*)x)[st - 1], v1 = v8[st - 1]; // (r-1,s-1) calculated in the last round
|
x1 = ((uint8_t*)x)[st - 1], v1 = v8[st - 1]; // (r-1,s-1) calculated in the last round
|
||||||
else x1 = v1 = 0; // not calculated; set to zeros
|
else x1 = v1 = 0; // not calculated; set to zeros
|
||||||
} else x1 = 0, v1 = r? q : 0;
|
} else x1 = 0, v1 = r? q : 0;
|
||||||
if (en >= r) ((uint8_t*)y)[r] = 0, u8[r] = r? q : 0;
|
if (en >= r) ((uint8_t*)y)[r] = 0, u8[r] = r? q : 0;
|
||||||
// loop fission: set scores first
|
// loop fission: set scores first
|
||||||
if (!(flag & KSW_EZ_GENERIC_SC)) {
|
if (!(flag & KSW_EZ_GENERIC_SC)) {
|
||||||
for (t = st0; t <= en0; t += 16) {
|
for (t = st0; t <= en0; t += 16) {
|
||||||
__m128i sq, st, tmp, mask;
|
__m128i sq, st, tmp, mask;
|
||||||
sq = _mm_loadu_si128((__m128i*)&sf[t]);
|
sq = _mm_loadu_si128((__m128i*)&sf[t]);
|
||||||
st = _mm_loadu_si128((__m128i*)&qrr[t]);
|
st = _mm_loadu_si128((__m128i*)&qrr[t]);
|
||||||
mask = _mm_or_si128(_mm_cmpeq_epi8(sq, m1_), _mm_cmpeq_epi8(st, m1_));
|
mask = _mm_or_si128(_mm_cmpeq_epi8(sq, m1_), _mm_cmpeq_epi8(st, m1_));
|
||||||
tmp = _mm_cmpeq_epi8(sq, st);
|
tmp = _mm_cmpeq_epi8(sq, st);
|
||||||
#ifdef __SSE4_1__
|
#ifdef __SSE4_1__
|
||||||
tmp = _mm_blendv_epi8(sc_mis_, sc_mch_, tmp);
|
tmp = _mm_blendv_epi8(sc_mis_, sc_mch_, tmp);
|
||||||
tmp = _mm_blendv_epi8(tmp, sc_N_, mask);
|
tmp = _mm_blendv_epi8(tmp, sc_N_, mask);
|
||||||
#else
|
#else
|
||||||
tmp = _mm_or_si128(_mm_andnot_si128(tmp, sc_mis_), _mm_and_si128(tmp, sc_mch_));
|
tmp = _mm_or_si128(_mm_andnot_si128(tmp, sc_mis_), _mm_and_si128(tmp, sc_mch_));
|
||||||
tmp = _mm_or_si128(_mm_andnot_si128(mask, tmp), _mm_and_si128(mask, sc_N_));
|
tmp = _mm_or_si128(_mm_andnot_si128(mask, tmp), _mm_and_si128(mask, sc_N_));
|
||||||
#endif
|
#endif
|
||||||
_mm_storeu_si128((__m128i*)((uint8_t*)s + t), tmp);
|
_mm_storeu_si128((__m128i*)((uint8_t*)s + t), tmp);
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
for (t = st0; t <= en0; ++t)
|
for (t = st0; t <= en0; ++t)
|
||||||
((uint8_t*)s)[t] = mat[sf[t] * m + qrr[t]];
|
((uint8_t*)s)[t] = mat[sf[t] * m + qrr[t]];
|
||||||
}
|
}
|
||||||
// core loop
|
// core loop
|
||||||
x1_ = _mm_cvtsi32_si128(x1);
|
x1_ = _mm_cvtsi32_si128(x1);
|
||||||
v1_ = _mm_cvtsi32_si128(v1);
|
v1_ = _mm_cvtsi32_si128(v1);
|
||||||
st_ = st / 16, en_ = en / 16;
|
st_ = st / 16, en_ = en / 16;
|
||||||
assert(en_ - st_ + 1 <= n_col_);
|
assert(en_ - st_ + 1 <= n_col_);
|
||||||
if (!with_cigar) { // score only
|
if (!with_cigar) { // score only
|
||||||
for (t = st_; t <= en_; ++t) {
|
for (t = st_; t <= en_; ++t) {
|
||||||
__m128i z, a, b, xt1, vt1, ut, tmp;
|
__m128i z, a, b, xt1, vt1, ut, tmp;
|
||||||
__dp_code_block1;
|
__dp_code_block1;
|
||||||
#ifdef __SSE4_1__
|
#ifdef __SSE4_1__
|
||||||
z = _mm_max_epi8(z, a); // z = z > a? z : a (signed)
|
z = _mm_max_epi8(z, a); // z = z > a? z : a (signed)
|
||||||
#else // we need to emulate SSE4.1 intrinsics _mm_max_epi8()
|
#else // we need to emulate SSE4.1 intrinsics _mm_max_epi8()
|
||||||
z = _mm_and_si128(z, _mm_cmpgt_epi8(z, zero_)); // z = z > 0? z : 0;
|
z = _mm_and_si128(z, _mm_cmpgt_epi8(z, zero_)); // z = z > 0? z : 0;
|
||||||
z = _mm_max_epu8(z, a); // z = max(z, a); this works because both are non-negative
|
z = _mm_max_epu8(z, a); // z = max(z, a); this works because both are non-negative
|
||||||
#endif
|
#endif
|
||||||
__dp_code_block2;
|
__dp_code_block2;
|
||||||
#ifdef __SSE4_1__
|
#ifdef __SSE4_1__
|
||||||
_mm_store_si128(&x[t], _mm_max_epi8(a, zero_));
|
_mm_store_si128(&x[t], _mm_max_epi8(a, zero_));
|
||||||
_mm_store_si128(&y[t], _mm_max_epi8(b, zero_));
|
_mm_store_si128(&y[t], _mm_max_epi8(b, zero_));
|
||||||
#else
|
#else
|
||||||
tmp = _mm_cmpgt_epi8(a, zero_);
|
tmp = _mm_cmpgt_epi8(a, zero_);
|
||||||
_mm_store_si128(&x[t], _mm_and_si128(a, tmp));
|
_mm_store_si128(&x[t], _mm_and_si128(a, tmp));
|
||||||
tmp = _mm_cmpgt_epi8(b, zero_);
|
tmp = _mm_cmpgt_epi8(b, zero_);
|
||||||
_mm_store_si128(&y[t], _mm_and_si128(b, tmp));
|
_mm_store_si128(&y[t], _mm_and_si128(b, tmp));
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
} else if (!(flag&KSW_EZ_RIGHT)) { // gap left-alignment
|
} else if (!(flag&KSW_EZ_RIGHT)) { // gap left-alignment
|
||||||
__m128i *pr = p + (size_t)r * n_col_ - st_;
|
__m128i *pr = p + (size_t)r * n_col_ - st_;
|
||||||
off[r] = st, off_end[r] = en;
|
off[r] = st, off_end[r] = en;
|
||||||
for (t = st_; t <= en_; ++t) {
|
for (t = st_; t <= en_; ++t) {
|
||||||
__m128i d, z, a, b, xt1, vt1, ut, tmp;
|
__m128i d, z, a, b, xt1, vt1, ut, tmp;
|
||||||
__dp_code_block1;
|
__dp_code_block1;
|
||||||
d = _mm_and_si128(_mm_cmpgt_epi8(a, z), flag1_); // d = a > z? 1 : 0
|
d = _mm_and_si128(_mm_cmpgt_epi8(a, z), flag1_); // d = a > z? 1 : 0
|
||||||
#ifdef __SSE4_1__
|
#ifdef __SSE4_1__
|
||||||
z = _mm_max_epi8(z, a); // z = z > a? z : a (signed)
|
z = _mm_max_epi8(z, a); // z = z > a? z : a (signed)
|
||||||
tmp = _mm_cmpgt_epi8(b, z);
|
tmp = _mm_cmpgt_epi8(b, z);
|
||||||
d = _mm_blendv_epi8(d, flag2_, tmp); // d = b > z? 2 : d
|
d = _mm_blendv_epi8(d, flag2_, tmp); // d = b > z? 2 : d
|
||||||
#else // we need to emulate SSE4.1 intrinsics _mm_max_epi8() and _mm_blendv_epi8()
|
#else // we need to emulate SSE4.1 intrinsics _mm_max_epi8() and _mm_blendv_epi8()
|
||||||
z = _mm_and_si128(z, _mm_cmpgt_epi8(z, zero_)); // z = z > 0? z : 0;
|
z = _mm_and_si128(z, _mm_cmpgt_epi8(z, zero_)); // z = z > 0? z : 0;
|
||||||
z = _mm_max_epu8(z, a); // z = max(z, a); this works because both are non-negative
|
z = _mm_max_epu8(z, a); // z = max(z, a); this works because both are non-negative
|
||||||
tmp = _mm_cmpgt_epi8(b, z);
|
tmp = _mm_cmpgt_epi8(b, z);
|
||||||
d = _mm_or_si128(_mm_andnot_si128(tmp, d), _mm_and_si128(tmp, flag2_)); // d = b > z? 2 : d; emulating blendv
|
d = _mm_or_si128(_mm_andnot_si128(tmp, d), _mm_and_si128(tmp, flag2_)); // d = b > z? 2 : d; emulating blendv
|
||||||
#endif
|
#endif
|
||||||
__dp_code_block2;
|
__dp_code_block2;
|
||||||
tmp = _mm_cmpgt_epi8(a, zero_);
|
tmp = _mm_cmpgt_epi8(a, zero_);
|
||||||
_mm_store_si128(&x[t], _mm_and_si128(tmp, a));
|
_mm_store_si128(&x[t], _mm_and_si128(tmp, a));
|
||||||
d = _mm_or_si128(d, _mm_and_si128(tmp, flag8_)); // d = a > 0? 0x08 : 0
|
d = _mm_or_si128(d, _mm_and_si128(tmp, flag8_)); // d = a > 0? 0x08 : 0
|
||||||
tmp = _mm_cmpgt_epi8(b, zero_);
|
tmp = _mm_cmpgt_epi8(b, zero_);
|
||||||
_mm_store_si128(&y[t], _mm_and_si128(tmp, b));
|
_mm_store_si128(&y[t], _mm_and_si128(tmp, b));
|
||||||
d = _mm_or_si128(d, _mm_and_si128(tmp, flag16_)); // d = b > 0? 0x10 : 0
|
d = _mm_or_si128(d, _mm_and_si128(tmp, flag16_)); // d = b > 0? 0x10 : 0
|
||||||
_mm_store_si128(&pr[t], d);
|
_mm_store_si128(&pr[t], d);
|
||||||
}
|
}
|
||||||
} else { // gap right-alignment
|
} else { // gap right-alignment
|
||||||
__m128i *pr = p + (size_t)r * n_col_ - st_;
|
__m128i *pr = p + (size_t)r * n_col_ - st_;
|
||||||
off[r] = st, off_end[r] = en;
|
off[r] = st, off_end[r] = en;
|
||||||
for (t = st_; t <= en_; ++t) {
|
for (t = st_; t <= en_; ++t) {
|
||||||
__m128i d, z, a, b, xt1, vt1, ut, tmp;
|
__m128i d, z, a, b, xt1, vt1, ut, tmp;
|
||||||
__dp_code_block1;
|
__dp_code_block1;
|
||||||
d = _mm_andnot_si128(_mm_cmpgt_epi8(z, a), flag1_); // d = z > a? 0 : 1
|
d = _mm_andnot_si128(_mm_cmpgt_epi8(z, a), flag1_); // d = z > a? 0 : 1
|
||||||
#ifdef __SSE4_1__
|
#ifdef __SSE4_1__
|
||||||
z = _mm_max_epi8(z, a); // z = z > a? z : a (signed)
|
z = _mm_max_epi8(z, a); // z = z > a? z : a (signed)
|
||||||
tmp = _mm_cmpgt_epi8(z, b);
|
tmp = _mm_cmpgt_epi8(z, b);
|
||||||
d = _mm_blendv_epi8(flag2_, d, tmp); // d = z > b? d : 2
|
d = _mm_blendv_epi8(flag2_, d, tmp); // d = z > b? d : 2
|
||||||
#else // we need to emulate SSE4.1 intrinsics _mm_max_epi8() and _mm_blendv_epi8()
|
#else // we need to emulate SSE4.1 intrinsics _mm_max_epi8() and _mm_blendv_epi8()
|
||||||
z = _mm_and_si128(z, _mm_cmpgt_epi8(z, zero_)); // z = z > 0? z : 0;
|
z = _mm_and_si128(z, _mm_cmpgt_epi8(z, zero_)); // z = z > 0? z : 0;
|
||||||
z = _mm_max_epu8(z, a); // z = max(z, a); this works because both are non-negative
|
z = _mm_max_epu8(z, a); // z = max(z, a); this works because both are non-negative
|
||||||
tmp = _mm_cmpgt_epi8(z, b);
|
tmp = _mm_cmpgt_epi8(z, b);
|
||||||
d = _mm_or_si128(_mm_andnot_si128(tmp, flag2_), _mm_and_si128(tmp, d)); // d = z > b? d : 2; emulating blendv
|
d = _mm_or_si128(_mm_andnot_si128(tmp, flag2_), _mm_and_si128(tmp, d)); // d = z > b? d : 2; emulating blendv
|
||||||
#endif
|
#endif
|
||||||
__dp_code_block2;
|
__dp_code_block2;
|
||||||
tmp = _mm_cmpgt_epi8(zero_, a);
|
tmp = _mm_cmpgt_epi8(zero_, a);
|
||||||
_mm_store_si128(&x[t], _mm_andnot_si128(tmp, a));
|
_mm_store_si128(&x[t], _mm_andnot_si128(tmp, a));
|
||||||
d = _mm_or_si128(d, _mm_andnot_si128(tmp, flag8_)); // d = 0 > a? 0 : 0x08
|
d = _mm_or_si128(d, _mm_andnot_si128(tmp, flag8_)); // d = 0 > a? 0 : 0x08
|
||||||
tmp = _mm_cmpgt_epi8(zero_, b);
|
tmp = _mm_cmpgt_epi8(zero_, b);
|
||||||
_mm_store_si128(&y[t], _mm_andnot_si128(tmp, b));
|
_mm_store_si128(&y[t], _mm_andnot_si128(tmp, b));
|
||||||
d = _mm_or_si128(d, _mm_andnot_si128(tmp, flag16_)); // d = 0 > b? 0 : 0x10
|
d = _mm_or_si128(d, _mm_andnot_si128(tmp, flag16_)); // d = 0 > b? 0 : 0x10
|
||||||
_mm_store_si128(&pr[t], d);
|
_mm_store_si128(&pr[t], d);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
if (!approx_max) { // find the exact max with a 32-bit score array
|
if (!approx_max) { // find the exact max with a 32-bit score array
|
||||||
int32_t max_H, max_t;
|
int32_t max_H, max_t;
|
||||||
// compute H[], max_H and max_t
|
// compute H[], max_H and max_t
|
||||||
if (r > 0) {
|
if (r > 0) {
|
||||||
int32_t HH[4], tt[4], en1 = st0 + (en0 - st0) / 4 * 4, i;
|
int32_t HH[4], tt[4], en1 = st0 + (en0 - st0) / 4 * 4, i;
|
||||||
__m128i max_H_, max_t_, qe_;
|
__m128i max_H_, max_t_, qe_;
|
||||||
max_H = H[en0] = en0 > 0? H[en0-1] + u8[en0] - qe : H[en0] + v8[en0] - qe; // special casing the last element
|
max_H = H[en0] = en0 > 0? H[en0-1] + u8[en0] - qe : H[en0] + v8[en0] - qe; // special casing the last element
|
||||||
max_t = en0;
|
max_t = en0;
|
||||||
max_H_ = _mm_set1_epi32(max_H);
|
max_H_ = _mm_set1_epi32(max_H);
|
||||||
max_t_ = _mm_set1_epi32(max_t);
|
max_t_ = _mm_set1_epi32(max_t);
|
||||||
qe_ = _mm_set1_epi32(q + e);
|
qe_ = _mm_set1_epi32(q + e);
|
||||||
for (t = st0; t < en1; t += 4) { // this implements: H[t]+=v8[t]-qe; if(H[t]>max_H) max_H=H[t],max_t=t;
|
for (t = st0; t < en1; t += 4) { // this implements: H[t]+=v8[t]-qe; if(H[t]>max_H) max_H=H[t],max_t=t;
|
||||||
__m128i H1, tmp, t_;
|
__m128i H1, tmp, t_;
|
||||||
H1 = _mm_loadu_si128((__m128i*)&H[t]);
|
H1 = _mm_loadu_si128((__m128i*)&H[t]);
|
||||||
t_ = _mm_setr_epi32(v8[t], v8[t+1], v8[t+2], v8[t+3]);
|
t_ = _mm_setr_epi32(v8[t], v8[t+1], v8[t+2], v8[t+3]);
|
||||||
H1 = _mm_add_epi32(H1, t_);
|
H1 = _mm_add_epi32(H1, t_);
|
||||||
H1 = _mm_sub_epi32(H1, qe_);
|
H1 = _mm_sub_epi32(H1, qe_);
|
||||||
_mm_storeu_si128((__m128i*)&H[t], H1);
|
_mm_storeu_si128((__m128i*)&H[t], H1);
|
||||||
t_ = _mm_set1_epi32(t);
|
t_ = _mm_set1_epi32(t);
|
||||||
tmp = _mm_cmpgt_epi32(H1, max_H_);
|
tmp = _mm_cmpgt_epi32(H1, max_H_);
|
||||||
#ifdef __SSE4_1__
|
#ifdef __SSE4_1__
|
||||||
max_H_ = _mm_blendv_epi8(max_H_, H1, tmp);
|
max_H_ = _mm_blendv_epi8(max_H_, H1, tmp);
|
||||||
max_t_ = _mm_blendv_epi8(max_t_, t_, tmp);
|
max_t_ = _mm_blendv_epi8(max_t_, t_, tmp);
|
||||||
#else
|
#else
|
||||||
max_H_ = _mm_or_si128(_mm_and_si128(tmp, H1), _mm_andnot_si128(tmp, max_H_));
|
max_H_ = _mm_or_si128(_mm_and_si128(tmp, H1), _mm_andnot_si128(tmp, max_H_));
|
||||||
max_t_ = _mm_or_si128(_mm_and_si128(tmp, t_), _mm_andnot_si128(tmp, max_t_));
|
max_t_ = _mm_or_si128(_mm_and_si128(tmp, t_), _mm_andnot_si128(tmp, max_t_));
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
_mm_storeu_si128((__m128i*)HH, max_H_);
|
_mm_storeu_si128((__m128i*)HH, max_H_);
|
||||||
_mm_storeu_si128((__m128i*)tt, max_t_);
|
_mm_storeu_si128((__m128i*)tt, max_t_);
|
||||||
for (i = 0; i < 4; ++i)
|
for (i = 0; i < 4; ++i)
|
||||||
if (max_H < HH[i]) max_H = HH[i], max_t = tt[i] + i;
|
if (max_H < HH[i]) max_H = HH[i], max_t = tt[i] + i;
|
||||||
for (; t < en0; ++t) { // for the rest of values that haven't been computed with SSE
|
for (; t < en0; ++t) { // for the rest of values that haven't been computed with SSE
|
||||||
H[t] += (int32_t)v8[t] - qe;
|
H[t] += (int32_t)v8[t] - qe;
|
||||||
if (H[t] > max_H)
|
if (H[t] > max_H)
|
||||||
max_H = H[t], max_t = t;
|
max_H = H[t], max_t = t;
|
||||||
}
|
}
|
||||||
} else H[0] = v8[0] - qe - qe, max_H = H[0], max_t = 0; // special casing r==0
|
} else H[0] = v8[0] - qe - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||||
// update ez
|
// update ez
|
||||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||||
ez->mte = H[en0], ez->mte_q = r - en;
|
ez->mte = H[en0], ez->mte_q = r - en;
|
||||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e)) break;
|
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e)) break;
|
||||||
if (r == qlen + tlen - 2 && en0 == tlen - 1)
|
if (r == qlen + tlen - 2 && en0 == tlen - 1)
|
||||||
ez->score = H[tlen - 1];
|
ez->score = H[tlen - 1];
|
||||||
} else { // find approximate max; Z-drop might be inaccurate, too.
|
} else { // find approximate max; Z-drop might be inaccurate, too.
|
||||||
if (r > 0) {
|
if (r > 0) {
|
||||||
if (last_H0_t >= st0 && last_H0_t <= en0 && last_H0_t + 1 >= st0 && last_H0_t + 1 <= en0) {
|
if (last_H0_t >= st0 && last_H0_t <= en0 && last_H0_t + 1 >= st0 && last_H0_t + 1 <= en0) {
|
||||||
int32_t d0 = v8[last_H0_t] - qe;
|
int32_t d0 = v8[last_H0_t] - qe;
|
||||||
int32_t d1 = u8[last_H0_t + 1] - qe;
|
int32_t d1 = u8[last_H0_t + 1] - qe;
|
||||||
if (d0 > d1) H0 += d0;
|
if (d0 > d1) H0 += d0;
|
||||||
else H0 += d1, ++last_H0_t;
|
else H0 += d1, ++last_H0_t;
|
||||||
} else if (last_H0_t >= st0 && last_H0_t <= en0) {
|
} else if (last_H0_t >= st0 && last_H0_t <= en0) {
|
||||||
H0 += v8[last_H0_t] - qe;
|
H0 += v8[last_H0_t] - qe;
|
||||||
} else {
|
} else {
|
||||||
++last_H0_t, H0 += u8[last_H0_t] - qe;
|
++last_H0_t, H0 += u8[last_H0_t] - qe;
|
||||||
}
|
}
|
||||||
if ((flag & KSW_EZ_APPROX_DROP) && ksw_apply_zdrop(ez, 1, H0, r, last_H0_t, zdrop, e)) break;
|
if ((flag & KSW_EZ_APPROX_DROP) && ksw_apply_zdrop(ez, 1, H0, r, last_H0_t, zdrop, e)) break;
|
||||||
} else H0 = v8[0] - qe - qe, last_H0_t = 0;
|
} else H0 = v8[0] - qe - qe, last_H0_t = 0;
|
||||||
if (r == qlen + tlen - 2 && en0 == tlen - 1)
|
if (r == qlen + tlen - 2 && en0 == tlen - 1)
|
||||||
ez->score = H0;
|
ez->score = H0;
|
||||||
}
|
}
|
||||||
last_st = st, last_en = en;
|
last_st = st, last_en = en;
|
||||||
//for (t = st0; t <= en0; ++t) printf("(%d,%d)\t(%d,%d,%d,%d)\t%d\n", r, t, ((int8_t*)u)[t], ((int8_t*)v)[t], ((int8_t*)x)[t], ((int8_t*)y)[t], H[t]); // for debugging
|
//for (t = st0; t <= en0; ++t) printf("(%d,%d)\t(%d,%d,%d,%d)\t%d\n", r, t, ((int8_t*)u)[t], ((int8_t*)v)[t], ((int8_t*)x)[t], ((int8_t*)y)[t], H[t]); // for debugging
|
||||||
}
|
}
|
||||||
kfree(km, mem);
|
kfree(km, mem);
|
||||||
if (!approx_max) kfree(km, H);
|
if (!approx_max) kfree(km, H);
|
||||||
if (with_cigar) { // backtrack
|
if (with_cigar) { // backtrack
|
||||||
int rev_cigar = !!(flag & KSW_EZ_REV_CIGAR);
|
int rev_cigar = !!(flag & KSW_EZ_REV_CIGAR);
|
||||||
if (!ez->zdropped && !(flag&KSW_EZ_EXTZ_ONLY)) {
|
if (!ez->zdropped && !(flag&KSW_EZ_EXTZ_ONLY)) {
|
||||||
ksw_backtrack(km, 1, rev_cigar, 0, (uint8_t*)p, off, off_end, n_col_*16, tlen-1, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
ksw_backtrack(km, 1, rev_cigar, 0, (uint8_t*)p, off, off_end, n_col_*16, tlen-1, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||||
} else if (!ez->zdropped && (flag&KSW_EZ_EXTZ_ONLY) && ez->mqe + end_bonus > (int)ez->max) {
|
} else if (!ez->zdropped && (flag&KSW_EZ_EXTZ_ONLY) && ez->mqe + end_bonus > (int)ez->max) {
|
||||||
ez->reach_end = 1;
|
ez->reach_end = 1;
|
||||||
ksw_backtrack(km, 1, rev_cigar, 0, (uint8_t*)p, off, off_end, n_col_*16, ez->mqe_t, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
ksw_backtrack(km, 1, rev_cigar, 0, (uint8_t*)p, off, off_end, n_col_*16, ez->mqe_t, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||||
} else if (ez->max_t >= 0 && ez->max_q >= 0) {
|
} else if (ez->max_t >= 0 && ez->max_q >= 0) {
|
||||||
ksw_backtrack(km, 1, rev_cigar, 0, (uint8_t*)p, off, off_end, n_col_*16, ez->max_t, ez->max_q, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
ksw_backtrack(km, 1, rev_cigar, 0, (uint8_t*)p, off, off_end, n_col_*16, ez->max_t, ez->max_q, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||||
}
|
}
|
||||||
kfree(km, mem2); kfree(km, off);
|
kfree(km, mem2); kfree(km, off);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
#endif // __SSE2__
|
#endif // __SSE2__
|
||||||
|
|||||||
+160
-160
@@ -1,160 +1,160 @@
|
|||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <pthread.h>
|
#include <pthread.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include <limits.h>
|
#include <limits.h>
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include "kthread.h"
|
#include "kthread.h"
|
||||||
|
|
||||||
#if (defined(WIN32) || defined(_WIN32)) && defined(_MSC_VER)
|
#if (defined(WIN32) || defined(_WIN32)) && defined(_MSC_VER)
|
||||||
#define __sync_fetch_and_add(ptr, addend) _InterlockedExchangeAdd((void*)ptr, addend)
|
#define __sync_fetch_and_add(ptr, addend) _InterlockedExchangeAdd((void*)ptr, addend)
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
/************
|
/************
|
||||||
* kt_for() *
|
* kt_for() *
|
||||||
************/
|
************/
|
||||||
|
|
||||||
struct kt_for_t;
|
struct kt_for_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
struct kt_for_t *t;
|
struct kt_for_t *t;
|
||||||
long i;
|
long i;
|
||||||
} ktf_worker_t;
|
} ktf_worker_t;
|
||||||
|
|
||||||
typedef struct kt_for_t {
|
typedef struct kt_for_t {
|
||||||
int n_threads;
|
int n_threads;
|
||||||
long n;
|
long n;
|
||||||
ktf_worker_t *w;
|
ktf_worker_t *w;
|
||||||
void (*func)(void*,long,int);
|
void (*func)(void*,long,int);
|
||||||
void *data;
|
void *data;
|
||||||
} kt_for_t;
|
} kt_for_t;
|
||||||
|
|
||||||
static inline long steal_work(kt_for_t *t)
|
static inline long steal_work(kt_for_t *t)
|
||||||
{
|
{
|
||||||
int i, min_i = -1;
|
int i, min_i = -1;
|
||||||
long k, min = LONG_MAX;
|
long k, min = LONG_MAX;
|
||||||
for (i = 0; i < t->n_threads; ++i)
|
for (i = 0; i < t->n_threads; ++i)
|
||||||
if (min > t->w[i].i) min = t->w[i].i, min_i = i;
|
if (min > t->w[i].i) min = t->w[i].i, min_i = i;
|
||||||
k = __sync_fetch_and_add(&t->w[min_i].i, t->n_threads);
|
k = __sync_fetch_and_add(&t->w[min_i].i, t->n_threads);
|
||||||
return k >= t->n? -1 : k;
|
return k >= t->n? -1 : k;
|
||||||
}
|
}
|
||||||
|
|
||||||
static void *ktf_worker(void *data)
|
static void *ktf_worker(void *data)
|
||||||
{
|
{
|
||||||
ktf_worker_t *w = (ktf_worker_t*)data;
|
ktf_worker_t *w = (ktf_worker_t*)data;
|
||||||
long i;
|
long i;
|
||||||
for (;;) {
|
for (;;) {
|
||||||
i = __sync_fetch_and_add(&w->i, w->t->n_threads);
|
i = __sync_fetch_and_add(&w->i, w->t->n_threads);
|
||||||
if (i >= w->t->n) break;
|
if (i >= w->t->n) break;
|
||||||
w->t->func(w->t->data, i, w - w->t->w);
|
w->t->func(w->t->data, i, w - w->t->w);
|
||||||
}
|
}
|
||||||
while ((i = steal_work(w->t)) >= 0)
|
while ((i = steal_work(w->t)) >= 0)
|
||||||
w->t->func(w->t->data, i, w - w->t->w);
|
w->t->func(w->t->data, i, w - w->t->w);
|
||||||
pthread_exit(0);
|
pthread_exit(0);
|
||||||
}
|
}
|
||||||
|
|
||||||
void kt_for(int n_threads, void (*func)(void*,long,int), void *data, long n)
|
void kt_for(int n_threads, void (*func)(void*,long,int), void *data, long n)
|
||||||
{
|
{
|
||||||
if (n_threads > 1) {
|
if (n_threads > 1) {
|
||||||
int i;
|
int i;
|
||||||
kt_for_t t;
|
kt_for_t t;
|
||||||
pthread_t *tid;
|
pthread_t *tid;
|
||||||
t.func = func, t.data = data, t.n_threads = n_threads, t.n = n;
|
t.func = func, t.data = data, t.n_threads = n_threads, t.n = n;
|
||||||
t.w = (ktf_worker_t*)calloc(n_threads, sizeof(ktf_worker_t));
|
t.w = (ktf_worker_t*)calloc(n_threads, sizeof(ktf_worker_t));
|
||||||
tid = (pthread_t*)calloc(n_threads, sizeof(pthread_t));
|
tid = (pthread_t*)calloc(n_threads, sizeof(pthread_t));
|
||||||
for (i = 0; i < n_threads; ++i)
|
for (i = 0; i < n_threads; ++i)
|
||||||
t.w[i].t = &t, t.w[i].i = i;
|
t.w[i].t = &t, t.w[i].i = i;
|
||||||
for (i = 0; i < n_threads; ++i) pthread_create(&tid[i], 0, ktf_worker, &t.w[i]);
|
for (i = 0; i < n_threads; ++i) pthread_create(&tid[i], 0, ktf_worker, &t.w[i]);
|
||||||
for (i = 0; i < n_threads; ++i) pthread_join(tid[i], 0);
|
for (i = 0; i < n_threads; ++i) pthread_join(tid[i], 0);
|
||||||
free(tid); free(t.w);
|
free(tid); free(t.w);
|
||||||
} else {
|
} else {
|
||||||
long j;
|
long j;
|
||||||
for (j = 0; j < n; ++j) func(data, j, 0);
|
for (j = 0; j < n; ++j) func(data, j, 0);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
/*****************
|
/*****************
|
||||||
* kt_pipeline() *
|
* kt_pipeline() *
|
||||||
*****************/
|
*****************/
|
||||||
|
|
||||||
struct ktp_t;
|
struct ktp_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
struct ktp_t *pl;
|
struct ktp_t *pl;
|
||||||
int64_t index;
|
int64_t index;
|
||||||
int step;
|
int step;
|
||||||
void *data;
|
void *data;
|
||||||
} ktp_worker_t;
|
} ktp_worker_t;
|
||||||
|
|
||||||
typedef struct ktp_t {
|
typedef struct ktp_t {
|
||||||
void *shared;
|
void *shared;
|
||||||
void *(*func)(void*, int, void*);
|
void *(*func)(void*, int, void*);
|
||||||
int64_t index;
|
int64_t index;
|
||||||
int n_workers, n_steps;
|
int n_workers, n_steps;
|
||||||
ktp_worker_t *workers;
|
ktp_worker_t *workers;
|
||||||
pthread_mutex_t mutex;
|
pthread_mutex_t mutex;
|
||||||
pthread_cond_t cv;
|
pthread_cond_t cv;
|
||||||
} ktp_t;
|
} ktp_t;
|
||||||
|
|
||||||
static void *ktp_worker(void *data)
|
static void *ktp_worker(void *data)
|
||||||
{
|
{
|
||||||
ktp_worker_t *w = (ktp_worker_t*)data;
|
ktp_worker_t *w = (ktp_worker_t*)data;
|
||||||
ktp_t *p = w->pl;
|
ktp_t *p = w->pl;
|
||||||
while (w->step < p->n_steps) {
|
while (w->step < p->n_steps) {
|
||||||
// test whether we can kick off the job with this worker
|
// test whether we can kick off the job with this worker
|
||||||
pthread_mutex_lock(&p->mutex);
|
pthread_mutex_lock(&p->mutex);
|
||||||
for (;;) {
|
for (;;) {
|
||||||
int i;
|
int i;
|
||||||
// test whether another worker is doing the same step
|
// test whether another worker is doing the same step
|
||||||
for (i = 0; i < p->n_workers; ++i) {
|
for (i = 0; i < p->n_workers; ++i) {
|
||||||
if (w == &p->workers[i]) continue; // ignore itself
|
if (w == &p->workers[i]) continue; // ignore itself
|
||||||
if (p->workers[i].step <= w->step && p->workers[i].index < w->index)
|
if (p->workers[i].step <= w->step && p->workers[i].index < w->index)
|
||||||
break;
|
break;
|
||||||
}
|
}
|
||||||
if (i == p->n_workers) break; // no workers with smaller indices are doing w->step or the previous steps
|
if (i == p->n_workers) break; // no workers with smaller indices are doing w->step or the previous steps
|
||||||
pthread_cond_wait(&p->cv, &p->mutex);
|
pthread_cond_wait(&p->cv, &p->mutex);
|
||||||
}
|
}
|
||||||
pthread_mutex_unlock(&p->mutex);
|
pthread_mutex_unlock(&p->mutex);
|
||||||
|
|
||||||
// working on w->step
|
// working on w->step
|
||||||
w->data = p->func(p->shared, w->step, w->step? w->data : 0); // for the first step, input is NULL
|
w->data = p->func(p->shared, w->step, w->step? w->data : 0); // for the first step, input is NULL
|
||||||
|
|
||||||
// update step and let other workers know
|
// update step and let other workers know
|
||||||
pthread_mutex_lock(&p->mutex);
|
pthread_mutex_lock(&p->mutex);
|
||||||
w->step = w->step == p->n_steps - 1 || w->data? (w->step + 1) % p->n_steps : p->n_steps;
|
w->step = w->step == p->n_steps - 1 || w->data? (w->step + 1) % p->n_steps : p->n_steps;
|
||||||
if (w->step == 0) w->index = p->index++;
|
if (w->step == 0) w->index = p->index++;
|
||||||
pthread_cond_broadcast(&p->cv);
|
pthread_cond_broadcast(&p->cv);
|
||||||
pthread_mutex_unlock(&p->mutex);
|
pthread_mutex_unlock(&p->mutex);
|
||||||
}
|
}
|
||||||
pthread_exit(0);
|
pthread_exit(0);
|
||||||
}
|
}
|
||||||
|
|
||||||
void kt_pipeline(int n_threads, void *(*func)(void*, int, void*), void *shared_data, int n_steps)
|
void kt_pipeline(int n_threads, void *(*func)(void*, int, void*), void *shared_data, int n_steps)
|
||||||
{
|
{
|
||||||
ktp_t aux;
|
ktp_t aux;
|
||||||
pthread_t *tid;
|
pthread_t *tid;
|
||||||
int i;
|
int i;
|
||||||
|
|
||||||
if (n_threads < 1) n_threads = 1;
|
if (n_threads < 1) n_threads = 1;
|
||||||
aux.n_workers = n_threads;
|
aux.n_workers = n_threads;
|
||||||
aux.n_steps = n_steps;
|
aux.n_steps = n_steps;
|
||||||
aux.func = func;
|
aux.func = func;
|
||||||
aux.shared = shared_data;
|
aux.shared = shared_data;
|
||||||
aux.index = 0;
|
aux.index = 0;
|
||||||
pthread_mutex_init(&aux.mutex, 0);
|
pthread_mutex_init(&aux.mutex, 0);
|
||||||
pthread_cond_init(&aux.cv, 0);
|
pthread_cond_init(&aux.cv, 0);
|
||||||
|
|
||||||
aux.workers = (ktp_worker_t*)calloc(n_threads, sizeof(ktp_worker_t));
|
aux.workers = (ktp_worker_t*)calloc(n_threads, sizeof(ktp_worker_t));
|
||||||
for (i = 0; i < n_threads; ++i) {
|
for (i = 0; i < n_threads; ++i) {
|
||||||
ktp_worker_t *w = &aux.workers[i];
|
ktp_worker_t *w = &aux.workers[i];
|
||||||
w->step = 0; w->pl = &aux; w->data = 0;
|
w->step = 0; w->pl = &aux; w->data = 0;
|
||||||
w->index = aux.index++;
|
w->index = aux.index++;
|
||||||
}
|
}
|
||||||
|
|
||||||
tid = (pthread_t*)calloc(n_threads, sizeof(pthread_t));
|
tid = (pthread_t*)calloc(n_threads, sizeof(pthread_t));
|
||||||
for (i = 0; i < n_threads; ++i) pthread_create(&tid[i], 0, ktp_worker, &aux.workers[i]);
|
for (i = 0; i < n_threads; ++i) pthread_create(&tid[i], 0, ktp_worker, &aux.workers[i]);
|
||||||
for (i = 0; i < n_threads; ++i) pthread_join(tid[i], 0);
|
for (i = 0; i < n_threads; ++i) pthread_join(tid[i], 0);
|
||||||
free(tid); free(aux.workers);
|
free(tid); free(aux.workers);
|
||||||
|
|
||||||
pthread_mutex_destroy(&aux.mutex);
|
pthread_mutex_destroy(&aux.mutex);
|
||||||
pthread_cond_destroy(&aux.cv);
|
pthread_cond_destroy(&aux.cv);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,15 +1,15 @@
|
|||||||
#ifndef KTHREAD_H
|
#ifndef KTHREAD_H
|
||||||
#define KTHREAD_H
|
#define KTHREAD_H
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
extern "C" {
|
extern "C" {
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
void kt_for(int n_threads, void (*func)(void*,long,int), void *data, long n);
|
void kt_for(int n_threads, void (*func)(void*,long,int), void *data, long n);
|
||||||
void kt_pipeline(int n_threads, void *(*func)(void*, int, void*), void *shared_data, int n_steps);
|
void kt_pipeline(int n_threads, void *(*func)(void*, int, void*), void *shared_data, int n_steps);
|
||||||
|
|
||||||
#ifdef __cplusplus
|
#ifdef __cplusplus
|
||||||
}
|
}
|
||||||
#endif
|
#endif
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,110 +1,110 @@
|
|||||||
/* The MIT License
|
/* The MIT License
|
||||||
|
|
||||||
Copyright (c) 2008, by Attractive Chaos <attractor@live.co.uk>
|
Copyright (c) 2008, by Attractive Chaos <attractor@live.co.uk>
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining
|
Permission is hereby granted, free of charge, to any person obtaining
|
||||||
a copy of this software and associated documentation files (the
|
a copy of this software and associated documentation files (the
|
||||||
"Software"), to deal in the Software without restriction, including
|
"Software"), to deal in the Software without restriction, including
|
||||||
without limitation the rights to use, copy, modify, merge, publish,
|
without limitation the rights to use, copy, modify, merge, publish,
|
||||||
distribute, sublicense, and/or sell copies of the Software, and to
|
distribute, sublicense, and/or sell copies of the Software, and to
|
||||||
permit persons to whom the Software is furnished to do so, subject to
|
permit persons to whom the Software is furnished to do so, subject to
|
||||||
the following conditions:
|
the following conditions:
|
||||||
|
|
||||||
The above copyright notice and this permission notice shall be
|
The above copyright notice and this permission notice shall be
|
||||||
included in all copies or substantial portions of the Software.
|
included in all copies or substantial portions of the Software.
|
||||||
|
|
||||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
SOFTWARE.
|
SOFTWARE.
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/*
|
/*
|
||||||
An example:
|
An example:
|
||||||
|
|
||||||
#include "kvec.h"
|
#include "kvec.h"
|
||||||
int main() {
|
int main() {
|
||||||
kvec_t(int) array;
|
kvec_t(int) array;
|
||||||
kv_init(array);
|
kv_init(array);
|
||||||
kv_push(int, array, 10); // append
|
kv_push(int, array, 10); // append
|
||||||
kv_a(int, array, 20) = 5; // dynamic
|
kv_a(int, array, 20) = 5; // dynamic
|
||||||
kv_A(array, 20) = 4; // static
|
kv_A(array, 20) = 4; // static
|
||||||
kv_destroy(array);
|
kv_destroy(array);
|
||||||
return 0;
|
return 0;
|
||||||
}
|
}
|
||||||
*/
|
*/
|
||||||
|
|
||||||
/*
|
/*
|
||||||
2008-09-22 (0.1.0):
|
2008-09-22 (0.1.0):
|
||||||
|
|
||||||
* The initial version.
|
* The initial version.
|
||||||
|
|
||||||
*/
|
*/
|
||||||
|
|
||||||
#ifndef AC_KVEC_H
|
#ifndef AC_KVEC_H
|
||||||
#define AC_KVEC_H
|
#define AC_KVEC_H
|
||||||
|
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
|
|
||||||
#define kv_roundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
#define kv_roundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
||||||
|
|
||||||
#define kvec_t(type) struct { size_t n, m; type *a; }
|
#define kvec_t(type) struct { size_t n, m; type *a; }
|
||||||
#define kv_init(v) ((v).n = (v).m = 0, (v).a = 0)
|
#define kv_init(v) ((v).n = (v).m = 0, (v).a = 0)
|
||||||
#define kv_destroy(v) free((v).a)
|
#define kv_destroy(v) free((v).a)
|
||||||
#define kv_A(v, i) ((v).a[(i)])
|
#define kv_A(v, i) ((v).a[(i)])
|
||||||
#define kv_pop(v) ((v).a[--(v).n])
|
#define kv_pop(v) ((v).a[--(v).n])
|
||||||
#define kv_size(v) ((v).n)
|
#define kv_size(v) ((v).n)
|
||||||
#define kv_max(v) ((v).m)
|
#define kv_max(v) ((v).m)
|
||||||
|
|
||||||
#define kv_resize(type, v, s) do { \
|
#define kv_resize(type, v, s) do { \
|
||||||
if ((v).m < (s)) { \
|
if ((v).m < (s)) { \
|
||||||
(v).m = (s); \
|
(v).m = (s); \
|
||||||
kv_roundup32((v).m); \
|
kv_roundup32((v).m); \
|
||||||
(v).a = (type*)realloc((v).a, sizeof(type) * (v).m); \
|
(v).a = (type*)realloc((v).a, sizeof(type) * (v).m); \
|
||||||
} \
|
} \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#define kv_copy(type, v1, v0) do { \
|
#define kv_copy(type, v1, v0) do { \
|
||||||
if ((v1).m < (v0).n) kv_resize(type, v1, (v0).n); \
|
if ((v1).m < (v0).n) kv_resize(type, v1, (v0).n); \
|
||||||
(v1).n = (v0).n; \
|
(v1).n = (v0).n; \
|
||||||
memcpy((v1).a, (v0).a, sizeof(type) * (v0).n); \
|
memcpy((v1).a, (v0).a, sizeof(type) * (v0).n); \
|
||||||
} while (0) \
|
} while (0) \
|
||||||
|
|
||||||
#define kv_push(type, v, x) do { \
|
#define kv_push(type, v, x) do { \
|
||||||
if ((v).n == (v).m) { \
|
if ((v).n == (v).m) { \
|
||||||
(v).m = (v).m? (v).m<<1 : 2; \
|
(v).m = (v).m? (v).m<<1 : 2; \
|
||||||
(v).a = (type*)realloc((v).a, sizeof(type) * (v).m); \
|
(v).a = (type*)realloc((v).a, sizeof(type) * (v).m); \
|
||||||
} \
|
} \
|
||||||
(v).a[(v).n++] = (x); \
|
(v).a[(v).n++] = (x); \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#define kv_pushp(type, v, p) do { \
|
#define kv_pushp(type, v, p) do { \
|
||||||
if ((v).n == (v).m) { \
|
if ((v).n == (v).m) { \
|
||||||
(v).m = (v).m? (v).m<<1 : 2; \
|
(v).m = (v).m? (v).m<<1 : 2; \
|
||||||
(v).a = (type*)realloc((v).a, sizeof(type) * (v).m); \
|
(v).a = (type*)realloc((v).a, sizeof(type) * (v).m); \
|
||||||
} \
|
} \
|
||||||
*(p) = &(v).a[(v).n++]; \
|
*(p) = &(v).a[(v).n++]; \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#define kv_a(type, v, i) ((v).m <= (size_t)(i)? \
|
#define kv_a(type, v, i) ((v).m <= (size_t)(i)? \
|
||||||
((v).m = (v).n = (i) + 1, kv_roundup32((v).m), \
|
((v).m = (v).n = (i) + 1, kv_roundup32((v).m), \
|
||||||
(v).a = (type*)realloc((v).a, sizeof(type) * (v).m), 0) \
|
(v).a = (type*)realloc((v).a, sizeof(type) * (v).m), 0) \
|
||||||
: (v).n <= (size_t)(i)? (v).n = (i) \
|
: (v).n <= (size_t)(i)? (v).n = (i) \
|
||||||
: 0), (v).a[(i)]
|
: 0), (v).a[(i)]
|
||||||
|
|
||||||
#define kv_reverse(type, v, start) do { \
|
#define kv_reverse(type, v, start) do { \
|
||||||
if ((v).m > 0 && (v).n > (start)) { \
|
if ((v).m > 0 && (v).n > (start)) { \
|
||||||
size_t __i, __end = (v).n - (start); \
|
size_t __i, __end = (v).n - (start); \
|
||||||
type *__a = (v).a + (start); \
|
type *__a = (v).a + (start); \
|
||||||
for (__i = 0; __i < __end>>1; ++__i) { \
|
for (__i = 0; __i < __end>>1; ++__i) { \
|
||||||
type __t = __a[__end - 1 - __i]; \
|
type __t = __a[__end - 1 - __i]; \
|
||||||
__a[__end - 1 - __i] = __a[__i]; __a[__i] = __t; \
|
__a[__end - 1 - __i] = __a[__i]; __a[__i] = __t; \
|
||||||
} \
|
} \
|
||||||
} \
|
} \
|
||||||
} while (0)
|
} while (0)
|
||||||
|
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
@@ -1,75 +1,75 @@
|
|||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <stdlib.h>
|
#include <stdlib.h>
|
||||||
#include "CommandLines.h"
|
#include "CommandLines.h"
|
||||||
#include "Process_Read.h"
|
#include "Process_Read.h"
|
||||||
#include "Assembly.h"
|
#include "Assembly.h"
|
||||||
#include "Levenshtein_distance.h"
|
#include "Levenshtein_distance.h"
|
||||||
#include "htab.h"
|
#include "htab.h"
|
||||||
|
|
||||||
int main(int argc, char *argv[])
|
int main(int argc, char *argv[])
|
||||||
{
|
{
|
||||||
int i, ret;
|
int i, ret;
|
||||||
yak_reset_realtime();
|
yak_reset_realtime();
|
||||||
init_opt(&asm_opt);
|
init_opt(&asm_opt);
|
||||||
if (!CommandLine_process(argc, argv, &asm_opt)) return 0;
|
if (!CommandLine_process(argc, argv, &asm_opt)) return 0;
|
||||||
|
|
||||||
// bit_extz_t exz, exz64; init_bit_extz_t(&exz, 2); init_bit_extz_t(&exz64, 2);
|
// bit_extz_t exz, exz64; init_bit_extz_t(&exz, 2); init_bit_extz_t(&exz64, 2);
|
||||||
|
|
||||||
// char *pstr = "GACCCAG", *tsrt = "GTTGTTAATTCCAT"; int32_t thre = 14; clear_align(exz); clear_align(exz64);
|
// char *pstr = "GACCCAG", *tsrt = "GTTGTTAATTCCAT"; int32_t thre = 14; clear_align(exz); clear_align(exz64);
|
||||||
// ed_band_cal_extension_64_0_w_trace((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz);
|
// ed_band_cal_extension_64_0_w_trace((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz);
|
||||||
// // // ed_band_cal_semi_64_w_absent_diag((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, 0, &exz);
|
// // // ed_band_cal_semi_64_w_absent_diag((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, 0, &exz);
|
||||||
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz.ps::%d, exz.pe::%d, exz.ts::%d, exz.te::%d\n", __func__,
|
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz.ps::%d, exz.pe::%d, exz.ts::%d, exz.te::%d\n", __func__,
|
||||||
// exz.err, exz.ps, exz.pe, exz.ts, exz.te);
|
// exz.err, exz.ps, exz.pe, exz.ts, exz.te);
|
||||||
// cigar_check((char*)pstr, (char*)tsrt, &(exz));
|
// cigar_check((char*)pstr, (char*)tsrt, &(exz));
|
||||||
// ed_band_cal_extension_64_0_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz);
|
// ed_band_cal_extension_64_0_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz);
|
||||||
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz.ps::%d, exz.pe::%d, exz.ts::%d, exz.te::%d\n", __func__,
|
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz.ps::%d, exz.pe::%d, exz.ts::%d, exz.te::%d\n", __func__,
|
||||||
// exz.err, exz.ps, exz.pe, exz.ts, exz.te);
|
// exz.err, exz.ps, exz.pe, exz.ts, exz.te);
|
||||||
// ed_band_cal_semi_infi_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, NULL, &exz);
|
// ed_band_cal_semi_infi_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, NULL, &exz);
|
||||||
// ed_band_cal_semi_64_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz64);
|
// ed_band_cal_semi_64_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz64);
|
||||||
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz64.err::%d, exz.ps::%d, exz64.ps::%d, exz.pe::%d, exz64.pe::%d, exz.ts::%d, exz64.ts::%d, exz.te::%d, exz64.te::%d\n", __func__,
|
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz64.err::%d, exz.ps::%d, exz64.ps::%d, exz.pe::%d, exz64.pe::%d, exz.ts::%d, exz64.ts::%d, exz.te::%d, exz64.te::%d\n", __func__,
|
||||||
// exz.err, exz64.err, exz.ps, exz64.ps, exz.pe, exz64.pe, exz.ts, exz64.ts, exz.te, exz64.te);
|
// exz.err, exz64.err, exz.ps, exz64.ps, exz.pe, exz64.pe, exz.ts, exz64.ts, exz.te, exz64.te);
|
||||||
// ed_band_cal_semi_64_w_trace((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz64);
|
// ed_band_cal_semi_64_w_trace((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz64);
|
||||||
// cigar_check((char*)pstr, (char*)tsrt, &(exz64));
|
// cigar_check((char*)pstr, (char*)tsrt, &(exz64));
|
||||||
|
|
||||||
|
|
||||||
// char *pstr = "TGT", *tsrt = "CTGT"; int32_t thre = 1;
|
// char *pstr = "TGT", *tsrt = "CTGT"; int32_t thre = 1;
|
||||||
// ed_band_cal_global_infi_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, NULL, &exz);
|
// ed_band_cal_global_infi_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, NULL, &exz);
|
||||||
// ed_band_cal_global_64_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz64);
|
// ed_band_cal_global_64_w((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz64);
|
||||||
// ed_band_cal_global_64_w_trace((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz64);
|
// ed_band_cal_global_64_w_trace((char*)pstr, strlen(pstr), (char*)tsrt, strlen(tsrt), thre, &exz64);
|
||||||
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz64.err::%d, exz.ps::%d, exz64.ps::%d, exz.pe::%d, exz64.pe::%d, exz.ts::%d, exz64.ts::%d, exz.te::%d, exz64.te::%d\n", __func__,
|
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz64.err::%d, exz.ps::%d, exz64.ps::%d, exz.pe::%d, exz64.pe::%d, exz.ts::%d, exz64.ts::%d, exz.te::%d, exz64.te::%d\n", __func__,
|
||||||
// exz.err, exz64.err, exz.ps, exz64.ps, exz.pe, exz64.pe, exz.ts, exz64.ts, exz.te, exz64.te);
|
// exz.err, exz64.err, exz.ps, exz64.ps, exz.pe, exz64.pe, exz.ts, exz64.ts, exz.te, exz64.te);
|
||||||
// cigar_check((char*)pstr, (char*)tsrt, &(exz64));
|
// cigar_check((char*)pstr, (char*)tsrt, &(exz64));
|
||||||
|
|
||||||
// ed_band_cal_extension_infi0_w((char *)"AAT", 3, (char *)"ACTTTTTT", 8, 2, NULL, &exz);
|
// ed_band_cal_extension_infi0_w((char *)"AAT", 3, (char *)"ACTTTTTT", 8, 2, NULL, &exz);
|
||||||
// ed_band_cal_extension_64_w((char *)"AAT", 3, (char *)"ACTTTTTT", 8, 2, &exz64);
|
// ed_band_cal_extension_64_w((char *)"AAT", 3, (char *)"ACTTTTTT", 8, 2, &exz64);
|
||||||
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz64.err::%d, exz.ps::%d, exz64.ps::%d, exz.pe::%d, exz64.pe::%d, exz.ts::%d, exz64.ts::%d, exz.te::%d, exz64.te::%d\n", __func__,
|
// fprintf(stderr, "\n[M::%s::] exz.err::%d, exz64.err::%d, exz.ps::%d, exz64.ps::%d, exz.pe::%d, exz64.pe::%d, exz.ts::%d, exz64.ts::%d, exz.te::%d, exz64.te::%d\n", __func__,
|
||||||
// exz.err, exz64.err, exz.ps, exz64.ps, exz.pe, exz64.pe, exz.ts, exz64.ts, exz.te, exz64.te);
|
// exz.err, exz64.err, exz.ps, exz64.ps, exz.pe, exz64.pe, exz.ts, exz64.ts, exz.te, exz64.te);
|
||||||
|
|
||||||
//bit_extz_t exz; ///ed_band_cal_global_128bit(t_string+r_ts, t_end+1-r_ts, q_string, ql, thres, &exz);
|
//bit_extz_t exz; ///ed_band_cal_global_128bit(t_string+r_ts, t_end+1-r_ts, q_string, ql, thres, &exz);
|
||||||
// ed_band_cal_extension_128bit((char *)"AAGTTTA", 7, (char *)"CCTTTTTT", 8, 4, &exz);
|
// ed_band_cal_extension_128bit((char *)"AAGTTTA", 7, (char *)"CCTTTTTT", 8, 4, &exz);
|
||||||
// ed_band_cal_extension_128bit((char *)"AA", 2, (char *)"ACTTTTTT", 8, 1, &exz);
|
// ed_band_cal_extension_128bit((char *)"AA", 2, (char *)"ACTTTTTT", 8, 1, &exz);
|
||||||
// fprintf(stderr, "ed_extension::%d, pe::%d, te::%d\n", exz.err, exz.pe, exz.te);
|
// fprintf(stderr, "ed_extension::%d, pe::%d, te::%d\n", exz.err, exz.pe, exz.te);
|
||||||
// exit(1);
|
// exit(1);
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
// fprintf(stderr, "[M::%s::] ed_global::%d, ed_global_128bit::%d\n", __func__,
|
// fprintf(stderr, "[M::%s::] ed_global::%d, ed_global_128bit::%d\n", __func__,
|
||||||
// ed_band_cal_global((char *)"ACT", 3, (char *)"AAT", 3, 1),
|
// ed_band_cal_global((char *)"ACT", 3, (char *)"AAT", 3, 1),
|
||||||
// ed_band_cal_global_128bit((char *)"ACT", 3, (char *)"AAT", 3, 1));
|
// ed_band_cal_global_128bit((char *)"ACT", 3, (char *)"AAT", 3, 1));
|
||||||
|
|
||||||
// fprintf(stderr, "[M::%s::] ed_global::%d, ed_global_128bit::%d\n", __func__,
|
// fprintf(stderr, "[M::%s::] ed_global::%d, ed_global_128bit::%d\n", __func__,
|
||||||
// ed_band_cal_global((char*)"ACTTTTTT", 8, (char*)"AATTTT", 6, 3),
|
// ed_band_cal_global((char*)"ACTTTTTT", 8, (char*)"AATTTT", 6, 3),
|
||||||
// ed_band_cal_global_128bit((char*)"ACTTTTTT", 8, (char*)"AATTTT", 6, 3));
|
// ed_band_cal_global_128bit((char*)"ACTTTTTT", 8, (char*)"AATTTT", 6, 3));
|
||||||
// exit(1);
|
// exit(1);
|
||||||
if(asm_opt.sec_in) ret = ha_assemble_pair();
|
if(asm_opt.sec_in) ret = ha_assemble_pair();
|
||||||
else if(asm_opt.dbg_ovec_cal) ret = ha_ec_dbg();
|
else if(asm_opt.dbg_ovec_cal) ret = ha_ec_dbg();
|
||||||
else ret = ha_assemble();
|
else ret = ha_assemble();
|
||||||
|
|
||||||
destory_opt(&asm_opt);
|
destory_opt(&asm_opt);
|
||||||
fprintf(stderr, "[M::%s] Version: %s\n", __func__, HA_VERSION);
|
fprintf(stderr, "[M::%s] Version: %s\n", __func__, HA_VERSION);
|
||||||
fprintf(stderr, "[M::%s] CMD:", __func__);
|
fprintf(stderr, "[M::%s] CMD:", __func__);
|
||||||
for (i = 0; i < argc; ++i)
|
for (i = 0; i < argc; ++i)
|
||||||
fprintf(stderr, " %s", argv[i]);
|
fprintf(stderr, " %s", argv[i]);
|
||||||
fprintf(stderr, "\n[M::%s] Real time: %.3f sec; CPU: %.3f sec; Peak RSS: %.3f GB\n", __func__, yak_realtime(), yak_cputime(), yak_peakrss_in_gb());
|
fprintf(stderr, "\n[M::%s] Real time: %.3f sec; CPU: %.3f sec; Peak RSS: %.3f GB\n", __func__, yak_realtime(), yak_cputime(), yak_peakrss_in_gb());
|
||||||
return ret;
|
return ret;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,131 +1,131 @@
|
|||||||
#ifndef __RCUT__
|
#ifndef __RCUT__
|
||||||
#define __RCUT__
|
#define __RCUT__
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <stdio.h>
|
#include <stdio.h>
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include "kvec.h"
|
#include "kvec.h"
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
#include "Purge_Dups.h"
|
#include "Purge_Dups.h"
|
||||||
#include "hic.h"
|
#include "hic.h"
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint32_t bS, bE;
|
uint32_t bS, bE;
|
||||||
uint32_t nS, nE;
|
uint32_t nS, nE;
|
||||||
uint32_t uID;
|
uint32_t uID;
|
||||||
uint8_t hs;
|
uint8_t hs;
|
||||||
}mc_interval_t;
|
}mc_interval_t;
|
||||||
|
|
||||||
#define mc_node_t int8_t
|
#define mc_node_t int8_t
|
||||||
#define mcg_node_t uint32_t
|
#define mcg_node_t uint32_t
|
||||||
// #define w_t int64_t
|
// #define w_t int64_t
|
||||||
// #define t_w_t int64_t
|
// #define t_w_t int64_t
|
||||||
// #define w_cast(x) ((t_w_t)((x) < 0 ? (x) - 0.5 : (x) + 0.5))
|
// #define w_cast(x) ((t_w_t)((x) < 0 ? (x) - 0.5 : (x) + 0.5))
|
||||||
|
|
||||||
#define w_t double
|
#define w_t double
|
||||||
#define t_w_t double
|
#define t_w_t double
|
||||||
#define w_cast(x) ((t_w_t)((x)))
|
#define w_cast(x) ((t_w_t)((x)))
|
||||||
#define MC_NAME "debug_mc.bin"
|
#define MC_NAME "debug_mc.bin"
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint64_t x; ///(uint64_t)nid1 << 32 | nid2;
|
uint64_t x; ///(uint64_t)nid1 << 32 | nid2;
|
||||||
w_t w; ///might be negative or positive
|
w_t w; ///might be negative or positive
|
||||||
} mc_edge_t;
|
} mc_edge_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
kvec_t(uint64_t) idx;
|
kvec_t(uint64_t) idx;
|
||||||
kvec_t(mc_edge_t) ma;
|
kvec_t(mc_edge_t) ma;
|
||||||
uint64_t* cc;
|
uint64_t* cc;
|
||||||
uint32_t n_seq;
|
uint32_t n_seq;
|
||||||
} mc_match_t;
|
} mc_match_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
kvec_t(mc_node_t) s;
|
kvec_t(mc_node_t) s;
|
||||||
ma_ug_t *ug;
|
ma_ug_t *ug;
|
||||||
asg_t *rg;
|
asg_t *rg;
|
||||||
mc_match_t* e;
|
mc_match_t* e;
|
||||||
}mc_g_t;
|
}mc_g_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint32_t a[2], occ[2];
|
uint32_t a[2], occ[2];
|
||||||
mc_node_t s[2];
|
mc_node_t s[2];
|
||||||
t_w_t z[4];
|
t_w_t z[4];
|
||||||
}mb_node_t;
|
}mb_node_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
kvec_t(uint32_t) bid;
|
kvec_t(uint32_t) bid;
|
||||||
kvec_t(uint32_t) idx;
|
kvec_t(uint32_t) idx;
|
||||||
kvec_t(mb_node_t) u;
|
kvec_t(mb_node_t) u;
|
||||||
}mb_nodes_t;
|
}mb_nodes_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
uint64_t x; ///(uint64_t)nid1 << 32 | nid2;
|
uint64_t x; ///(uint64_t)nid1 << 32 | nid2;
|
||||||
t_w_t w[4]; ///might be negative or positive
|
t_w_t w[4]; ///might be negative or positive
|
||||||
} mb_edge_t;
|
} mb_edge_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
kvec_t(uint64_t) idx;
|
kvec_t(uint64_t) idx;
|
||||||
kvec_t(mb_edge_t) ma;
|
kvec_t(mb_edge_t) ma;
|
||||||
uint64_t* cc;
|
uint64_t* cc;
|
||||||
uint32_t n_seq;
|
uint32_t n_seq;
|
||||||
} mb_match_t;
|
} mb_match_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
mb_nodes_t* u;
|
mb_nodes_t* u;
|
||||||
mb_match_t* e;
|
mb_match_t* e;
|
||||||
}mb_g_t;
|
}mb_g_t;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
mcg_node_t s;
|
mcg_node_t s;
|
||||||
uint16_t h[2], hc;
|
uint16_t h[2], hc;
|
||||||
double hw[2];
|
double hw[2];
|
||||||
}mc_gg_status;
|
}mc_gg_status;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
mc_gg_status *a;
|
mc_gg_status *a;
|
||||||
size_t n, m;
|
size_t n, m;
|
||||||
}kv_gg_status;
|
}kv_gg_status;
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
mcg_node_t *a;
|
mcg_node_t *a;
|
||||||
size_t n, m;
|
size_t n, m;
|
||||||
}mcb_t;
|
}mcb_t;
|
||||||
|
|
||||||
|
|
||||||
typedef struct {
|
typedef struct {
|
||||||
kv_gg_status *s;
|
kv_gg_status *s;
|
||||||
// ma_ug_t *ug;
|
// ma_ug_t *ug;
|
||||||
// asg_t *rg;
|
// asg_t *rg;
|
||||||
uint32_t un;
|
uint32_t un;
|
||||||
mc_match_t* e;
|
mc_match_t* e;
|
||||||
kvec_t(mcb_t) m;
|
kvec_t(mcb_t) m;
|
||||||
mcg_node_t mask;
|
mcg_node_t mask;
|
||||||
uint16_t hN;
|
uint16_t hN;
|
||||||
}mc_gg_t;
|
}mc_gg_t;
|
||||||
|
|
||||||
|
|
||||||
static inline uint64_t kr_splitmix64(uint64_t x)
|
static inline uint64_t kr_splitmix64(uint64_t x)
|
||||||
{
|
{
|
||||||
uint64_t z = (x += 0x9E3779B97F4A7C15ULL);
|
uint64_t z = (x += 0x9E3779B97F4A7C15ULL);
|
||||||
z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9ULL;
|
z = (z ^ (z >> 30)) * 0xBF58476D1CE4E5B9ULL;
|
||||||
z = (z ^ (z >> 27)) * 0x94D049BB133111EBULL;
|
z = (z ^ (z >> 27)) * 0x94D049BB133111EBULL;
|
||||||
return z ^ (z >> 31);
|
return z ^ (z >> 31);
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline double kr_drand_r(uint64_t *x)
|
static inline double kr_drand_r(uint64_t *x)
|
||||||
{
|
{
|
||||||
union { uint64_t i; double d; } u;
|
union { uint64_t i; double d; } u;
|
||||||
*x = kr_splitmix64(*x);
|
*x = kr_splitmix64(*x);
|
||||||
u.i = 0x3FFULL << 52 | (*x) >> 12;
|
u.i = 0x3FFULL << 52 | (*x) >> 12;
|
||||||
return u.d - 1.0;
|
return u.d - 1.0;
|
||||||
}
|
}
|
||||||
|
|
||||||
void mc_solve(hap_overlaps_list* ovlp, trans_chain* t_ch, kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g, double f_rate, uint8_t* trio_flag, uint32_t renew_s, int8_t *s, uint32_t is_sys, bubble_type* bub, kv_u_trans_t *ref, int clean_ov, int is_dump);
|
void mc_solve(hap_overlaps_list* ovlp, trans_chain* t_ch, kv_u_trans_t *ta, ma_ug_t *ug, asg_t *read_g, double f_rate, uint8_t* trio_flag, uint32_t renew_s, int8_t *s, uint32_t is_sys, bubble_type* bub, kv_u_trans_t *ref, int clean_ov, int is_dump);
|
||||||
void debug_mc_g_t(const char* name);
|
void debug_mc_g_t(const char* name);
|
||||||
void mc_solve_general(kv_u_trans_t *ta, uint32_t un, kv_gg_status *s, uint16_t hapN, uint16_t update_ta, uint16_t write_dump);
|
void mc_solve_general(kv_u_trans_t *ta, uint32_t un, kv_gg_status *s, uint16_t hapN, uint16_t update_ta, uint16_t write_dump);
|
||||||
kv_gg_status *init_mc_gg_status(ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut,
|
kv_gg_status *init_mc_gg_status(ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut,
|
||||||
ma_hit_t_alloc* sources, R_to_U* ruIndex, uint64_t t_cov, uint16_t hapN);
|
ma_hit_t_alloc* sources, R_to_U* ruIndex, uint64_t t_cov, uint16_t hapN);
|
||||||
void destory_mc_gg_t(mc_gg_t **p);
|
void destory_mc_gg_t(mc_gg_t **p);
|
||||||
void debug_mc_gg_t(const char* fn, uint32_t update_ta, uint32_t convert_mc_g_t);
|
void debug_mc_gg_t(const char* fn, uint32_t update_ta, uint32_t convert_mc_g_t);
|
||||||
void quick_debug_phasing(const char* fn);
|
void quick_debug_phasing(const char* fn);
|
||||||
#endif
|
#endif
|
||||||
+581
-581
File diff suppressed because it is too large
Load Diff
@@ -1,59 +1,59 @@
|
|||||||
#include <sys/resource.h>
|
#include <sys/resource.h>
|
||||||
#include <sys/time.h>
|
#include <sys/time.h>
|
||||||
#include "htab.h"
|
#include "htab.h"
|
||||||
|
|
||||||
int yak_verbose = 3;
|
int yak_verbose = 3;
|
||||||
|
|
||||||
static double yak_realtime0;
|
static double yak_realtime0;
|
||||||
|
|
||||||
double yak_cputime(void)
|
double yak_cputime(void)
|
||||||
{
|
{
|
||||||
struct rusage r;
|
struct rusage r;
|
||||||
getrusage(RUSAGE_SELF, &r);
|
getrusage(RUSAGE_SELF, &r);
|
||||||
return r.ru_utime.tv_sec + r.ru_stime.tv_sec + 1e-6 * (r.ru_utime.tv_usec + r.ru_stime.tv_usec);
|
return r.ru_utime.tv_sec + r.ru_stime.tv_sec + 1e-6 * (r.ru_utime.tv_usec + r.ru_stime.tv_usec);
|
||||||
}
|
}
|
||||||
|
|
||||||
static inline double yak_realtime_core(void)
|
static inline double yak_realtime_core(void)
|
||||||
{
|
{
|
||||||
struct timeval tp;
|
struct timeval tp;
|
||||||
struct timezone tzp;
|
struct timezone tzp;
|
||||||
gettimeofday(&tp, &tzp);
|
gettimeofday(&tp, &tzp);
|
||||||
return tp.tv_sec + tp.tv_usec * 1e-6;
|
return tp.tv_sec + tp.tv_usec * 1e-6;
|
||||||
}
|
}
|
||||||
|
|
||||||
void yak_reset_realtime(void)
|
void yak_reset_realtime(void)
|
||||||
{
|
{
|
||||||
yak_realtime0 = yak_realtime_core();
|
yak_realtime0 = yak_realtime_core();
|
||||||
}
|
}
|
||||||
|
|
||||||
double yak_realtime(void)
|
double yak_realtime(void)
|
||||||
{
|
{
|
||||||
return yak_realtime_core() - yak_realtime0;
|
return yak_realtime_core() - yak_realtime0;
|
||||||
}
|
}
|
||||||
|
|
||||||
double yak_realtime_0(void)
|
double yak_realtime_0(void)
|
||||||
{
|
{
|
||||||
return yak_realtime_core();
|
return yak_realtime_core();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
long yak_peakrss(void)
|
long yak_peakrss(void)
|
||||||
{
|
{
|
||||||
struct rusage r;
|
struct rusage r;
|
||||||
getrusage(RUSAGE_SELF, &r);
|
getrusage(RUSAGE_SELF, &r);
|
||||||
#ifdef __linux__
|
#ifdef __linux__
|
||||||
return r.ru_maxrss * 1024;
|
return r.ru_maxrss * 1024;
|
||||||
#else
|
#else
|
||||||
return r.ru_maxrss;
|
return r.ru_maxrss;
|
||||||
#endif
|
#endif
|
||||||
}
|
}
|
||||||
|
|
||||||
double yak_peakrss_in_gb(void)
|
double yak_peakrss_in_gb(void)
|
||||||
{
|
{
|
||||||
return yak_peakrss() / 1073741824.0;
|
return yak_peakrss() / 1073741824.0;
|
||||||
}
|
}
|
||||||
|
|
||||||
double yak_cpu_usage(void)
|
double yak_cpu_usage(void)
|
||||||
{
|
{
|
||||||
return (yak_cputime() + 1e-9) / (yak_realtime() + 1e-9);
|
return (yak_cputime() + 1e-9) / (yak_realtime() + 1e-9);
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,28 +1,28 @@
|
|||||||
#ifndef __TOVLP__
|
#ifndef __TOVLP__
|
||||||
#define __TOVLP__
|
#define __TOVLP__
|
||||||
|
|
||||||
#define __STDC_LIMIT_MACROS
|
#define __STDC_LIMIT_MACROS
|
||||||
#include <stdint.h>
|
#include <stdint.h>
|
||||||
#include "Overlaps.h"
|
#include "Overlaps.h"
|
||||||
|
|
||||||
typedef struct {///[cBeg, cEnd)
|
typedef struct {///[cBeg, cEnd)
|
||||||
uint32_t ui, len, cBeg, cEnd;
|
uint32_t ui, len, cBeg, cEnd;
|
||||||
uint32_t *a, an;
|
uint32_t *a, an;
|
||||||
ma_ug_t *ug;
|
ma_ug_t *ug;
|
||||||
utg_trans_t *o;
|
utg_trans_t *o;
|
||||||
} utg_trans_hit_idx;
|
} utg_trans_hit_idx;
|
||||||
|
|
||||||
utg_trans_t *init_utg_trans_t(ma_ug_t *ug, ma_hit_t_alloc* reverse_sources, ma_sub_t *coverage_cut, R_to_U* ruIndex, asg_t *read_g, int max_hang, int min_ovlp);
|
utg_trans_t *init_utg_trans_t(ma_ug_t *ug, ma_hit_t_alloc* reverse_sources, ma_sub_t *coverage_cut, R_to_U* ruIndex, asg_t *read_g, int max_hang, int min_ovlp);
|
||||||
void destroy_utg_trans_t(utg_trans_t **o);
|
void destroy_utg_trans_t(utg_trans_t **o);
|
||||||
void asg_bub_collect_ovlp(ma_ug_t *ug, uint32_t v0, buf_t *b, utg_trans_t *o);
|
void asg_bub_collect_ovlp(ma_ug_t *ug, uint32_t v0, buf_t *b, utg_trans_t *o);
|
||||||
void collect_trans_ovlp(const char* cmd, buf_t* pri, uint64_t pri_offset, buf_t* aux, uint64_t aux_offset,
|
void collect_trans_ovlp(const char* cmd, buf_t* pri, uint64_t pri_offset, buf_t* aux, uint64_t aux_offset,
|
||||||
ma_ug_t *ug, utg_trans_t *o);
|
ma_ug_t *ug, utg_trans_t *o);
|
||||||
int asg_arc_decompress(asg_t *g, ma_ug_t *ug, asg_t *read_sg, ma_hit_t_alloc* reverse_sources,
|
int asg_arc_decompress(asg_t *g, ma_ug_t *ug, asg_t *read_sg, ma_hit_t_alloc* reverse_sources,
|
||||||
R_to_U* ruIndex, utg_trans_t *o);
|
R_to_U* ruIndex, utg_trans_t *o);
|
||||||
int asg_arc_decompress_mul(asg_t *g, ma_ug_t *ug, asg_t *read_sg, uint32_t positive_flag, uint32_t negative_flag,
|
int asg_arc_decompress_mul(asg_t *g, ma_ug_t *ug, asg_t *read_sg, uint32_t positive_flag, uint32_t negative_flag,
|
||||||
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, utg_trans_t *o);
|
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, utg_trans_t *o);
|
||||||
kv_u_trans_t *pt_pdist(ma_ug_t *ug, asg_t *read_g, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources,
|
kv_u_trans_t *pt_pdist(ma_ug_t *ug, asg_t *read_g, ma_sub_t *coverage_cut, ma_hit_t_alloc* sources,
|
||||||
kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp, uint32_t min_chain_cnt);
|
kvec_asg_arc_t_warp* edge, int max_hang, int min_ovlp, uint32_t min_chain_cnt);
|
||||||
void reset_utg_trans_hit_idx(utg_trans_hit_idx *t, uint32_t* i_x_a, uint32_t i_x_n, ma_ug_t *i_ug,
|
void reset_utg_trans_hit_idx(utg_trans_hit_idx *t, uint32_t* i_x_a, uint32_t i_x_n, ma_ug_t *i_ug,
|
||||||
utg_trans_t *i_o, uint32_t i_cBeg, uint32_t i_cEnd);
|
utg_trans_t *i_o, uint32_t i_cBeg, uint32_t i_cEnd);
|
||||||
#endif
|
#endif
|
||||||
|
|||||||
Reference in New Issue
Block a user