mirror of
https://github.com/chhylp123/hifiasm.git
synced 2026-09-15 20:57:57 +08:00
Compare commits
166 Commits
0.1.0
...
hifiasm_hi
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9d659fd2e0 | ||
|
|
ace34ae2b8 | ||
|
|
94008c02ac | ||
|
|
1e4d67ee64 | ||
|
|
ff4ef89fe3 | ||
|
|
952471657f | ||
|
|
88ab801717 | ||
|
|
a309e610ca | ||
|
|
fda858f9e7 | ||
|
|
c01b593d7c | ||
|
|
a25c37ed89 | ||
|
|
2e56ef7db5 | ||
|
|
309e885365 | ||
|
|
5cb42cafc2 | ||
|
|
428cc01d10 | ||
|
|
3481a2fe2b | ||
|
|
f1e78720aa | ||
|
|
e401f88e57 | ||
|
|
7a2e02cb77 | ||
|
|
3a88079a6e | ||
|
|
be26acc0e0 | ||
|
|
03c9028848 | ||
|
|
0071251c82 | ||
|
|
39cf3f5dd8 | ||
|
|
acba806ce1 | ||
|
|
46987e792a | ||
|
|
1ec650e3e5 | ||
|
|
61657f4b29 | ||
|
|
35ee94f040 | ||
|
|
ed137c2ac6 | ||
|
|
9cc97472d9 | ||
|
|
449950cddb | ||
|
|
71f6c93d58 | ||
|
|
97f6efc4da | ||
|
|
eddbc61952 | ||
|
|
e247c5f042 | ||
|
|
d36e1e7782 | ||
|
|
f6ea45b58a | ||
|
|
e544360043 | ||
|
|
3364a29b87 | ||
|
|
331e437c72 | ||
|
|
7f6725ead3 | ||
|
|
85c57fd087 | ||
|
|
57e979ca17 | ||
|
|
a06ee39433 | ||
|
|
8907e9c64f | ||
|
|
946a0f4115 | ||
|
|
b333addb6c | ||
|
|
4216384df6 | ||
|
|
3120db1340 | ||
|
|
69e8282b9a | ||
|
|
b93baa3cbd | ||
|
|
cc166c8ff5 | ||
|
|
cdd5f3e4e0 | ||
|
|
2d985569f1 | ||
|
|
57b16e9e1f | ||
|
|
b245e9a760 | ||
|
|
2c4de3a326 | ||
|
|
75a89c214d | ||
|
|
22b681b830 | ||
|
|
4d7600361c | ||
|
|
c4397a9400 | ||
|
|
e6b5b666a2 | ||
|
|
54fa1a9aed | ||
|
|
20c6c1d2e6 | ||
|
|
c710c6ea48 | ||
|
|
c37ea00d6a | ||
|
|
ad1b79a6bb | ||
|
|
943b6947cb | ||
|
|
22b2e1e9e1 | ||
|
|
15591c9038 | ||
|
|
71328f7423 | ||
|
|
f32bfc904a | ||
|
|
7a35bd7fcc | ||
|
|
c3f032da37 | ||
|
|
c41aae0630 | ||
|
|
23dbfdef77 | ||
|
|
a288415111 | ||
|
|
8efcdefcaf | ||
|
|
9cc563c7c6 | ||
|
|
a79fd0f360 | ||
|
|
1bbca54d8d | ||
|
|
2d0087104f | ||
|
|
7f580850e8 | ||
|
|
87fc103e63 | ||
|
|
adbc65feda | ||
|
|
3773610532 | ||
|
|
46f83e152d | ||
|
|
0f1994fc1f | ||
|
|
31356f9e02 | ||
|
|
16a3d58dc9 | ||
|
|
5917aa2f36 | ||
|
|
c57b63653d | ||
|
|
f0c53e9948 | ||
|
|
ac3fcc339a | ||
|
|
8c27dbb52f | ||
|
|
4e3c15ec83 | ||
|
|
dbe66d259d | ||
|
|
69c32066d4 | ||
|
|
18b69f1a72 | ||
|
|
76de054331 | ||
|
|
992322f8e3 | ||
|
|
6f4d0debe4 | ||
|
|
d529dcea3f | ||
|
|
ffba3ce0ef | ||
|
|
968b4caef9 | ||
|
|
3838482851 | ||
|
|
483ceb852c | ||
|
|
7776dee103 | ||
|
|
878fe943c3 | ||
|
|
b7e5d1c4d3 | ||
|
|
e75b93ae12 | ||
|
|
1d4f34a36c | ||
|
|
4fa3cbfd46 | ||
|
|
5e16483b43 | ||
|
|
19648873b9 | ||
|
|
cb20e50219 | ||
|
|
7079a9f306 | ||
|
|
bd89bd500c | ||
|
|
30a70bc307 | ||
|
|
1bb174a04e | ||
|
|
78e8f2f28a | ||
|
|
8dbb4140bd | ||
|
|
293f4b6b58 | ||
|
|
7271c106e4 | ||
|
|
10ed8b36ec | ||
|
|
4aa4ebaf96 | ||
|
|
814a3705e2 | ||
|
|
8f664f80ce | ||
|
|
79bc553da6 | ||
|
|
387d6336d8 | ||
|
|
0dd927bdce | ||
|
|
c3bbaa8a39 | ||
|
|
60d94790a2 | ||
|
|
ff8194d291 | ||
|
|
716713685c | ||
|
|
5a6b97145a | ||
|
|
e116f6e09d | ||
|
|
585bbbe213 | ||
|
|
f8351e557a | ||
|
|
3e897560c4 | ||
|
|
84361ba6dd | ||
|
|
3341cf20ba | ||
|
|
f70257c260 | ||
|
|
ec750cdd7d | ||
|
|
2a848d254a | ||
|
|
64edb06e08 | ||
|
|
398e73022b | ||
|
|
59a9d62df9 | ||
|
|
e6ef2fb56f | ||
|
|
3a27d104c6 | ||
|
|
cfb0a5c8ec | ||
|
|
3c1d3cf6a1 | ||
|
|
ee2573a05c | ||
|
|
d3cb016aba | ||
|
|
670bd10093 | ||
|
|
90e290636a | ||
|
|
c72d419711 | ||
|
|
4536141b9a | ||
|
|
6ae51c84f6 | ||
|
|
6246132702 | ||
|
|
0a3fdd9599 | ||
|
|
6edf54048c | ||
|
|
c63ce670f7 | ||
|
|
387c64c5fc | ||
|
|
67f1173eb6 |
1797
Assembly.cpp
1797
Assembly.cpp
File diff suppressed because it is too large
Load Diff
@@ -8,9 +8,6 @@
|
||||
#define Get_Cigar_Type(RECORD) (RECORD&3)
|
||||
#define Get_Cigar_Length(RECORD) (RECORD>>2)
|
||||
|
||||
void Counting_multiple_thr();
|
||||
void Build_hash_table_multiple_thr();
|
||||
///int load_pre_cauculated_index();
|
||||
void Overlap_calculate_multipe_thr();
|
||||
void Correct_Reads(int last_round);
|
||||
int ha_assemble(void);
|
||||
|
||||
#endif
|
||||
|
||||
240
CommandLines.cpp
240
CommandLines.cpp
@@ -1,15 +1,32 @@
|
||||
#include "CommandLines.h"
|
||||
#include <zlib.h>
|
||||
#include <stdlib.h>
|
||||
#include <stdio.h>
|
||||
#include "ketopt.h"
|
||||
#include <stdint.h>
|
||||
#include <limits.h>
|
||||
#include <sys/time.h>
|
||||
#include "CommandLines.h"
|
||||
#include "ketopt.h"
|
||||
|
||||
#define VERSION "0.1.0"
|
||||
#define DEFAULT_OUTPUT "hifiasm.asm"
|
||||
|
||||
hifiasm_opt_t asm_opt;
|
||||
|
||||
static ko_longopt_t long_options[] = {
|
||||
{ "version", ko_no_argument, 300 },
|
||||
{ "dbg-gfa", ko_no_argument, 301 },
|
||||
{ "write-paf", ko_no_argument, 302 },
|
||||
{ "write-ec", ko_no_argument, 303 },
|
||||
{ "skip-triobin", ko_no_argument, 304 },
|
||||
{ "max-od-ec", ko_no_argument, 305 },
|
||||
{ "max-od-final", ko_no_argument, 306 },
|
||||
{ "ex-list", ko_required_argument, 307 },
|
||||
{ "ex-iter", ko_required_argument, 308 },
|
||||
{ "purge-cov", ko_required_argument, 309 },
|
||||
{ "pri-range", ko_required_argument, 310 },
|
||||
{ "high-het", ko_no_argument, 311 },
|
||||
{ 0, 0, 0 }
|
||||
};
|
||||
|
||||
double Get_T(void)
|
||||
{
|
||||
struct timeval t;
|
||||
@@ -21,45 +38,79 @@ void Print_H(hifiasm_opt_t* asm_opt)
|
||||
{
|
||||
fprintf(stderr, "Usage: hifiasm [options] <in_1.fq> <in_2.fq> <...>\n");
|
||||
fprintf(stderr, "Options:\n");
|
||||
fprintf(stderr, " -o FILE prefix of output files [%s]\n", asm_opt->output_file_name);
|
||||
fprintf(stderr, " -t INT number of threads [%d]\n", asm_opt->thread_num);
|
||||
fprintf(stderr, " -r INT round of correction [%d]\n", asm_opt->number_of_round);
|
||||
fprintf(stderr, " -a INT round of assembly cleaning [%d]\n", asm_opt->clean_round);
|
||||
fprintf(stderr, " -k INT k-mer length [%d] (must be < 64)\n", asm_opt->k_mer_length);
|
||||
///fprintf(stderr, " -w write all overlaps to disk, can accelerate assembly next time [%d]\n", asm_opt->write_index_to_disk);
|
||||
///fprintf(stderr, " -l load all overlaps from disk, can avoid overlap calculation [%d]\n", asm_opt->load_index_from_disk);
|
||||
///fprintf(stderr, " -i ignore saved overlaps in *.ovlp*.bin files\n");
|
||||
fprintf(stderr, " -i ignore saved overlaps in *.ovlp* files\n");
|
||||
fprintf(stderr, " -z INT length of adapters that should be removed [%d]\n", asm_opt->adapterLen);
|
||||
fprintf(stderr, " -m INT size of popped large bubbles for contig graph [%lld]\n",
|
||||
asm_opt->large_pop_bubble_size);
|
||||
fprintf(stderr, " -p INT size of popped small bubbles for haplotype-resolved unitig graph [%lld]\n",
|
||||
asm_opt->small_pop_bubble_size);
|
||||
fprintf(stderr, " -n INT small removed unitig threshold [%d]\n", asm_opt->max_short_tip);
|
||||
fprintf(stderr, " -x FLOAT max overlap drop ratio [%.2g]\n", asm_opt->max_drop_rate);
|
||||
fprintf(stderr, " -y FLOAT min overlap drop ratio [%.2g]\n", asm_opt->min_drop_rate);
|
||||
fprintf(stderr, " -v show version number\n");
|
||||
fprintf(stderr, " -h show help information\n");
|
||||
fprintf(stderr, " Input/Output:\n");
|
||||
fprintf(stderr, " -o STR prefix of output files [%s]\n", asm_opt->output_file_name);
|
||||
fprintf(stderr, " -i ignore saved read correction and overlaps\n");
|
||||
fprintf(stderr, " -t INT number of threads [%d]\n", asm_opt->thread_num);
|
||||
fprintf(stderr, " -z INT length of adapters that should be removed [%d]\n", asm_opt->adapterLen);
|
||||
fprintf(stderr, " --version show version number\n");
|
||||
fprintf(stderr, " Overlap/Error correction:\n");
|
||||
fprintf(stderr, " -k INT k-mer length (must be <64) [%d]\n", asm_opt->k_mer_length);
|
||||
fprintf(stderr, " -w INT minimizer window size [%d]\n", asm_opt->mz_win);
|
||||
fprintf(stderr, " -f INT number of bits for bloom filter; 0 to disable [%d]\n", asm_opt->bf_shift);
|
||||
fprintf(stderr, " -D FLOAT drop k-mers occurring >FLOAT*coverage times [%.1f]\n", asm_opt->high_factor);
|
||||
fprintf(stderr, " -N INT consider up to max(-D*coverage,-N) overlaps for each oriented read [%d]\n", asm_opt->max_n_chain);
|
||||
fprintf(stderr, " -r INT round of correction [%d]\n", asm_opt->number_of_round);
|
||||
fprintf(stderr, " Assembly:\n");
|
||||
fprintf(stderr, " -a INT round of assembly cleaning [%d]\n", asm_opt->clean_round);
|
||||
fprintf(stderr, " -m INT pop bubbles of <INT in size in contig graphs [%lld]\n", asm_opt->large_pop_bubble_size);
|
||||
fprintf(stderr, " -p INT pop bubbles of <INT in size in unitig graphs [%lld]\n", asm_opt->small_pop_bubble_size);
|
||||
fprintf(stderr, " -n INT remove tip unitigs composed of <=INT reads [%d]\n", asm_opt->max_short_tip);
|
||||
fprintf(stderr, " -x FLOAT max overlap drop ratio [%.2g]\n", asm_opt->max_drop_rate);
|
||||
fprintf(stderr, " -y FLOAT min overlap drop ratio [%.2g]\n", asm_opt->min_drop_rate);
|
||||
fprintf(stderr, " -u disable post join contigs step which may improve N50\n");
|
||||
// fprintf(stderr, " --pri-range INT1[,INT2]\n");
|
||||
// fprintf(stderr, " keep contigs with coverage in this range in p_ctg.gfa; -1 to disable [auto,inf]\n");
|
||||
|
||||
fprintf(stderr, " Trio-partition:\n");
|
||||
fprintf(stderr, " -1 FILE hap1/paternal k-mer dump generated by \"yak count\" []\n");
|
||||
fprintf(stderr, " -2 FILE hap2/maternal k-mer dump generated by \"yak count\" []\n");
|
||||
fprintf(stderr, " -c INT lower bound of the binned k-mer's frequency [%d]\n", asm_opt->min_cnt);
|
||||
fprintf(stderr, " -d INT upper bound of the binned k-mer's frequency [%d]\n", asm_opt->mid_cnt);
|
||||
fprintf(stderr, " -3 FILE list of hap1/paternal read names []\n");
|
||||
fprintf(stderr, " -4 FILE list of hap2/maternal read names []\n");
|
||||
|
||||
fprintf(stderr, " Purge-dups:\n");
|
||||
fprintf(stderr, " -l INT purge level. 0: no purging; 1: light; 2: aggressive [0 for trio; 2 for unzip]\n");
|
||||
fprintf(stderr, " -s FLOAT similarity threshold for duplicate haplotigs [%g]\n",
|
||||
asm_opt->purge_simi_rate);
|
||||
fprintf(stderr, " -O INT min number of overlapped reads for duplicate haplotigs [%d]\n",
|
||||
asm_opt->purge_overlap_len);
|
||||
fprintf(stderr, " --purge-cov INT\n");
|
||||
fprintf(stderr, " coverage upper bound of Purge-dups [auto]\n");
|
||||
fprintf(stderr, " --high-het enable this mode for high heterozygosity sample\n");
|
||||
|
||||
|
||||
fprintf(stderr, "Example: ./hifiasm -o NA12878.asm -t 32 NA12878.fq.gz\n");
|
||||
fprintf(stderr, "See `man ./hifiasm.1' for detailed description of these command-line options.\n");
|
||||
}
|
||||
|
||||
void init_opt(hifiasm_opt_t* asm_opt)
|
||||
{
|
||||
memset(asm_opt, 0, sizeof(hifiasm_opt_t));
|
||||
asm_opt->flag = 0;
|
||||
asm_opt->coverage = -1;
|
||||
asm_opt->num_reads = 0;
|
||||
asm_opt->read_file_names = NULL;
|
||||
asm_opt->output_file_name = (char*)(DEFAULT_OUTPUT);
|
||||
asm_opt->required_read_name = NULL;
|
||||
asm_opt->thread_num = 1;
|
||||
asm_opt->k_mer_length = 40;
|
||||
asm_opt->k_mer_length = 51;
|
||||
asm_opt->mz_win = 51;
|
||||
asm_opt->bf_shift = 37;
|
||||
asm_opt->high_factor = 5.0;
|
||||
asm_opt->max_ov_diff_ec = 0.04;
|
||||
asm_opt->max_ov_diff_final = 0.03;
|
||||
asm_opt->hom_cov = 20;
|
||||
asm_opt->het_cov = -1024;
|
||||
asm_opt->max_n_chain = 100;
|
||||
asm_opt->k_mer_min_freq = 3;
|
||||
asm_opt->k_mer_max_freq = 66;
|
||||
asm_opt->load_index_from_disk = 1;
|
||||
asm_opt->write_index_to_disk = 1;
|
||||
asm_opt->number_of_round = 2;
|
||||
asm_opt->number_of_round = 3;
|
||||
asm_opt->adapterLen = 0;
|
||||
asm_opt->clean_round = 4;
|
||||
asm_opt->complete_threads = 0;
|
||||
asm_opt->small_pop_bubble_size = 100000;
|
||||
asm_opt->large_pop_bubble_size = 10000000;
|
||||
asm_opt->min_drop_rate = 0.2;
|
||||
@@ -70,6 +121,15 @@ void init_opt(hifiasm_opt_t* asm_opt)
|
||||
asm_opt->min_overlap_Len = 50;
|
||||
asm_opt->min_overlap_coverage = 0;
|
||||
asm_opt->max_short_tip = 3;
|
||||
asm_opt->min_cnt = 2;
|
||||
asm_opt->mid_cnt = 5;
|
||||
asm_opt->purge_level_primary = 2;
|
||||
asm_opt->purge_level_trio = 0;
|
||||
asm_opt->purge_simi_rate = 0.75;
|
||||
asm_opt->purge_overlap_len = 1;
|
||||
asm_opt->recover_atg_cov_min = -1024;
|
||||
asm_opt->recover_atg_cov_max = INT_MAX;
|
||||
asm_opt->hom_global_coverage = -1;
|
||||
}
|
||||
|
||||
void destory_opt(hifiasm_opt_t* asm_opt)
|
||||
@@ -80,15 +140,42 @@ void destory_opt(hifiasm_opt_t* asm_opt)
|
||||
}
|
||||
}
|
||||
|
||||
void clear_opt(hifiasm_opt_t* asm_opt, int last_round)
|
||||
void ha_opt_reset_to_round(hifiasm_opt_t* asm_opt, int round)
|
||||
{
|
||||
asm_opt->complete_threads = 0;
|
||||
asm_opt->num_bases = 0;
|
||||
asm_opt->num_corrected_bases = 0;
|
||||
asm_opt->num_recorrected_bases = 0;
|
||||
asm_opt->roundID = asm_opt->number_of_round - last_round;
|
||||
asm_opt->mem_buf = 0;
|
||||
asm_opt->roundID = round;
|
||||
}
|
||||
|
||||
void ha_opt_update_cov(hifiasm_opt_t *opt, int hom_cov)
|
||||
{
|
||||
int max_n_chain = (int)(hom_cov * opt->high_factor + .499);
|
||||
opt->hom_cov = hom_cov;
|
||||
if (opt->max_n_chain < max_n_chain)
|
||||
opt->max_n_chain = max_n_chain;
|
||||
fprintf(stderr, "[M::%s] updated max_n_chain to %d\n", __func__, opt->max_n_chain);
|
||||
}
|
||||
|
||||
static int check_file(char* name, const char* opt)
|
||||
{
|
||||
if(!name)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] file does not exist (-%s)\n", opt);
|
||||
return 0;
|
||||
}
|
||||
FILE* is_exist = NULL;
|
||||
is_exist = fopen(name,"r");
|
||||
if(!is_exist)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] %s does not exist (-%s)\n", name, opt);
|
||||
return 0;
|
||||
}
|
||||
|
||||
fclose(is_exist);
|
||||
return 1;
|
||||
}
|
||||
|
||||
int check_option(hifiasm_opt_t* asm_opt)
|
||||
{
|
||||
@@ -198,6 +285,15 @@ int check_option(hifiasm_opt_t* asm_opt)
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (asm_opt->max_ov_diff_ec < asm_opt->max_ov_diff_final) {
|
||||
fprintf(stderr, "[ERROR] max_ov_diff_ec shouldn't be smaller than max_ov_diff_final\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (asm_opt->max_ov_diff_ec < HA_MIN_OV_DIFF) {
|
||||
fprintf(stderr, "[ERROR] max_ov_diff_ec shouldn't be smaller than %g\n", HA_MIN_OV_DIFF);
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->max_short_tip < 0)
|
||||
{
|
||||
@@ -205,6 +301,30 @@ int check_option(hifiasm_opt_t* asm_opt)
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->purge_level_primary < 0 || asm_opt->purge_level_primary > 2)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] the level of purge-dup should be [0, 2] (-l)\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(ha_opt_triobin(asm_opt) && ((asm_opt->purge_level_trio < 0 || asm_opt->purge_level_trio > 1)))
|
||||
{
|
||||
fprintf(stderr, "[ERROR] the level of purge-dup for trio should be [0, 1] (-l)\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
if(asm_opt->hom_global_coverage < 0 && asm_opt->hom_global_coverage != -1)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] purge duplication coverage threshold should be >= 0 (--purge-cov)\n");
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
if(asm_opt->fn_bin_yak[0] != NULL && check_file(asm_opt->fn_bin_yak[0], "YAK1") == 0) return 0;
|
||||
if(asm_opt->fn_bin_yak[1] != NULL && check_file(asm_opt->fn_bin_yak[1], "YAK2") == 0) return 0;
|
||||
if(asm_opt->fn_bin_list[0] != NULL && check_file(asm_opt->fn_bin_list[0], "LIST1") == 0) return 0;
|
||||
if(asm_opt->fn_bin_list[1] != NULL && check_file(asm_opt->fn_bin_list[1], "LIST2") == 0) return 0;
|
||||
|
||||
// fprintf(stderr, "input file num: %d\n", asm_opt->num_reads);
|
||||
// fprintf(stderr, "output file: %s\n", asm_opt->output_file_name);
|
||||
// fprintf(stderr, "number of threads: %d\n", asm_opt->thread_num);
|
||||
@@ -217,6 +337,13 @@ int check_option(hifiasm_opt_t* asm_opt)
|
||||
// fprintf(stderr, "size of popped small bubbles: %lld\n", asm_opt->small_pop_bubble_size);
|
||||
// fprintf(stderr, "size of popped large bubbles: %lld\n", asm_opt->large_pop_bubble_size);
|
||||
// fprintf(stderr, "small removed unitig threshold: %d\n", asm_opt->max_short_tip);
|
||||
// fprintf(stderr, "small removed unitig threshold: %d\n", asm_opt->max_short_tip);
|
||||
// fprintf(stderr, "min_cnt: %d\n", asm_opt->min_cnt);
|
||||
// fprintf(stderr, "mid_cnt: %d\n", asm_opt->mid_cnt);
|
||||
// fprintf(stderr, "purge_level_primary: %d\n", asm_opt->purge_level_primary);
|
||||
// fprintf(stderr, "purge_level_trio: %d\n", asm_opt->purge_level_trio);
|
||||
// fprintf(stderr, "purge_simi_rate: %f\n", asm_opt->purge_simi_rate);
|
||||
// fprintf(stderr, "purge_overlap_len: %d\n", asm_opt->purge_overlap_len);
|
||||
|
||||
return 1;
|
||||
}
|
||||
@@ -254,49 +381,88 @@ int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt)
|
||||
|
||||
int c;
|
||||
|
||||
while ((c = ketopt(&opt, argc, argv, 1, "hvt:o:k:lwm:n:r:a:b:z:x:y:p:i", 0)) >= 0) {
|
||||
while ((c = ketopt(&opt, argc, argv, 1, "hvt:o:k:w:m:n:r:a:b:z:x:y:p:c:d:M:P:if:D:FN:1:2:3:4:l:s:O:eu", long_options)) >= 0) {
|
||||
if (c == 'h')
|
||||
{
|
||||
Print_H(asm_opt);
|
||||
return 0;
|
||||
}
|
||||
else if (c == 'v')
|
||||
else if (c == 'v' || c == 300)
|
||||
{
|
||||
fprintf(stderr, "[Version] %s\n", VERSION);
|
||||
puts(HA_VERSION);
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
else if (c == 'f') asm_opt->bf_shift = atoi(opt.arg);
|
||||
else if (c == 't') asm_opt->thread_num = atoi(opt.arg);
|
||||
else if (c == 'o') asm_opt->output_file_name = opt.arg;
|
||||
else if (c == 'r') asm_opt->number_of_round = atoi(opt.arg);
|
||||
else if (c == 'k') asm_opt->k_mer_length = atoi(opt.arg);
|
||||
else if (c == 'i') asm_opt->load_index_from_disk = 0;
|
||||
else if (c == 'l') asm_opt->load_index_from_disk = 1;
|
||||
else if (c == 'w') asm_opt->write_index_to_disk = 1;
|
||||
else if (c == 'w') asm_opt->mz_win = atoi(opt.arg);
|
||||
else if (c == 'D') asm_opt->high_factor = atof(opt.arg);
|
||||
else if (c == 'F') asm_opt->flag |= HA_F_NO_KMER_FLT;
|
||||
else if (c == 'N') asm_opt->max_n_chain = atoi(opt.arg);
|
||||
else if (c == 'a') asm_opt->clean_round = atoi(opt.arg);
|
||||
else if (c == 'z') asm_opt->adapterLen = atoi(opt.arg);
|
||||
else if (c == 'b') asm_opt->required_read_name = opt.arg;
|
||||
else if (c == 'c') asm_opt->min_cnt = atoi(opt.arg);
|
||||
else if (c == 'd') asm_opt->mid_cnt = atoi(opt.arg);
|
||||
else if (c == '1' || c == 'P') asm_opt->fn_bin_yak[0] = opt.arg; // -P/-M reserved for backward compatibility
|
||||
else if (c == '2' || c == 'M') asm_opt->fn_bin_yak[1] = opt.arg;
|
||||
else if (c == '3') asm_opt->fn_bin_list[0] = opt.arg;
|
||||
else if (c == '4') asm_opt->fn_bin_list[1] = opt.arg;
|
||||
else if (c == 'x') asm_opt->max_drop_rate = atof(opt.arg);
|
||||
else if (c == 'y') asm_opt->min_drop_rate = atof(opt.arg);
|
||||
else if (c == 'p') asm_opt->small_pop_bubble_size = atoll(opt.arg);
|
||||
else if (c == 'm') asm_opt->large_pop_bubble_size = atoll(opt.arg);
|
||||
else if (c == 'n') asm_opt->max_short_tip = atoll(opt.arg);
|
||||
else if (c == 'e') asm_opt->flag |= HA_F_BAN_ASSEMBLY;
|
||||
else if (c == 'u') asm_opt->flag |= HA_F_BAN_POST_JOIN;
|
||||
else if (c == 301) asm_opt->flag |= HA_F_VERBOSE_GFA;
|
||||
else if (c == 302) asm_opt->flag |= HA_F_WRITE_PAF;
|
||||
else if (c == 303) asm_opt->flag |= HA_F_WRITE_EC;
|
||||
else if (c == 304) asm_opt->flag |= HA_F_SKIP_TRIOBIN;
|
||||
else if (c == 305) asm_opt->max_ov_diff_ec = atof(opt.arg);
|
||||
else if (c == 306) asm_opt->max_ov_diff_final = atof(opt.arg);
|
||||
else if (c == 307) asm_opt->extract_list = opt.arg;
|
||||
else if (c == 308) asm_opt->extract_iter = atoi(opt.arg);
|
||||
else if (c == 309) asm_opt->hom_global_coverage = atoi(opt.arg);
|
||||
else if (c == 310)
|
||||
{
|
||||
char* s = NULL;
|
||||
asm_opt->recover_atg_cov_min = strtol(opt.arg, &s, 10);
|
||||
if (*s == ',') asm_opt->recover_atg_cov_max = strtol(s + 1, &s, 10);
|
||||
if(asm_opt->recover_atg_cov_min == -1 || asm_opt->recover_atg_cov_max == -1)
|
||||
{
|
||||
asm_opt->recover_atg_cov_min = asm_opt->recover_atg_cov_max = -1;
|
||||
}
|
||||
}
|
||||
else if (c == 311) asm_opt->flag |= HA_F_HIGH_HET;
|
||||
else if (c == 'l')
|
||||
{ ///0: disable purge_dup; 1: purge containment; 2: purge overlap
|
||||
asm_opt->purge_level_primary = asm_opt->purge_level_trio = atoi(opt.arg);
|
||||
}
|
||||
else if (c == 's') asm_opt->purge_simi_rate = atof(opt.arg);
|
||||
else if (c == 'O') asm_opt->purge_overlap_len = atoll(opt.arg);
|
||||
else if (c == ':')
|
||||
{
|
||||
fprintf(stderr, "[ERROR] missing option argument in \"%s\"\n", argv[opt.i - 1]);
|
||||
return 0;
|
||||
return 1;
|
||||
}
|
||||
else if (c == '?')
|
||||
{
|
||||
fprintf(stderr, "[ERROR] unknown option in \"%s\"\n", argv[opt.i - 1]);
|
||||
return 0;
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
if (argc == 1)
|
||||
|
||||
if (argc == opt.ind)
|
||||
{
|
||||
Print_H(asm_opt);
|
||||
return 0;
|
||||
}
|
||||
///fprintf(stderr, "max_ov_diff_ec: %f, max_ov_diff_final: %f\n", asm_opt->max_ov_diff_ec, asm_opt->max_ov_diff_final);
|
||||
|
||||
get_queries(argc, argv, &opt, asm_opt);
|
||||
|
||||
|
||||
@@ -3,15 +3,44 @@
|
||||
|
||||
#include <pthread.h>
|
||||
|
||||
#define HA_VERSION "0.9-r297"
|
||||
|
||||
#define VERBOSE 0
|
||||
|
||||
#define HA_F_NO_HPC 0x1
|
||||
#define HA_F_NO_KMER_FLT 0x2
|
||||
#define HA_F_VERBOSE_GFA 0x4
|
||||
#define HA_F_WRITE_EC 0x8
|
||||
#define HA_F_WRITE_PAF 0x10
|
||||
#define HA_F_SKIP_TRIOBIN 0x20
|
||||
#define HA_F_PURGE_CONTAIN 0x40
|
||||
#define HA_F_PURGE_JOIN 0x80
|
||||
#define HA_F_BAN_POST_JOIN 0x100
|
||||
#define HA_F_BAN_ASSEMBLY 0x200
|
||||
#define HA_F_HIGH_HET 0x400
|
||||
|
||||
#define HA_MIN_OV_DIFF 0.02 // min sequence divergence in an overlap
|
||||
|
||||
typedef struct {
|
||||
int flag;
|
||||
int num_reads;
|
||||
char** read_file_names;
|
||||
char* output_file_name;
|
||||
char* required_read_name;
|
||||
char *fn_bin_yak[2];
|
||||
char *fn_bin_list[2];
|
||||
char *extract_list;
|
||||
int extract_iter;
|
||||
int thread_num;
|
||||
int k_mer_length;
|
||||
int mz_win;
|
||||
int bf_shift;
|
||||
double high_factor; // coverage cutoff set to high_factor*hom_cov
|
||||
double max_ov_diff_ec;
|
||||
double max_ov_diff_final;
|
||||
int hom_cov;
|
||||
int het_cov;
|
||||
int max_n_chain; // fall-back max number of chains to consider
|
||||
int k_mer_min_freq;
|
||||
int k_mer_max_freq;
|
||||
int load_index_from_disk;
|
||||
@@ -19,31 +48,48 @@ typedef struct {
|
||||
int number_of_round;
|
||||
int adapterLen;
|
||||
int clean_round;
|
||||
int complete_threads;
|
||||
int roundID;
|
||||
int max_hang_Len;
|
||||
int gap_fuzz;
|
||||
int min_overlap_Len;
|
||||
int min_overlap_coverage;
|
||||
int max_short_tip;
|
||||
int min_cnt;
|
||||
int mid_cnt;
|
||||
int purge_level_primary;
|
||||
int purge_level_trio;
|
||||
int purge_overlap_len;
|
||||
int recover_atg_cov_min;
|
||||
int recover_atg_cov_max;
|
||||
int hom_global_coverage;
|
||||
|
||||
float max_hang_rate;
|
||||
float min_drop_rate;
|
||||
float max_drop_rate;
|
||||
float purge_simi_rate;
|
||||
|
||||
long long small_pop_bubble_size;
|
||||
long long large_pop_bubble_size;
|
||||
long long num_bases;
|
||||
long long num_corrected_bases;
|
||||
long long num_recorrected_bases;
|
||||
long long mem_buf;
|
||||
long long coverage;
|
||||
|
||||
} hifiasm_opt_t;
|
||||
|
||||
extern hifiasm_opt_t asm_opt;
|
||||
|
||||
void init_opt(hifiasm_opt_t* asm_opt);
|
||||
void destory_opt(hifiasm_opt_t* asm_opt);
|
||||
void clear_opt(hifiasm_opt_t* asm_opt, int last_round);
|
||||
int CommandLine_process (int argc, char *argv[], hifiasm_opt_t* asm_opt);
|
||||
void ha_opt_reset_to_round(hifiasm_opt_t* asm_opt, int round);
|
||||
void ha_opt_update_cov(hifiasm_opt_t *opt, int hom_cov);
|
||||
int CommandLine_process(int argc, char *argv[], hifiasm_opt_t* asm_opt);
|
||||
double Get_T(void);
|
||||
|
||||
#endif
|
||||
static inline int ha_opt_triobin(const hifiasm_opt_t *opt)
|
||||
{
|
||||
return ((opt->fn_bin_yak[0] && opt->fn_bin_yak[1]) || (opt->fn_bin_list[0] && opt->fn_bin_list[1]));
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
907
Correct.cpp
907
Correct.cpp
File diff suppressed because it is too large
Load Diff
19
Correct.h
19
Correct.h
@@ -5,6 +5,7 @@
|
||||
#include "Levenshtein_distance.h"
|
||||
#include "POA.h"
|
||||
#include "Process_Read.h"
|
||||
#include "Correct.h"
|
||||
|
||||
//#define CORRECT_THRESHOLD 0.70
|
||||
#define CORRECT_THRESHOLD 0.60
|
||||
@@ -16,6 +17,7 @@
|
||||
#define INSERTION 2
|
||||
#define DELETION 3
|
||||
|
||||
#define WINDOW_MAX_SIZE (WINDOW + (int)(1.0 / HA_MIN_OV_DIFF) + 3) // TODO: why 1/max_ov_diff?
|
||||
|
||||
///#define FLAG_THRE 0
|
||||
|
||||
@@ -1160,10 +1162,17 @@ void init_Cigar_record_alloc(Cigar_record_alloc* x);
|
||||
void resize_Cigar_record_alloc(Cigar_record_alloc* x, long long new_size);
|
||||
void destory_Cigar_record_alloc(Cigar_record_alloc* x);
|
||||
|
||||
void afine_gap_alignment(const char *tseq, uint8_t* tnum, const int tl,
|
||||
const char *qseq, uint8_t* qnum, const int ql, const uint8_t *c2n, const int strand,
|
||||
void afine_gap_alignment(const char *qseq, uint8_t* qnum, const int ql,
|
||||
const char *tseq, uint8_t* tnum, const int tl, const uint8_t *c2n, const int strand,
|
||||
int sc_mch, int sc_mis, int gapo, int gape, int bandLen, int zdrop, int end_bonus,
|
||||
long long* max_t_pos, long long* max_q_pos, long long* score, long long* droped);
|
||||
long long* max_q_pos, long long* max_t_pos, long long* global_score,
|
||||
long long* extention_score, long long* q_boundary_score, long long* q_boundary_t_coordinate,
|
||||
long long* t_boundary_score, long long* t_boundary_q_coordinate,
|
||||
long long* droped, int mode);
|
||||
void correct_overlap_high_het(overlap_region_alloc* overlap_list, All_reads* R_INF,
|
||||
UC_Read* g_read, Correct_dumy* dumy, UC_Read* overlap_read);
|
||||
long long get_affine_gap_score(overlap_region* ovc, UC_Read* g_read, UC_Read* overlap_read, uint8_t* x_num,
|
||||
uint8_t* y_num, uint64_t EstimateXOlen, uint64_t EstimateYOlen);
|
||||
|
||||
#define FORWARD_KSW 0
|
||||
#define BACKWARD_KSW 1
|
||||
@@ -1172,5 +1181,5 @@ long long* max_t_pos, long long* max_q_pos, long long* score, long long* droped)
|
||||
#define GAP_OPEN_KSW 4
|
||||
#define GAP_EXT_KSW 2
|
||||
#define Z_DROP_KSW 400
|
||||
#define BAND_KSW 50
|
||||
#endif
|
||||
#define BAND_KSW 500
|
||||
#endif
|
||||
|
||||
1209
Hash_Table.cpp
1209
Hash_Table.cpp
File diff suppressed because it is too large
Load Diff
453
Hash_Table.h
453
Hash_Table.h
@@ -1,34 +1,20 @@
|
||||
#ifndef __HASHTABLE__
|
||||
#define __HASHTABLE__
|
||||
#include "khash.h"
|
||||
#include "kmer.h"
|
||||
|
||||
KHASH_MAP_INIT_INT64(COUNT64, int)
|
||||
typedef khash_t(COUNT64) Count_Table;
|
||||
|
||||
KHASH_MAP_INIT_INT64(POS64, uint64_t)
|
||||
typedef khash_t(POS64) Pos_Table;
|
||||
#include "htab.h"
|
||||
|
||||
#define PREFIX_BITS 16
|
||||
#define MAX_SUFFIX_BITS 64
|
||||
#define MODE_VALUE 101
|
||||
|
||||
///#define WINDOW 350
|
||||
///#define THRESHOLD 14
|
||||
|
||||
#define WINDOW 375
|
||||
//#define WINDOW_BOUNDARY 150
|
||||
#define WINDOW_BOUNDARY 375
|
||||
///for one side, the first or last WINDOW_UNCORRECT_SINGLE_SIDE_BOUNDARY bases should not be corrected
|
||||
#define WINDOW_UNCORRECT_SINGLE_SIDE_BOUNDARY 25
|
||||
#define THRESHOLD 15
|
||||
#define THRESHOLD_RATE 0.04
|
||||
#define TAIL_LENGTH int(1/THRESHOLD_RATE)
|
||||
///#define OVERLAP_THRESHOLD 0.9
|
||||
#define OVERLAP_THRESHOLD_FILTER 0.9
|
||||
#define WINDOW_MAX_SIZE WINDOW + TAIL_LENGTH + 3
|
||||
#define HIGH_HET_OVERLAP_THRESHOLD_FILTER 0.3
|
||||
#define HIGH_HET_ERROR_RATE 0.08
|
||||
#define THRESHOLD_MAX_SIZE 31
|
||||
#define FINAL_OVERLAP_ERROR_RATE 0.03
|
||||
|
||||
#define GROUP_SIZE 4
|
||||
///the max cigar likes 10M10D10M10D10M
|
||||
@@ -37,26 +23,8 @@ typedef khash_t(POS64) Pos_Table;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
volatile int lock;
|
||||
|
||||
}Hash_table_spin_lock;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
Count_Table** sub_h;
|
||||
Hash_table_spin_lock* sub_h_lock;
|
||||
int prefix_bits;
|
||||
int suffix_bits;
|
||||
///number of subtable
|
||||
int size;
|
||||
uint64_t suffix_mode;
|
||||
uint64_t non_unique_k_mer;
|
||||
} Total_Count_Table;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
uint64_t offset;
|
||||
uint64_t readID;
|
||||
uint32_t offset;
|
||||
uint32_t readID:31, rev:1;
|
||||
} k_mer_pos;
|
||||
|
||||
typedef struct
|
||||
@@ -70,17 +38,9 @@ typedef struct
|
||||
|
||||
typedef struct
|
||||
{
|
||||
k_mer_pos_list* list;
|
||||
uint64_t size;
|
||||
uint64_t length;
|
||||
} k_mer_pos_list_alloc;
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
int C_L[CIGAR_MAX_LENGTH];
|
||||
char C_C[CIGAR_MAX_LENGTH];
|
||||
int length;
|
||||
int C_L[CIGAR_MAX_LENGTH];
|
||||
char C_C[CIGAR_MAX_LENGTH];
|
||||
int length;
|
||||
} CIGAR;
|
||||
|
||||
typedef struct
|
||||
@@ -97,407 +57,96 @@ typedef struct
|
||||
CIGAR cigar;
|
||||
} window_list;
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
window_list* buffer;
|
||||
long long length;
|
||||
long long size;
|
||||
}window_list_alloc;
|
||||
|
||||
int32_t length;
|
||||
int32_t size;
|
||||
} window_list_alloc;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
uint64_t* buffer;
|
||||
uint64_t length;
|
||||
uint64_t size;
|
||||
}Fake_Cigar;
|
||||
uint32_t length;
|
||||
uint32_t size;
|
||||
} Fake_Cigar;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
uint64_t x_id;
|
||||
uint32_t x_id;
|
||||
///the begining and end of the whole overlap
|
||||
uint64_t x_pos_s;
|
||||
uint64_t x_pos_e;
|
||||
uint64_t x_pos_strand;
|
||||
uint32_t x_pos_s;
|
||||
uint32_t x_pos_e;
|
||||
uint32_t x_pos_strand;
|
||||
|
||||
uint64_t y_id;
|
||||
uint64_t y_pos_s;
|
||||
uint64_t y_pos_e;
|
||||
uint64_t y_pos_strand;
|
||||
uint32_t y_id;
|
||||
uint32_t y_pos_s;
|
||||
uint32_t y_pos_e;
|
||||
uint32_t y_pos_strand;
|
||||
|
||||
uint64_t overlapLen;
|
||||
uint64_t shared_seed;
|
||||
uint64_t align_length;
|
||||
///uint64_t total_errors;
|
||||
uint32_t overlapLen;
|
||||
int32_t shared_seed;
|
||||
uint32_t align_length;
|
||||
uint8_t is_match;
|
||||
uint8_t without_large_indel;
|
||||
uint64_t non_homopolymer_errors;
|
||||
int8_t strong;
|
||||
uint32_t non_homopolymer_errors;
|
||||
|
||||
window_list* w_list;
|
||||
uint64_t w_list_size;
|
||||
uint64_t w_list_length;
|
||||
int8_t strong;
|
||||
uint32_t w_list_size;
|
||||
uint32_t w_list_length;
|
||||
Fake_Cigar f_cigar;
|
||||
|
||||
window_list_alloc boundary_cigars;
|
||||
} overlap_region;
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
overlap_region* list;
|
||||
uint64_t size;
|
||||
uint64_t length;
|
||||
///uint64_t mapped_overlaps_length;
|
||||
long long mapped_overlaps_length;
|
||||
int64_t mapped_overlaps_length;
|
||||
} overlap_region_alloc;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
///uint64_t offset;
|
||||
long long offset;
|
||||
///uint64_t self_offset;
|
||||
long long self_offset;
|
||||
uint64_t readID;
|
||||
uint8_t strand;
|
||||
uint32_t readID:30, strand:1, good:1;
|
||||
uint32_t offset, self_offset;
|
||||
} k_mer_hit;
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
k_mer_hit node;
|
||||
uint64_t ID;
|
||||
} ElemType;
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
ElemType* heap;
|
||||
uint64_t* index_i;
|
||||
int len;
|
||||
int MaxSize;
|
||||
} HeapSq;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
long long* score;
|
||||
long long* pre;
|
||||
long long* indels;
|
||||
long long* self_length;
|
||||
long long length;
|
||||
long long size;
|
||||
typedef struct {
|
||||
int32_t *score;
|
||||
int64_t *pre;
|
||||
int32_t *indels;
|
||||
int32_t *self_length;
|
||||
int64_t *tmp; // MUST BE 64-bit integer
|
||||
int64_t length;
|
||||
int64_t size;
|
||||
} Chain_Data;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
k_mer_hit* list;
|
||||
k_mer_hit* tmp;
|
||||
long long length;
|
||||
long long size;
|
||||
uint64_t foward_pos;
|
||||
uint64_t rc_pos;
|
||||
Chain_Data chainDP;
|
||||
} Candidates_list;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
Pos_Table** sub_h;
|
||||
Hash_table_spin_lock* sub_h_lock;
|
||||
int prefix_bits;
|
||||
int suffix_bits;
|
||||
///number of subtable
|
||||
int size;
|
||||
uint64_t suffix_mode;
|
||||
k_mer_pos* pos;
|
||||
uint64_t useful_k_mer;
|
||||
uint64_t total_occ;
|
||||
uint64_t* k_mer_index;
|
||||
} Total_Pos_Table;
|
||||
|
||||
|
||||
|
||||
|
||||
inline uint64_t mod_d(uint64_t h_key, uint64_t low_key, uint64_t d)
|
||||
{
|
||||
uint64_t result = (h_key >> 32) % d;
|
||||
result = ((result << 32) + (h_key & (uint64_t)0xffffffff)) % d;
|
||||
result = ((result << 32) + (low_key >> 32)) % d;
|
||||
result = ((result << 32) + (low_key & (uint64_t)0xffffffff)) % d;
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
|
||||
|
||||
////suffix_bits = 64 in default
|
||||
inline int recover_hash_code(uint64_t sub_ID, uint64_t sub_key, Hash_code* code,
|
||||
uint64_t suffix_mode, int suffix_bits, int k)
|
||||
{
|
||||
uint64_t h_key, low_key;
|
||||
h_key = low_key = 0;
|
||||
|
||||
low_key = sub_ID << SAFE_SHIFT(suffix_bits);
|
||||
low_key = low_key | sub_key;
|
||||
|
||||
h_key = sub_ID >> (64 - suffix_bits);
|
||||
|
||||
code->x[0] = code->x[1] = 0;
|
||||
uint64_t mask = ALL >> (64 - k);
|
||||
code->x[0] = low_key & mask;
|
||||
|
||||
code->x[1] = h_key << (64 - k);
|
||||
code->x[1] = code->x[1] | (low_key >> SAFE_SHIFT(k));
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
///inline int get_sub_table(uint64_t* get_sub_ID, uint64_t* get_sub_key, Total_Count_Table* TCB, Hash_code* code, int k)
|
||||
inline int get_sub_table(uint64_t* get_sub_ID, uint64_t* get_sub_key, uint64_t suffix_mode, int suffix_bits,
|
||||
Hash_code* code, int k)
|
||||
{
|
||||
uint64_t h_key, low_key;
|
||||
///k might be 64,so it is unsafe
|
||||
///low_key = code->x[0] | (code->x[1] << k);
|
||||
low_key = code->x[0] | (code->x[1] << SAFE_SHIFT(k));
|
||||
//k cannot be 0, so this shift is safe
|
||||
h_key = code->x[1] >> (64 - k);
|
||||
|
||||
if(mod_d(h_key, low_key, MODE_VALUE) > 3)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint64_t sub_ID = (low_key >> SAFE_SHIFT(suffix_bits)) | (h_key << (64 - suffix_bits));
|
||||
uint64_t sub_key = (low_key & suffix_mode);
|
||||
|
||||
*get_sub_ID = sub_ID;
|
||||
*get_sub_key = sub_key;
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
inline int insert_Total_Count_Table(Total_Count_Table* TCB, Hash_code* code, int k)
|
||||
{
|
||||
uint64_t sub_ID, sub_key;
|
||||
if(!get_sub_table(&sub_ID, &sub_key, TCB->suffix_mode, TCB->suffix_bits, code, k))
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
khint_t t;
|
||||
int absent;
|
||||
|
||||
|
||||
while (__sync_lock_test_and_set(&TCB->sub_h_lock[sub_ID].lock, 1))
|
||||
{
|
||||
while (TCB->sub_h_lock[sub_ID].lock);
|
||||
}
|
||||
|
||||
t = kh_put(COUNT64, TCB->sub_h[sub_ID], sub_key, &absent);
|
||||
if (absent)
|
||||
{
|
||||
kh_value(TCB->sub_h[sub_ID], t) = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
//kh_value(TCB->sub_h[sub_ID], t) = kh_value(TCB->sub_h[sub_ID], t) + 1;
|
||||
kh_value(TCB->sub_h[sub_ID], t)++;
|
||||
}
|
||||
|
||||
__sync_lock_release(&TCB->sub_h_lock[sub_ID].lock);
|
||||
|
||||
return 1;
|
||||
}
|
||||
|
||||
inline int get_Total_Count_Table(Total_Count_Table* TCB, Hash_code* code, int k)
|
||||
{
|
||||
|
||||
uint64_t sub_ID, sub_key;
|
||||
if(!get_sub_table(&sub_ID, &sub_key, TCB->suffix_mode, TCB->suffix_bits, code, k))
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
khint_t t;
|
||||
|
||||
///query hash table,key is k
|
||||
t = kh_get(COUNT64, TCB->sub_h[sub_ID], sub_key);
|
||||
|
||||
if (t != kh_end(TCB->sub_h[sub_ID]))
|
||||
{
|
||||
return kh_value(TCB->sub_h[sub_ID], t);
|
||||
}
|
||||
else
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
inline uint64_t get_Total_Pos_Table(Total_Pos_Table* PCB, Hash_code* code, int k, uint64_t* r_sub_ID)
|
||||
{
|
||||
|
||||
uint64_t sub_ID, sub_key;
|
||||
if(!get_sub_table(&sub_ID, &sub_key, PCB->suffix_mode, PCB->suffix_bits, code, k))
|
||||
{
|
||||
return (uint64_t)-1;
|
||||
}
|
||||
|
||||
khint_t t;
|
||||
|
||||
///query hash table,key is k
|
||||
t = kh_get(POS64, PCB->sub_h[sub_ID], sub_key);
|
||||
|
||||
if (t != kh_end(PCB->sub_h[sub_ID]))
|
||||
{
|
||||
*r_sub_ID = sub_ID;
|
||||
return kh_value(PCB->sub_h[sub_ID], t);
|
||||
}
|
||||
else
|
||||
{
|
||||
return (uint64_t)-1;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
inline uint64_t count_Total_Pos_Table(Total_Pos_Table* PCB, Hash_code* code, int k)
|
||||
{
|
||||
uint64_t sub_ID;
|
||||
uint64_t ret = get_Total_Pos_Table(PCB, code, k, &sub_ID);
|
||||
if(ret != (uint64_t)-1)
|
||||
{
|
||||
return PCB->k_mer_index[ret + 1] - PCB->k_mer_index[ret];
|
||||
}
|
||||
else
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
inline uint64_t locate_Total_Pos_Table(Total_Pos_Table* PCB, Hash_code* code, k_mer_pos** list, int k, uint64_t* r_sub_ID)
|
||||
{
|
||||
uint64_t ret = get_Total_Pos_Table(PCB, code, k, r_sub_ID);
|
||||
if(ret != (uint64_t)-1)
|
||||
{
|
||||
*list = PCB->k_mer_index[ret] + PCB->pos;
|
||||
return PCB->k_mer_index[ret + 1] - PCB->k_mer_index[ret];
|
||||
}
|
||||
else
|
||||
{
|
||||
*list = NULL;
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
int cmp_k_mer_pos(const void * a, const void * b);
|
||||
|
||||
|
||||
inline uint64_t insert_Total_Pos_Table(Total_Pos_Table* PCB, Hash_code* code, int k, uint64_t readID, uint64_t pos)
|
||||
{
|
||||
k_mer_pos* list;
|
||||
int flag = 0;
|
||||
uint64_t sub_ID;
|
||||
uint64_t occ = locate_Total_Pos_Table(PCB, code, &list, k, &sub_ID);
|
||||
|
||||
if (occ)
|
||||
{
|
||||
|
||||
while (__sync_lock_test_and_set(&PCB->sub_h_lock[sub_ID].lock, 1))
|
||||
{
|
||||
while (PCB->sub_h_lock[sub_ID].lock);
|
||||
}
|
||||
|
||||
if (list[0].offset + 1 < occ)
|
||||
{
|
||||
list[0].offset++;
|
||||
list[list[0].offset].readID = readID;
|
||||
///list[list[0].offset].readID = readID|direction;
|
||||
list[list[0].offset].offset = pos;
|
||||
}
|
||||
else
|
||||
{
|
||||
list[0].readID = readID;
|
||||
///list[0].readID = readID|direction;
|
||||
list[0].offset = pos;
|
||||
flag = 1;
|
||||
}
|
||||
|
||||
__sync_lock_release(&PCB->sub_h_lock[sub_ID].lock);
|
||||
|
||||
//if all pos has been saved, it is safe to sort
|
||||
if (flag && occ>1)
|
||||
{
|
||||
qsort(list, occ, sizeof(k_mer_pos), cmp_k_mer_pos);
|
||||
}
|
||||
|
||||
|
||||
return 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
|
||||
void init_Total_Count_Table(int k, Total_Count_Table* TCB);
|
||||
void init_Total_Pos_Table(Total_Pos_Table* TCB, Total_Count_Table* pre_TCB);
|
||||
void destory_Total_Count_Table(Total_Count_Table* TCB);
|
||||
|
||||
void init_Count_Table(Count_Table** table);
|
||||
void init_Pos_Table(Count_Table** pre_table, Pos_Table** table);
|
||||
void destory_Total_Pos_Table(Total_Pos_Table* TCB);
|
||||
void write_Total_Pos_Table(Total_Pos_Table* TCB, char* read_file_name);
|
||||
int load_Total_Pos_Table(Total_Pos_Table* TCB, char* read_file_name);
|
||||
|
||||
|
||||
|
||||
void Traverse_Counting_Table(Total_Count_Table* TCB, Total_Pos_Table* PCB, int k_mer_min_freq, int k_mer_max_freq);
|
||||
|
||||
void init_Candidates_list(Candidates_list* l);
|
||||
void clear_Candidates_list(Candidates_list* l);
|
||||
void destory_Candidates_list(Candidates_list* l);
|
||||
|
||||
|
||||
void init_k_mer_pos_list_alloc(k_mer_pos_list_alloc* list);
|
||||
void destory_k_mer_pos_list_alloc(k_mer_pos_list_alloc* list);
|
||||
void clear_k_mer_pos_list_alloc(k_mer_pos_list_alloc* list);
|
||||
void append_k_mer_pos_list_alloc(k_mer_pos_list_alloc* list, k_mer_pos* n_list, uint64_t n_length,
|
||||
uint64_t n_end_pos, uint8_t n_direction);
|
||||
|
||||
|
||||
|
||||
void merge_k_mer_pos_list_alloc_heap_sort(k_mer_pos_list_alloc* list, Candidates_list* candidates, HeapSq* HBT);
|
||||
|
||||
void Init_Heap(HeapSq* HBT);
|
||||
void destory_Heap(HeapSq* HBT);
|
||||
void clear_Heap(HeapSq* HBT);
|
||||
|
||||
void init_overlap_region_alloc(overlap_region_alloc* list);
|
||||
void clear_overlap_region_alloc(overlap_region_alloc* list);
|
||||
void destory_overlap_region_alloc(overlap_region_alloc* list);
|
||||
void append_window_list(overlap_region* region, uint64_t x_start, uint64_t x_end, int y_start, int y_end, int error,
|
||||
int extra_begin, int extra_end, int error_threshold);
|
||||
|
||||
|
||||
|
||||
void overlap_region_sort_y_id(overlap_region *a, long long n);
|
||||
|
||||
|
||||
void calculate_overlap_region_by_chaining(Candidates_list* candidates, overlap_region_alloc* overlap_list,
|
||||
uint64_t readID, uint64_t readLength, All_reads* R_INF, double band_width_threshold, int add_beg_end);
|
||||
|
||||
|
||||
|
||||
void init_fake_cigar(Fake_Cigar* x);
|
||||
void destory_fake_cigar(Fake_Cigar* x);
|
||||
void clear_fake_cigar(Fake_Cigar* x);
|
||||
@@ -505,14 +154,14 @@ void add_fake_cigar(Fake_Cigar* x, uint32_t gap_site, int32_t gap_shift);
|
||||
void resize_fake_cigar(Fake_Cigar* x, uint64_t size);
|
||||
int get_fake_gap_pos(Fake_Cigar* x, int index);
|
||||
int get_fake_gap_shift(Fake_Cigar* x, int index);
|
||||
inline long long y_start_offset(long long x_start, Fake_Cigar* o)
|
||||
|
||||
static inline long long y_start_offset(long long x_start, Fake_Cigar* o)
|
||||
{
|
||||
if(x_start == get_fake_gap_pos(o, o->length - 1))
|
||||
{
|
||||
return get_fake_gap_shift(o, o->length - 1);
|
||||
}
|
||||
|
||||
|
||||
long long i;
|
||||
for (i = 0; i < (long long)o->length; i++)
|
||||
{
|
||||
@@ -524,7 +173,7 @@ inline long long y_start_offset(long long x_start, Fake_Cigar* o)
|
||||
|
||||
if(i == 0 || i == (long long)o->length)
|
||||
{
|
||||
fprintf(stderr, "ERROR\n");
|
||||
fprintf(stderr, "ERROR at %s:%d\n", __FILE__, __LINE__);
|
||||
exit(0);
|
||||
}
|
||||
|
||||
@@ -532,23 +181,11 @@ inline long long y_start_offset(long long x_start, Fake_Cigar* o)
|
||||
return get_fake_gap_shift(o, i - 1);
|
||||
}
|
||||
|
||||
inline void print_fake_gap(Fake_Cigar* o)
|
||||
{
|
||||
long long i;
|
||||
for (i = 0; i < (long long)o->length; i++)
|
||||
{
|
||||
fprintf(stderr, "**i: %lld, gap_pos_in_x: %d, gap_shift: %d\n",
|
||||
i, get_fake_gap_pos(o, i),
|
||||
get_fake_gap_shift(o, i));
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
void resize_Chain_Data(Chain_Data* x, long long size);
|
||||
void init_window_list_alloc(window_list_alloc* x);
|
||||
void clear_window_list_alloc(window_list_alloc* x);
|
||||
void destory_window_list_alloc(window_list_alloc* x);
|
||||
void resize_window_list_alloc(window_list_alloc* x, long long size);
|
||||
void chain_DP(k_mer_hit* a, long long a_n, Chain_Data* dp, overlap_region* result, double band_width_threshold, int max_skip, int x_readLen, int y_readLen);
|
||||
|
||||
|
||||
#endif
|
||||
#endif
|
||||
|
||||
56
Makefile
56
Makefile
@@ -1,9 +1,10 @@
|
||||
CXX= g++
|
||||
CXXFLAGS= -g -O3 -msse4.2 -mpopcnt -fomit-frame-pointer -Wall #-Winline
|
||||
CXXFLAGS= -g -O3 -msse4.2 -mpopcnt -fomit-frame-pointer -Wall
|
||||
CPPFLAGS=
|
||||
INCLUDES=
|
||||
OBJS= Output.o CommandLines.o Process_Read.o Assembly.o kmer.o Hash_Table.o \
|
||||
POA.o Correct.o Levenshtein_distance.o Overlaps.o #ksw2_extz2_sse.o
|
||||
OBJS= CommandLines.o Process_Read.o Assembly.o Hash_Table.o \
|
||||
POA.o Correct.o Levenshtein_distance.o Overlaps.o Trio.o kthread.o Purge_Dups.o \
|
||||
htab.o hist.o sketch.o anchor.o extract.o sys.o ksw2_extz2_sse.o
|
||||
EXE= hifiasm
|
||||
LIBS= -lz -lpthread -lm
|
||||
|
||||
@@ -31,24 +32,37 @@ depend:
|
||||
|
||||
# DO NOT DELETE
|
||||
|
||||
Assembly.o: Assembly.h Process_Read.h kseq.h Overlaps.h kvec.h kdq.h
|
||||
Assembly.o: CommandLines.h kmer.h Hash_Table.h khash.h POA.h Correct.h
|
||||
Assembly.o: Levenshtein_distance.h Output.h
|
||||
Assembly.o: Assembly.h CommandLines.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||
Assembly.o: Hash_Table.h htab.h POA.h Correct.h Levenshtein_distance.h
|
||||
Assembly.o: kthread.h
|
||||
CommandLines.o: CommandLines.h ketopt.h
|
||||
Correct.o: Correct.h Hash_Table.h khash.h kmer.h Process_Read.h kseq.h
|
||||
Correct.o: Overlaps.h kvec.h kdq.h CommandLines.h Levenshtein_distance.h
|
||||
Correct.o: POA.h Assembly.h #ksw2.h
|
||||
Hash_Table.o: Hash_Table.h khash.h kmer.h Process_Read.h kseq.h Overlaps.h
|
||||
Hash_Table.o: kvec.h kdq.h CommandLines.h Correct.h Levenshtein_distance.h
|
||||
Hash_Table.o: POA.h ksort.h
|
||||
Correct.o: Correct.h Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h ksw2.h
|
||||
Correct.o: kdq.h CommandLines.h Levenshtein_distance.h POA.h Assembly.h
|
||||
Hash_Table.o: Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||
Hash_Table.o: CommandLines.h ksort.h
|
||||
Levenshtein_distance.o: Levenshtein_distance.h
|
||||
Output.o: Output.h CommandLines.h
|
||||
Overlaps.o: Overlaps.h kvec.h kdq.h ksort.h Process_Read.h kseq.h
|
||||
Overlaps.o: CommandLines.h
|
||||
POA.o: POA.h Hash_Table.h khash.h kmer.h Process_Read.h kseq.h Overlaps.h
|
||||
POA.o: kvec.h kdq.h CommandLines.h Correct.h Levenshtein_distance.h
|
||||
Process_Read.o: Process_Read.h kseq.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||
kmer.o: kmer.h Process_Read.h kseq.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||
main.o: CommandLines.h Process_Read.h kseq.h Overlaps.h kvec.h kdq.h
|
||||
main.o: Assembly.h Levenshtein_distance.h
|
||||
#ksw2_extz2_sse.o: ksw2.h
|
||||
Overlaps.o: Overlaps.h kvec.h kdq.h ksort.h Process_Read.h CommandLines.h
|
||||
Overlaps.o: Hash_Table.h htab.h Correct.h Levenshtein_distance.h POA.h
|
||||
Overlaps.o: Purge_Dups.h
|
||||
POA.o: POA.h Hash_Table.h htab.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||
POA.o: CommandLines.h Correct.h Levenshtein_distance.h
|
||||
Process_Read.o: Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||
Purge_Dups.o: ksort.h Purge_Dups.h kvec.h kdq.h Overlaps.h Hash_Table.h
|
||||
Purge_Dups.o: htab.h Process_Read.h CommandLines.h Correct.h
|
||||
Purge_Dups.o: Levenshtein_distance.h POA.h kthread.h
|
||||
Trio.o: khashl.h kthread.h kseq.h Process_Read.h Overlaps.h kvec.h kdq.h
|
||||
Trio.o: CommandLines.h htab.h
|
||||
anchor.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||
anchor.o: ksort.h Hash_Table.h
|
||||
extract.o: Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h khashl.h
|
||||
extract.o: kseq.h
|
||||
hist.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||
htab.o: kthread.h khashl.h kseq.h ksort.h htab.h Process_Read.h Overlaps.h
|
||||
htab.o: kvec.h kdq.h CommandLines.h
|
||||
kthread.o: kthread.h
|
||||
main.o: CommandLines.h Process_Read.h Overlaps.h kvec.h kdq.h Assembly.h
|
||||
main.o: Levenshtein_distance.h htab.h
|
||||
sketch.o: kvec.h htab.h Process_Read.h Overlaps.h kdq.h CommandLines.h
|
||||
sys.o: htab.h Process_Read.h Overlaps.h kvec.h kdq.h CommandLines.h
|
||||
ksw2_extz2_sse.o: ksw2.h
|
||||
20797
Overlaps.cpp
20797
Overlaps.cpp
File diff suppressed because it is too large
Load Diff
778
Overlaps.h
778
Overlaps.h
@@ -1,5 +1,6 @@
|
||||
#ifndef __OVERLAPS__
|
||||
#define __OVERLAPS__
|
||||
#include <stdio.h>
|
||||
#include <stdint.h>
|
||||
#include "kvec.h"
|
||||
#include "kdq.h"
|
||||
@@ -17,6 +18,16 @@
|
||||
///#define MAX_BUBBLE_DIST 10000000
|
||||
#define SMALL_BUBBLE_SIZE (uint32_t)-1
|
||||
//#define SMALL_BUBBLE_SIZE 1000
|
||||
#define PRIMARY_LABLE 0
|
||||
#define ALTER_LABLE 1
|
||||
#define HAP_LABLE 2
|
||||
#define TRIO_THRES 0.9
|
||||
#define DOUBLE_CHECK_THRES 0.1
|
||||
#define FINAL_DOUBLE_CHECK_THRES 0.2
|
||||
#define CHIMERIC_TRIM_THRES 4
|
||||
// #define PRIMARY_LABLE 1
|
||||
// #define ALTER_LABLE 2
|
||||
// #define HAP_LABLE 4
|
||||
|
||||
|
||||
#define Get_qn(RECORD) ((uint32_t)((RECORD).qns>>32))
|
||||
@@ -35,6 +46,10 @@
|
||||
#define LONG_TIPS_UNDER_MAX_EXT 6
|
||||
#define LOOP 7
|
||||
|
||||
#define TRIM 10
|
||||
#define CUT 11
|
||||
#define CUT_DIF_HAP 12
|
||||
|
||||
|
||||
///query is the read itself
|
||||
typedef struct {
|
||||
@@ -46,7 +61,6 @@ typedef struct {
|
||||
uint8_t no_l_indel;
|
||||
} ma_hit_t;
|
||||
|
||||
|
||||
typedef struct {
|
||||
ma_hit_t* buffer;
|
||||
uint32_t size;
|
||||
@@ -58,7 +72,7 @@ typedef struct {
|
||||
|
||||
void init_ma_hit_t_alloc(ma_hit_t_alloc* x);
|
||||
void clear_ma_hit_t_alloc(ma_hit_t_alloc* x);
|
||||
void resize_ma_hit_t_alloc(ma_hit_t_alloc* x, uint64_t size);
|
||||
void resize_ma_hit_t_alloc(ma_hit_t_alloc* x, uint32_t size);
|
||||
void destory_ma_hit_t_alloc(ma_hit_t_alloc* x);
|
||||
void add_ma_hit_t_alloc(ma_hit_t_alloc* x, ma_hit_t* element);
|
||||
void ma_hit_sort_tn(ma_hit_t *a, long long n);
|
||||
@@ -67,7 +81,6 @@ void ma_hit_sort_qns(ma_hit_t *a, long long n);
|
||||
int load_all_data_from_disk(ma_hit_t_alloc **sources, ma_hit_t_alloc **reverse_sources,
|
||||
char* output_file_name);
|
||||
|
||||
void normalize_ma_hit_t(ma_hit_t_alloc* sources, long long num_sources);
|
||||
|
||||
|
||||
typedef struct {
|
||||
@@ -93,6 +106,17 @@ typedef struct {
|
||||
uint8_t no_l_indel;
|
||||
} asg_arc_t;
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint32_t len:31, circ:1; // len: length of the unitig; circ: circular if non-zero
|
||||
uint32_t start, end; // start: starting vertex in the string graph; end: ending vertex
|
||||
uint32_t m, n; // number of reads
|
||||
uint64_t *a; // list of reads
|
||||
char *s; // unitig sequence is not null
|
||||
} ma_utg_t;
|
||||
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint32_t len:31, del:1;
|
||||
uint8_t c;
|
||||
@@ -102,15 +126,42 @@ typedef struct {
|
||||
uint32_t m_arc, n_arc:31, is_srt:1;
|
||||
asg_arc_t *arc;
|
||||
uint32_t m_seq, n_seq:31, is_symm:1;
|
||||
uint32_t r_seq;
|
||||
|
||||
asg_seq_t *seq;
|
||||
uint64_t *idx;
|
||||
|
||||
uint8_t* seq_vis;
|
||||
|
||||
uint32_t n_F_seq;
|
||||
ma_utg_t* F_seq;
|
||||
} asg_t;
|
||||
|
||||
asg_t *asg_init(void);
|
||||
void asg_destroy(asg_t *g);
|
||||
void asg_arc_sort(asg_t *g);
|
||||
void asg_seq_set(asg_t *g, int sid, int len, int del);
|
||||
void asg_arc_index(asg_t *g);
|
||||
void asg_cleanup(asg_t *g);
|
||||
void asg_symm(asg_t *g);
|
||||
void print_gfa(asg_t *g);
|
||||
|
||||
|
||||
typedef struct { size_t n, m; uint64_t *a; } asg64_v;
|
||||
|
||||
|
||||
typedef struct { size_t n, m; ma_utg_t *a; } ma_utg_v;
|
||||
|
||||
typedef struct {
|
||||
ma_utg_v u;
|
||||
asg_t *g;
|
||||
} ma_ug_t;
|
||||
|
||||
typedef struct {
|
||||
uint32_t utg:31, ori:1, start, len;
|
||||
} utg_intv_t;
|
||||
|
||||
|
||||
#define MA_HT_INT (-1)
|
||||
#define MA_HT_QCONT (-2)
|
||||
#define MA_HT_TCONT (-3)
|
||||
@@ -201,6 +252,23 @@ static inline int ma_hit2arc(const ma_hit_t *h, int ql, int tl, int max_hang, fl
|
||||
#define asg_arc_n(g, v) ((uint32_t)(g)->idx[(v)])
|
||||
#define asg_arc_a(g, v) (&(g)->arc[(g)->idx[(v)]>>32])
|
||||
|
||||
static inline uint32_t asg_get_arc(asg_t *g, uint32_t v, uint32_t w, asg_arc_t* t)
|
||||
{
|
||||
uint32_t i, nv = asg_arc_n(g, v);
|
||||
asg_arc_t *av = asg_arc_a(g, v);
|
||||
for (i = 0; i < nv; ++i)
|
||||
{
|
||||
if(av[i].del) continue;
|
||||
if(av[i].v == w)
|
||||
{
|
||||
(*t) = av[i];
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
// append an arc
|
||||
static inline asg_arc_t *asg_arc_pushp(asg_t *g)
|
||||
{
|
||||
@@ -239,7 +307,7 @@ static inline void asg_seq_del(asg_t *g, uint32_t s)
|
||||
static inline void asg_seq_drop(asg_t *g, uint32_t s)
|
||||
{
|
||||
///s is not at primary
|
||||
if(g->seq[s].c)
|
||||
if(g->seq[s].c == ALTER_LABLE)
|
||||
{
|
||||
uint32_t k;
|
||||
for (k = 0; k < 2; ++k)
|
||||
@@ -252,8 +320,11 @@ static inline void asg_seq_drop(asg_t *g, uint32_t s)
|
||||
{
|
||||
if(av[i].del) continue;
|
||||
///if output node is at primary
|
||||
if(g->seq[(av[i].v>>1)].c == 0)
|
||||
{
|
||||
/****************************may have hap bugs********************************/
|
||||
///if(g->seq[(av[i].v>>1)].c == PRIMARY_LABLE)
|
||||
///if(g->seq[(av[i].v>>1)].c == PRIMARY_LABLE || g->seq[(av[i].v>>1)].c == HAP_LABLE)
|
||||
if(g->seq[(av[i].v>>1)].c != ALTER_LABLE)
|
||||
{/****************************may have hap bugs********************************/
|
||||
av[i].del = 1;
|
||||
asg_arc_del(g, av[i].v^1, v^1, 1);
|
||||
}
|
||||
@@ -263,24 +334,7 @@ static inline void asg_seq_drop(asg_t *g, uint32_t s)
|
||||
}
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint32_t len:31, circ:1; // len: length of the unitig; circ: circular if non-zero
|
||||
uint32_t start, end; // start: starting vertex in the string graph; end: ending vertex
|
||||
uint32_t m, n; // number of reads
|
||||
uint64_t *a; // list of reads
|
||||
char *s; // unitig sequence is not null
|
||||
} ma_utg_t;
|
||||
|
||||
typedef struct { size_t n, m; ma_utg_t *a; } ma_utg_v;
|
||||
|
||||
typedef struct {
|
||||
ma_utg_v u;
|
||||
asg_t *g;
|
||||
} ma_ug_t;
|
||||
|
||||
typedef struct {
|
||||
uint32_t utg:31, ori:1, start, len;
|
||||
} utg_intv_t;
|
||||
|
||||
|
||||
/******************
|
||||
@@ -290,7 +344,10 @@ typedef struct {
|
||||
typedef struct {
|
||||
uint32_t p; // the optimal parent vertex
|
||||
uint32_t d; // the shortest distance from the initial vertex
|
||||
uint32_t c; // max count of reads
|
||||
uint32_t c; // max count of positive reads
|
||||
uint32_t m; // max count of negative reads
|
||||
uint32_t np; // max count of non-positive reads
|
||||
uint32_t nc; // max count of reads, no matter positive or negative
|
||||
uint32_t r:31, s:1; // r: the number of remaining incoming arc; s: state
|
||||
//s: state, s=0, this edge has not been visited, otherwise, s=1
|
||||
} binfo_t;
|
||||
@@ -304,6 +361,60 @@ typedef struct {
|
||||
kvec_t(uint32_t) e; // visited edges/arcs
|
||||
} buf_t;
|
||||
|
||||
|
||||
typedef struct {
|
||||
kvec_t(uint64_t) Nodes;
|
||||
kvec_t(uint64_t) Edges;
|
||||
uint32_t pre_n_seq, seqID;
|
||||
} C_graph;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(uint8_t) a;
|
||||
uint32_t i;
|
||||
} kvec_t_u8_warp;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(uint32_t) a;
|
||||
uint32_t i;
|
||||
} kvec_t_u32_warp;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(int32_t) a;
|
||||
uint32_t i;
|
||||
} kvec_t_i32_warp;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(uint64_t) a;
|
||||
uint64_t i;
|
||||
} kvec_t_u64_warp;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(asg_arc_t) a;
|
||||
uint64_t i;
|
||||
}kvec_asg_arc_t_warp;
|
||||
|
||||
void sort_kvec_t_u64_warp(kvec_t_u64_warp* u_vecs, uint32_t is_descend);
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint32_t q_pos;
|
||||
uint32_t t_pos;
|
||||
uint32_t t_id;
|
||||
uint32_t is_color;
|
||||
} Hap_Align;
|
||||
|
||||
typedef struct {
|
||||
kvec_t(Hap_Align) x;
|
||||
uint64_t i;
|
||||
} Hap_Align_warp;
|
||||
|
||||
typedef struct {
|
||||
buf_t* b_0;
|
||||
uint32_t untigI;
|
||||
uint32_t readI;
|
||||
uint32_t offset;
|
||||
} rIdContig;
|
||||
|
||||
// count the number of outgoing arcs, including reduced arcs
|
||||
static inline int count_out_with_del(const asg_t *g, uint32_t v)
|
||||
{
|
||||
@@ -340,5 +451,620 @@ void add_overlaps(ma_hit_t_alloc* source_paf, ma_hit_t_alloc* dest_paf, uint64_t
|
||||
void remove_overlaps(ma_hit_t_alloc* source_paf, uint64_t* source_index, long long listLen);
|
||||
void add_overlaps_from_different_sources(ma_hit_t_alloc* source_paf_list, ma_hit_t_alloc* dest_paf,
|
||||
uint64_t* source_index, long long listLen);
|
||||
void print_revise_edges(ma_hit_t_alloc* source_paf, uint64_t* source_index, long long listLen);
|
||||
#endif
|
||||
|
||||
#define EvaluateLen(U, id) ((U).a[(id)].start)
|
||||
#define IsMerge(U, id) ((U).a[(id)].end)
|
||||
#define kv_reuse(v, rn, rm, r) ((v).n = (rn), (v).m = (rm), (v).a = (r))
|
||||
#define long_tip(U, id, threshold) ((EvaluateLen((U), (id))>=(threshold))&&(!((U).a[(id)].circ)))
|
||||
///there are threee cases:
|
||||
///1. if this untig is too long (>maxShortUntig), it must be not short untig/must be a long untig
|
||||
///2. if this untig is long (>minLongUntig && EvaluateLen(ug->u, av[i].v>>1) > (EvaluateLen(ug->u, v>>1)*l_untig_rate)), it might be a long tip
|
||||
#define check_long_tip(U, id, minLongUntig, maxShortUntig, ShortUntigRate, mainLen) \
|
||||
((!((U).a[(id)].circ)) \
|
||||
&& \
|
||||
((EvaluateLen((U), (id)) > (maxShortUntig))\
|
||||
||\
|
||||
((long_tip((U), (id), (minLongUntig)))\
|
||||
&&\
|
||||
(EvaluateLen((U), (id)) > (ShortUntigRate)*(mainLen)))))
|
||||
#define Get_vis(visit, v, d) (((visit)[(v)>>1])&(((((v)<<(d))&1)+1)))
|
||||
#define Set_vis(visit, v, d) (((visit)[(v)>>1])|=(((((v)<<(d))&1)+1)))
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint64_t len;
|
||||
uint32_t* index;
|
||||
} R_to_U;
|
||||
|
||||
void init_R_to_U(R_to_U* x, uint64_t len);
|
||||
void destory_R_to_U(R_to_U* x);
|
||||
void set_R_to_U(R_to_U* x, uint32_t rID, uint32_t uID, uint32_t is_Unitig);
|
||||
void get_R_to_U(R_to_U* x, uint32_t rID, uint32_t* uID, uint32_t* is_Unitig);
|
||||
void transfor_R_to_U(R_to_U* x);
|
||||
void debug_utg_graph(ma_ug_t *ug, asg_t* read_g, int require_equal_nv, int test_tangle);
|
||||
void clean_untig_graph(ma_ug_t *ug, asg_t *read_g, ma_hit_t_alloc* reverse_sources,
|
||||
long long bubble_dist, long long tipsLen, float tip_drop_ratio, long long stops_threshold,
|
||||
R_to_U* ruIndex, buf_t* b_0, uint8_t* visit, float density, uint32_t miniHapLen,
|
||||
uint32_t miniBiGraph, float chimeric_rate, int is_final_clean);
|
||||
int asg_pop_bubble_primary(asg_t *g, int max_dist);
|
||||
long long asg_arc_del_simple_circle_untig(ma_hit_t_alloc* sources, ma_sub_t* coverage_cut, asg_t *g, long long circleLen, int is_drop);
|
||||
|
||||
typedef struct {
|
||||
asg_t* g;
|
||||
|
||||
asg_arc_t *av;
|
||||
uint32_t nv;
|
||||
uint32_t av_i;
|
||||
|
||||
asg_arc_t* new_edges;
|
||||
uint32_t new_edges_n;
|
||||
uint32_t new_edges_i;
|
||||
} Edge_iter;
|
||||
|
||||
void init_Edge_iter(asg_t* g, uint32_t v, asg_arc_t* new_edges, uint32_t new_edges_n, Edge_iter* x);
|
||||
int get_arc_t(Edge_iter* x, asg_arc_t* get);
|
||||
int asg_pop_bubble_primary_trio(ma_ug_t *ug, int max_dist, uint32_t positive_flag, uint32_t negative_flag);
|
||||
|
||||
|
||||
inline int get_real_length(asg_t *g, uint32_t v, uint32_t* v_s)
|
||||
{
|
||||
uint32_t i, kv = 0;
|
||||
for (i = 0, kv = 0; i < asg_arc_n(g, v); i++)
|
||||
{
|
||||
if(!asg_arc_a(g, v)[i].del)
|
||||
{
|
||||
if(v_s) v_s[kv] = asg_arc_a(g, v)[i].v;
|
||||
kv++;
|
||||
}
|
||||
}
|
||||
|
||||
return kv;
|
||||
}
|
||||
|
||||
inline uint32_t check_tip(asg_t *sg, uint32_t begNode, uint32_t* endNode, buf_t* b, uint32_t max_ext)
|
||||
{
|
||||
///cut tip of length <= max_ext
|
||||
uint32_t v = begNode, w;
|
||||
uint32_t kv;
|
||||
uint32_t eLen = 0;
|
||||
(*endNode) = (uint32_t)-1;
|
||||
b->b.n = 0;
|
||||
while (1)
|
||||
{
|
||||
kv = get_real_length(sg, v, NULL);
|
||||
(*endNode) = v;
|
||||
eLen++;
|
||||
if(b) kv_push(uint32_t, b->b, v);
|
||||
if(kv == 0) return END_TIPS;
|
||||
if(kv > 1) return MUL_OUTPUT;
|
||||
///if(eLen > max_ext) return LONG_TIPS;
|
||||
///kv must be 1 here
|
||||
kv = get_real_length(sg, v, &w);
|
||||
///here this value must be >= 1
|
||||
if(get_real_length(sg, w^1, NULL)!=1) return MUL_INPUT;
|
||||
v = w;
|
||||
if(v == begNode) return LOOP;
|
||||
if(eLen >= max_ext) return LONG_TIPS;
|
||||
}
|
||||
}
|
||||
|
||||
inline uint32_t get_unitig_back(asg_t *sg, ma_ug_t *ug, uint32_t begNode, uint32_t* endNode,
|
||||
long long* nodeLen, long long* baseLen, buf_t* b)
|
||||
{
|
||||
ma_utg_v* u = NULL;
|
||||
uint32_t v = begNode, w, k;
|
||||
uint32_t kv;
|
||||
(*nodeLen) = (*baseLen) = 0;
|
||||
(*endNode) = (uint32_t)-1;
|
||||
if(ug!=NULL) u = &(ug->u);
|
||||
|
||||
while (1)
|
||||
{
|
||||
kv = get_real_length(sg, v, NULL);
|
||||
(*endNode) = v;
|
||||
if(u == NULL)
|
||||
{
|
||||
(*nodeLen)++;
|
||||
}
|
||||
else
|
||||
{
|
||||
(*nodeLen) += EvaluateLen((*u), v>>1);
|
||||
}
|
||||
if(b) kv_push(uint32_t, b->b, v);
|
||||
///means reach the end of a unitig
|
||||
if(kv!=1) (*baseLen) += sg->seq[v>>1].len;
|
||||
if(kv==0) return END_TIPS;
|
||||
if(kv>1) return MUL_OUTPUT;
|
||||
///kv must be 1 here
|
||||
kv = get_real_length(sg, v, &w);
|
||||
///means reach the end of a unitig
|
||||
if(get_real_length(sg, w^1, NULL)!=1)
|
||||
{
|
||||
(*baseLen) += sg->seq[v>>1].len;
|
||||
return MUL_INPUT;
|
||||
}
|
||||
|
||||
for (k = 0; k < asg_arc_n(sg, v); k++)
|
||||
{
|
||||
if(asg_arc_a(sg, v)[k].del) continue;
|
||||
///here is just one undeleted edge
|
||||
(*baseLen) += asg_arc_len(asg_arc_a(sg, v)[k]);
|
||||
break;
|
||||
}
|
||||
|
||||
v = w;
|
||||
if(v == begNode) return LOOP;
|
||||
}
|
||||
}
|
||||
|
||||
inline uint32_t get_unitig(asg_t *sg, ma_ug_t *ug, uint32_t begNode, uint32_t* endNode,
|
||||
long long* nodeLen, long long* baseLen, long long* max_stop_nodeLen, long long* max_stop_baseLen,
|
||||
uint32_t stops_threshold, buf_t* b)
|
||||
{
|
||||
ma_utg_v* u = NULL;
|
||||
uint32_t v = begNode, w, k;
|
||||
uint32_t kv, return_flag, n_stops = 0;
|
||||
long long pre_baseLen = 0, pre_nodeLen = 0;
|
||||
long long cur_baseLen = 0, cur_nodeLen = 0;
|
||||
(*max_stop_nodeLen) = (*max_stop_baseLen) = (*nodeLen) = (*baseLen) = 0;
|
||||
(*endNode) = (uint32_t)-1;
|
||||
if(ug!=NULL) u = &(ug->u);
|
||||
|
||||
while (1)
|
||||
{
|
||||
kv = get_real_length(sg, v, NULL);
|
||||
(*endNode) = v;
|
||||
if(u == NULL)
|
||||
{
|
||||
(*nodeLen)++;
|
||||
}
|
||||
else
|
||||
{
|
||||
(*nodeLen) += EvaluateLen((*u), v>>1);
|
||||
}
|
||||
if(b) kv_push(uint32_t, b->b, v);
|
||||
///means reach the end of a unitig
|
||||
if(kv!=1) (*baseLen) += sg->seq[v>>1].len;
|
||||
if(kv==0)
|
||||
{
|
||||
return_flag = END_TIPS;
|
||||
break;
|
||||
///return END_TIPS;
|
||||
}
|
||||
if(kv>1)
|
||||
{
|
||||
return_flag = MUL_OUTPUT;
|
||||
break;
|
||||
///return MUL_OUTPUT;
|
||||
}
|
||||
///kv must be 1 here
|
||||
kv = get_real_length(sg, v, &w);
|
||||
///means reach the end of a unitig
|
||||
if(get_real_length(sg, w^1, NULL)!=1)
|
||||
{
|
||||
|
||||
n_stops++;
|
||||
if(n_stops >= stops_threshold)
|
||||
{
|
||||
(*baseLen) += sg->seq[v>>1].len;
|
||||
return_flag = MUL_INPUT;
|
||||
break;
|
||||
///return MUL_INPUT;
|
||||
}
|
||||
else
|
||||
{
|
||||
for (k = 0; k < asg_arc_n(sg, v); k++)
|
||||
{
|
||||
if(asg_arc_a(sg, v)[k].del) continue;
|
||||
///here is just one undeleted edge
|
||||
(*baseLen) += asg_arc_len(asg_arc_a(sg, v)[k]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
cur_baseLen = (*baseLen) - pre_baseLen;
|
||||
pre_baseLen = (*baseLen);
|
||||
if(cur_baseLen > (*max_stop_baseLen))
|
||||
{
|
||||
(*max_stop_baseLen) = cur_baseLen;
|
||||
}
|
||||
|
||||
|
||||
cur_nodeLen = (*nodeLen) - pre_nodeLen;
|
||||
pre_nodeLen = (*nodeLen);
|
||||
if(cur_nodeLen > (*max_stop_nodeLen))
|
||||
{
|
||||
(*max_stop_nodeLen) = cur_nodeLen;
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
for (k = 0; k < asg_arc_n(sg, v); k++)
|
||||
{
|
||||
if(asg_arc_a(sg, v)[k].del) continue;
|
||||
///here is just one undeleted edge
|
||||
(*baseLen) += asg_arc_len(asg_arc_a(sg, v)[k]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
v = w;
|
||||
if(v == begNode)
|
||||
{
|
||||
return_flag = LOOP;
|
||||
break;
|
||||
///return LOOP;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
cur_baseLen = (*baseLen) - pre_baseLen;
|
||||
pre_baseLen = (*baseLen);
|
||||
if(cur_baseLen > (*max_stop_baseLen))
|
||||
{
|
||||
(*max_stop_baseLen) = cur_baseLen;
|
||||
}
|
||||
|
||||
|
||||
cur_nodeLen = (*nodeLen) - pre_nodeLen;
|
||||
pre_nodeLen = (*nodeLen);
|
||||
if(cur_nodeLen > (*max_stop_nodeLen))
|
||||
{
|
||||
(*max_stop_nodeLen) = cur_nodeLen;
|
||||
}
|
||||
|
||||
return return_flag;
|
||||
}
|
||||
|
||||
#define UNAVAILABLE (uint32_t)-1
|
||||
#define PLOID 0
|
||||
#define NON_PLOID 1
|
||||
#define DIFF_HAP_RATE 0.75
|
||||
#define TRIO_DROP_THRES 0.9
|
||||
#define TRIO_DROP_LENGTH_THRES 0.8
|
||||
#define MAX_STOP_RATE 0.6
|
||||
#define TANGLE_MISSED_THRES 0.6
|
||||
///if ug == NULL, nsg should be equal to read_sg
|
||||
inline uint32_t check_different_haps(asg_t *nsg, ma_ug_t *ug, asg_t *read_sg,
|
||||
uint32_t v_0, uint32_t v_1, ma_hit_t_alloc* reverse_sources, buf_t* b_0, buf_t* b_1,
|
||||
R_to_U* ruIndex, uint32_t min_edge_length, uint32_t stops_threshold)
|
||||
{
|
||||
uint32_t vEnd, qn, tn, j, is_Unitig, uId;
|
||||
long long ELen_0, ELen_1, tmp, max_stop_nodeLen, max_stop_baseLen;
|
||||
|
||||
b_0->b.n = b_1->b.n = 0;
|
||||
if(get_unitig(nsg, ug, v_0, &vEnd, &ELen_0, &tmp, &max_stop_nodeLen, &max_stop_baseLen,
|
||||
stops_threshold, b_0) == LOOP)
|
||||
{
|
||||
return UNAVAILABLE;
|
||||
}
|
||||
if(get_unitig(nsg, ug, v_1, &vEnd, &ELen_1, &tmp, &max_stop_nodeLen, &max_stop_baseLen,
|
||||
stops_threshold, b_1) == LOOP)
|
||||
{
|
||||
return UNAVAILABLE;
|
||||
}
|
||||
|
||||
if(ELen_0<=min_edge_length || ELen_1<=min_edge_length) return UNAVAILABLE;
|
||||
|
||||
rIdContig b_max, b_min;
|
||||
b_max.b_0 = b_min.b_0 = NULL;
|
||||
b_max.offset = b_max.readI = b_max.untigI = 0;
|
||||
b_min.offset = b_min.readI = b_min.untigI = 0;
|
||||
|
||||
if(ELen_0<=ELen_1)
|
||||
{
|
||||
b_min.b_0 = b_0;
|
||||
b_max.b_0 = b_1;
|
||||
}
|
||||
else
|
||||
{
|
||||
b_min.b_0 = b_1;
|
||||
b_max.b_0 = b_0;
|
||||
}
|
||||
|
||||
uint32_t max_count = 0, min_count = 0;
|
||||
ma_utg_t *node_min = NULL, *node_max = NULL;
|
||||
|
||||
if(ug != NULL)
|
||||
{
|
||||
/*****************************label all unitigs****************************************/
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
node_max = &(ug->u.a[b_max.b_0->b.a[b_max.untigI]>>1]);
|
||||
///each read
|
||||
for (b_max.readI = 0; b_max.readI < node_max->n; b_max.readI++)
|
||||
{
|
||||
qn = (node_max->a[b_max.readI]>>33);
|
||||
set_R_to_U(ruIndex, qn, (b_max.b_0->b.a[b_max.untigI]>>1), 1);
|
||||
}
|
||||
}
|
||||
/*****************************label all unitigs****************************************/
|
||||
|
||||
|
||||
///each unitig
|
||||
for (b_min.untigI = 0; b_min.untigI < b_min.b_0->b.n; b_min.untigI++)
|
||||
{
|
||||
|
||||
node_min = &(ug->u.a[(b_min.b_0->b.a[b_min.untigI]>>1)]);
|
||||
|
||||
///each read
|
||||
for (b_min.readI = 0; b_min.readI < node_min->n; b_min.readI++)
|
||||
{
|
||||
qn = node_min->a[b_min.readI]>>33;
|
||||
|
||||
/************************BUG: don't forget****************************/
|
||||
if(reverse_sources[qn].length > 0) min_count++;
|
||||
///if(reverse_sources[qn].length >= 0) min_count++;
|
||||
/************************BUG: don't forget****************************/
|
||||
for (j = 0; j < (long long)reverse_sources[qn].length; j++)
|
||||
{
|
||||
tn = Get_tn(reverse_sources[qn].buffer[j]);
|
||||
if(read_sg->seq[tn].del == 1)
|
||||
{
|
||||
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
|
||||
if(tn == (uint32_t)-1 || is_Unitig == 1 || read_sg->seq[tn].del == 1) continue;
|
||||
}
|
||||
|
||||
get_R_to_U(ruIndex, tn, &uId, &is_Unitig);
|
||||
if(uId!=(uint32_t)-1 && is_Unitig == 1)
|
||||
{
|
||||
max_count++;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*****************************label all unitigs****************************************/
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
node_max = &(ug->u.a[b_max.b_0->b.a[b_max.untigI]>>1]);
|
||||
///each read
|
||||
for (b_max.readI = 0; b_max.readI < node_max->n; b_max.readI++)
|
||||
{
|
||||
qn = (node_max->a[b_max.readI]>>33);
|
||||
ruIndex->index[qn] = (uint32_t)-1;
|
||||
}
|
||||
}
|
||||
/*****************************label all unitigs****************************************/
|
||||
|
||||
}
|
||||
else
|
||||
{
|
||||
/*****************************label all reads****************************************/
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
qn = (b_max.b_0->b.a[b_max.untigI]>>1);
|
||||
set_R_to_U(ruIndex, qn, 1, 1);
|
||||
}
|
||||
/*****************************label all reads****************************************/
|
||||
|
||||
///each read
|
||||
for (b_min.untigI = 0; b_min.untigI < b_min.b_0->b.n; b_min.untigI++)
|
||||
{
|
||||
qn = (b_min.b_0->b.a[b_min.untigI]>>1);
|
||||
|
||||
/************************BUG: don't forget****************************/
|
||||
if(reverse_sources[qn].length > 0) min_count++;
|
||||
///if(reverse_sources[qn].length >= 0) min_count++;
|
||||
/************************BUG: don't forget****************************/
|
||||
|
||||
for (j = 0; j < (long long)reverse_sources[qn].length; j++)
|
||||
{
|
||||
tn = Get_tn(reverse_sources[qn].buffer[j]);
|
||||
if(nsg->seq[tn].del == 1)
|
||||
{
|
||||
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
|
||||
if(tn == (uint32_t)-1 || is_Unitig == 1 || nsg->seq[tn].del == 1) continue;
|
||||
}
|
||||
|
||||
|
||||
get_R_to_U(ruIndex, tn, &uId, &is_Unitig);
|
||||
if(uId!=(uint32_t)-1 && is_Unitig == 1)
|
||||
{
|
||||
max_count++;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/*****************************label all reads****************************************/
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
qn = (b_max.b_0->b.a[b_max.untigI]>>1);
|
||||
ruIndex->index[qn] = (uint32_t)-1;
|
||||
}
|
||||
/*****************************label all reads****************************************/
|
||||
}
|
||||
|
||||
// if(((v_0==7707) && (v_1==26867))||((v_1==7707) && (v_0==26867)))
|
||||
// {
|
||||
// fprintf(stderr, "******\nv_0>>1: %u, v_0&1: %u, ELen_0: %u\n", v_0>>1, v_0&1, (uint32_t)ELen_0);
|
||||
// fprintf(stderr, "v_1>>1: %u, v_1&1: %u, ELen_1: %u\n", v_1>>1, v_1&1, (uint32_t)ELen_1);
|
||||
// fprintf(stderr, "min_count: %u, max_count: %u, DIFF_HAP_RATE: %f\n\n",
|
||||
// min_count, max_count, DIFF_HAP_RATE);
|
||||
// }
|
||||
|
||||
if(min_count == 0) return UNAVAILABLE;
|
||||
if(max_count > min_count*DIFF_HAP_RATE) return PLOID;
|
||||
return NON_PLOID;
|
||||
}
|
||||
|
||||
|
||||
|
||||
inline uint32_t check_different_haps_naive(asg_t *nsg, ma_ug_t *ug, asg_t *read_sg,
|
||||
uint32_t v_0, uint32_t v_1, ma_hit_t_alloc* reverse_sources, buf_t* b_0, buf_t* b_1,
|
||||
R_to_U* ruIndex, uint32_t min_edge_length, uint32_t stops_threshold)
|
||||
{
|
||||
uint32_t vEnd, qn, tn, j, is_Unitig;
|
||||
long long ELen_0, ELen_1, tmp, max_stop_nodeLen, max_stop_baseLen;
|
||||
|
||||
b_0->b.n = b_1->b.n = 0;
|
||||
if(get_unitig(nsg, ug, v_0, &vEnd, &ELen_0, &tmp, &max_stop_nodeLen, &max_stop_baseLen,
|
||||
stops_threshold, b_0) == LOOP)
|
||||
{
|
||||
return UNAVAILABLE;
|
||||
}
|
||||
if(get_unitig(nsg, ug, v_1, &vEnd, &ELen_1, &tmp, &max_stop_nodeLen, &max_stop_baseLen,
|
||||
stops_threshold, b_1) == LOOP)
|
||||
{
|
||||
return UNAVAILABLE;
|
||||
}
|
||||
|
||||
if(ELen_0<=min_edge_length || ELen_1<=min_edge_length) return UNAVAILABLE;
|
||||
|
||||
rIdContig b_max, b_min;
|
||||
b_max.b_0 = b_min.b_0 = NULL;
|
||||
b_max.offset = b_max.readI = b_max.untigI = 0;
|
||||
b_min.offset = b_min.readI = b_min.untigI = 0;
|
||||
|
||||
if(ELen_0<=ELen_1)
|
||||
{
|
||||
b_min.b_0 = b_0;
|
||||
b_max.b_0 = b_1;
|
||||
}
|
||||
else
|
||||
{
|
||||
b_min.b_0 = b_1;
|
||||
b_max.b_0 = b_0;
|
||||
}
|
||||
|
||||
uint32_t max_count = 0, min_count = 0;
|
||||
ma_utg_t *node_min = NULL, *node_max = NULL;
|
||||
|
||||
if(ug != NULL)
|
||||
{
|
||||
///each unitig
|
||||
for (b_min.untigI = 0; b_min.untigI < b_min.b_0->b.n; b_min.untigI++)
|
||||
{
|
||||
|
||||
node_min = &(ug->u.a[(b_min.b_0->b.a[b_min.untigI]>>1)]);
|
||||
|
||||
///each read
|
||||
for (b_min.readI = 0; b_min.readI < node_min->n; b_min.readI++)
|
||||
{
|
||||
qn = node_min->a[b_min.readI]>>33;
|
||||
|
||||
if(reverse_sources[qn].length > 0) min_count++;
|
||||
for (j = 0; j < (long long)reverse_sources[qn].length; j++)
|
||||
{
|
||||
tn = Get_tn(reverse_sources[qn].buffer[j]);
|
||||
if(read_sg->seq[tn].del == 1)
|
||||
{
|
||||
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
|
||||
if(tn == (uint32_t)-1 || is_Unitig == 1 || read_sg->seq[tn].del == 1) continue;
|
||||
}
|
||||
|
||||
///each unitig
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
node_max = &(ug->u.a[b_max.b_0->b.a[b_max.untigI]>>1]);
|
||||
///each read
|
||||
for (b_max.readI = 0; b_max.readI < node_max->n; b_max.readI++)
|
||||
{
|
||||
if(tn == (node_max->a[b_max.readI]>>33))
|
||||
{
|
||||
max_count++;
|
||||
goto end_check_different_haps_ug;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
end_check_different_haps_ug:;
|
||||
}
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
///each read
|
||||
for (b_min.untigI = 0; b_min.untigI < b_min.b_0->b.n; b_min.untigI++)
|
||||
{
|
||||
qn = (b_min.b_0->b.a[b_min.untigI]>>1);
|
||||
|
||||
if(reverse_sources[qn].length > 0) min_count++;
|
||||
|
||||
for (j = 0; j < (long long)reverse_sources[qn].length; j++)
|
||||
{
|
||||
tn = Get_tn(reverse_sources[qn].buffer[j]);
|
||||
if(nsg->seq[tn].del == 1)
|
||||
{
|
||||
get_R_to_U(ruIndex, tn, &tn, &is_Unitig);
|
||||
if(tn == (uint32_t)-1 || is_Unitig == 1 || nsg->seq[tn].del == 1) continue;
|
||||
}
|
||||
|
||||
///each read
|
||||
for (b_max.untigI = 0; b_max.untigI < b_max.b_0->b.n; b_max.untigI++)
|
||||
{
|
||||
if((b_max.b_0->b.a[b_max.untigI]>>1) == tn)
|
||||
{
|
||||
max_count++;
|
||||
goto end_check_different_haps_non_ug;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
end_check_different_haps_non_ug:;
|
||||
}
|
||||
}
|
||||
|
||||
// if(((v_0==7707) && (v_1==26867))||((v_1==7707) && (v_0==26867)))
|
||||
// {
|
||||
// fprintf(stderr, "******\nv_0>>1: %u, v_0&1: %u, ELen_0: %u\n", v_0>>1, v_0&1, (uint32_t)ELen_0);
|
||||
// fprintf(stderr, "v_1>>1: %u, v_1&1: %u, ELen_1: %u\n", v_1>>1, v_1&1, (uint32_t)ELen_1);
|
||||
// fprintf(stderr, "min_count: %u, max_count: %u, DIFF_HAP_RATE: %f\n\n",
|
||||
// min_count, max_count, DIFF_HAP_RATE);
|
||||
// }
|
||||
|
||||
if(min_count == 0) return UNAVAILABLE;
|
||||
if(max_count > min_count*DIFF_HAP_RATE) return PLOID;
|
||||
return NON_PLOID;
|
||||
}
|
||||
|
||||
|
||||
|
||||
typedef struct {
|
||||
uint32_t father_occ;
|
||||
uint32_t mother_occ;
|
||||
uint32_t ambig_occ;
|
||||
uint32_t drop_occ;
|
||||
uint32_t total;
|
||||
} Trio_counter;
|
||||
|
||||
void resolve_tangles(ma_ug_t *src, asg_t *read_g, ma_hit_t_alloc* reverse_sources, long long minLongUntig,
|
||||
long long maxShortUntig, float l_untig_rate, float max_node_threshold, R_to_U* ruIndex, uint32_t trio_flag,
|
||||
float drop_ratio);
|
||||
void adjust_utg_advance(asg_t *sg, ma_ug_t *ug, ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex);
|
||||
void rescue_contained_reads_aggressive(ma_ug_t *i_ug, asg_t *r_g, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, long long bubble_dist, uint32_t chainLenThres, uint32_t is_bubble_check,
|
||||
uint32_t is_primary_check, kvec_asg_arc_t_warp* new_rtg_edges, kvec_t_u32_warp* new_rtg_nodes);
|
||||
void rescue_missing_overlaps_aggressive(ma_ug_t *i_ug, asg_t *r_g, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, long long bubble_dist, uint32_t is_bubble_check,
|
||||
uint32_t is_primary_check, kvec_asg_arc_t_warp* new_rtg_edges);
|
||||
void deduplicate(ma_ug_t *src, asg_t *read_g, ma_hit_t_alloc* reverse_sources, long long minLongUntig,
|
||||
long long maxShortUntig, float l_untig_rate, float max_node_threshold, R_to_U* ruIndex, uint32_t resolve_tangle);
|
||||
void all_to_all_deduplicate(ma_ug_t* ug, asg_t* read_g, ma_sub_t* coverage_cut,
|
||||
ma_hit_t_alloc* sources, uint8_t postive_flag, float drop_rate, ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, float double_check_rate);
|
||||
void drop_semi_circle(ma_ug_t *ug, asg_t* nsg, asg_t* read_g, ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex);
|
||||
void rescue_wrong_overlaps_to_unitigs(ma_ug_t *i_ug, asg_t *r_g, ma_hit_t_alloc* sources, ma_hit_t_alloc* reverse_sources,
|
||||
ma_sub_t *coverage_cut, R_to_U* ruIndex, int max_hang, int min_ovlp, long long bubble_dist, kvec_asg_arc_t_warp* keep_edges);
|
||||
void get_unitig_trio_flag(ma_utg_t* nsu, uint32_t flag, uint32_t* require, uint32_t* non_require, uint32_t* ambigious);
|
||||
void rescue_missing_overlaps_backward(ma_ug_t *i_ug, asg_t *r_g, ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, long long bubble_dist, uint32_t backward_steps,
|
||||
uint32_t is_bubble_check, uint32_t is_primary_check);
|
||||
uint32_t get_edge_from_source(ma_hit_t_alloc* sources, ma_sub_t *coverage_cut,
|
||||
R_to_U* ruIndex, int max_hang, int min_ovlp, uint32_t query, uint32_t target, asg_arc_t* t);
|
||||
uint64_t asg_bub_pop1_primary_trio(asg_t *g, ma_ug_t *utg, uint32_t v0, int max_dist, buf_t *b,
|
||||
uint32_t positive_flag, uint32_t negative_flag, uint32_t is_pop);
|
||||
int unitig_arc_del_short_diploid_by_length(asg_t *g, float drop_ratio);
|
||||
|
||||
#define JUNK_COV 5
|
||||
#define DISCARD_RATE 0.8
|
||||
|
||||
#endif
|
||||
|
||||
204
POA.cpp
204
POA.cpp
@@ -1,29 +1,27 @@
|
||||
#include "POA.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include "POA.h"
|
||||
#include "Correct.h"
|
||||
#include "Process_Read.h"
|
||||
#define INIT_EDGE_SIZE 50
|
||||
#define INCREASE_EDGE_SIZE 5
|
||||
#define INIT_NODE_SIZE 16000
|
||||
|
||||
|
||||
|
||||
/********
|
||||
* Edge *
|
||||
********/
|
||||
|
||||
void init_Edge_alloc(Edge_alloc* list)
|
||||
{
|
||||
if (list->list == NULL)
|
||||
{
|
||||
list->size = INIT_EDGE_SIZE;
|
||||
list->length = 0;
|
||||
list->delete_length = 0;
|
||||
list->list = (Edge*)malloc(sizeof(Edge)*list->size);
|
||||
}
|
||||
else
|
||||
{
|
||||
list->length = 0;
|
||||
list->delete_length = 0;
|
||||
}
|
||||
|
||||
if (list->list == NULL) {
|
||||
list->size = INIT_EDGE_SIZE;
|
||||
list->length = 0;
|
||||
list->delete_length = 0;
|
||||
list->list = (Edge*)malloc(sizeof(Edge)*list->size);
|
||||
} else {
|
||||
list->length = 0;
|
||||
list->delete_length = 0;
|
||||
}
|
||||
}
|
||||
|
||||
void clear_Edge_alloc(Edge_alloc* list)
|
||||
@@ -34,28 +32,29 @@ void clear_Edge_alloc(Edge_alloc* list)
|
||||
|
||||
void destory_Edge_alloc(Edge_alloc* list)
|
||||
{
|
||||
free(list->list);
|
||||
if (list && list->list)
|
||||
free(list->list);
|
||||
}
|
||||
|
||||
void append_Edge_alloc(Edge_alloc* list, uint64_t in_node, uint64_t out_node, uint64_t weight, uint64_t length)
|
||||
{
|
||||
if (list->length + 1 > list->size)
|
||||
{
|
||||
list->size = list->size + INCREASE_EDGE_SIZE;
|
||||
list->list = (Edge*)realloc(list->list, sizeof(Edge)*list->size);
|
||||
}
|
||||
if (list->length + 1 > list->size) {
|
||||
uint64_t old_size = list->size;
|
||||
list->size = list->size + INCREASE_EDGE_SIZE;
|
||||
list->list = (Edge*)realloc(list->list, sizeof(Edge)*list->size);
|
||||
memset(&list->list[old_size], 0, (list->size - old_size) * sizeof(Edge));
|
||||
}
|
||||
|
||||
list->list[list->length].in_node = in_node;
|
||||
list->list[list->length].out_node = out_node;
|
||||
list->list[list->length].weight = weight;
|
||||
list->list[list->length].length = length;
|
||||
list->list[list->length].num_insertions = 0;
|
||||
list->list[list->length].self_edge_ID = list->length;
|
||||
list->list[list->length].in_node = in_node;
|
||||
list->list[list->length].out_node = out_node;
|
||||
list->list[list->length].weight = weight;
|
||||
list->list[list->length].length = length;
|
||||
list->list[list->length].num_insertions = 0;
|
||||
list->list[list->length].self_edge_ID = list->length;
|
||||
|
||||
list->length++;
|
||||
list->length++;
|
||||
}
|
||||
|
||||
|
||||
int add_and_check_bi_direction_edge(Graph* graph, Node* in_node, Node* out_node, uint64_t weight, uint64_t flag)
|
||||
{
|
||||
Edge* e_forward;
|
||||
@@ -81,7 +80,6 @@ int add_and_check_bi_direction_edge(Graph* graph, Node* in_node, Node* out_node,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
void add_bi_direction_edge(Graph* graph, Node* in_node, Node* out_node, uint64_t weight, uint64_t flag)
|
||||
{
|
||||
|
||||
@@ -95,8 +93,6 @@ void add_bi_direction_edge(Graph* graph, Node* in_node, Node* out_node, uint64_t
|
||||
= Output_Edges((*in_node)).length - 1;
|
||||
}
|
||||
|
||||
|
||||
|
||||
int remove_and_check_bi_direction_edge_from_nodes(Graph* graph, Node* in_node, Node* out_node)
|
||||
{
|
||||
Edge* e_forward;
|
||||
@@ -135,8 +131,6 @@ int remove_and_check_bi_direction_edge_from_nodes(Graph* graph, Node* in_node, N
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
int remove_and_check_bi_direction_edge_from_edge(Graph* graph, Edge* e)
|
||||
{
|
||||
Edge* e_forward;
|
||||
@@ -171,109 +165,71 @@ int remove_and_check_bi_direction_edge_from_edge(Graph* graph, Edge* e)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
|
||||
/********
|
||||
* Node *
|
||||
********/
|
||||
|
||||
void init_Node_alloc(Node_alloc* list)
|
||||
{
|
||||
list->size = INIT_NODE_SIZE;
|
||||
list->length = 0;
|
||||
list->delete_length = 0;
|
||||
list->list = (Node*)malloc(sizeof(Node)*list->size);
|
||||
list->sort.size = 0;
|
||||
list->sort.list = NULL;
|
||||
list->sort.visit = NULL;
|
||||
|
||||
list->sort.iterative_buffer = NULL;
|
||||
list->sort.iterative_buffer_visit = NULL;
|
||||
|
||||
|
||||
uint64_t i;
|
||||
for (i = 0; i < list->size; i++)
|
||||
{
|
||||
list->list[i].insertion_edges.list=NULL;
|
||||
list->list[i].mismatch_edges.list=NULL;
|
||||
list->list[i].deletion_edges.list=NULL;
|
||||
}
|
||||
memset(list, 0, sizeof(Node_alloc));
|
||||
list->size = INIT_NODE_SIZE;
|
||||
list->list = (Node*)calloc(list->size, sizeof(Node));
|
||||
}
|
||||
|
||||
void destory_Node_alloc(Node_alloc* list)
|
||||
{
|
||||
uint64_t i =0;
|
||||
for (i = 0; i < list->length; i++)
|
||||
{
|
||||
destory_Edge_alloc(&list->list[i].deletion_edges);
|
||||
destory_Edge_alloc(&list->list[i].insertion_edges);
|
||||
destory_Edge_alloc(&list->list[i].mismatch_edges);
|
||||
}
|
||||
|
||||
free(list->list);
|
||||
free(list->sort.list);
|
||||
free(list->sort.visit);
|
||||
free(list->sort.iterative_buffer);
|
||||
free(list->sort.iterative_buffer_visit);
|
||||
///free(list->topo_order);
|
||||
uint64_t i;
|
||||
for (i = 0; i < list->size; i++) {
|
||||
destory_Edge_alloc(&list->list[i].deletion_edges);
|
||||
destory_Edge_alloc(&list->list[i].insertion_edges);
|
||||
destory_Edge_alloc(&list->list[i].mismatch_edges);
|
||||
}
|
||||
free(list->list);
|
||||
free(list->sort.list);
|
||||
free(list->sort.visit);
|
||||
free(list->sort.iterative_buffer);
|
||||
free(list->sort.iterative_buffer_visit);
|
||||
}
|
||||
|
||||
void clear_Node_alloc(Node_alloc* list)
|
||||
{
|
||||
uint64_t i =0;
|
||||
for (i = 0; i < list->length; i++)
|
||||
{
|
||||
clear_Edge_alloc(&list->list[i].insertion_edges);
|
||||
clear_Edge_alloc(&list->list[i].mismatch_edges);
|
||||
clear_Edge_alloc(&list->list[i].deletion_edges);
|
||||
}
|
||||
|
||||
list->length = 0;
|
||||
list->delete_length = 0;
|
||||
uint64_t i =0;
|
||||
for (i = 0; i < list->length; i++) { // TODO: is this list->size or list->length? The original version is list->length.
|
||||
clear_Edge_alloc(&list->list[i].insertion_edges);
|
||||
clear_Edge_alloc(&list->list[i].mismatch_edges);
|
||||
clear_Edge_alloc(&list->list[i].deletion_edges);
|
||||
}
|
||||
list->length = 0;
|
||||
list->delete_length = 0;
|
||||
}
|
||||
|
||||
|
||||
uint64_t append_Node_alloc(Node_alloc* list, char base)
|
||||
{
|
||||
|
||||
if (list->length + 1 > list->size)
|
||||
{
|
||||
uint64_t i = list->size;
|
||||
if (list->length + 1 > list->size) {
|
||||
uint64_t old_size = list->size;
|
||||
list->size = list->size * 2;
|
||||
list->list = (Node*)realloc(list->list, sizeof(Node) * list->size);
|
||||
memset(&list->list[old_size], 0, (list->size - old_size) * sizeof(Node));
|
||||
}
|
||||
|
||||
list->size = list->size * 2;
|
||||
list->list = (Node*)realloc(list->list, sizeof(Node)*list->size);
|
||||
///list->topo_order = (uint64_t*)realloc(list->topo_order, sizeof(uint64_t)*list->size);
|
||||
|
||||
for (; i < list->size; i++)
|
||||
{
|
||||
list->list[i].deletion_edges.list=NULL;
|
||||
list->list[i].insertion_edges.list=NULL;
|
||||
list->list[i].mismatch_edges.list=NULL;
|
||||
}
|
||||
}
|
||||
|
||||
list->list[list->length].ID = list->length;
|
||||
list->list[list->length].base = base;
|
||||
list->list[list->length].weight = 1;
|
||||
list->list[list->length].num_insertions = 0;
|
||||
init_Edge_alloc(&list->list[list->length].deletion_edges);
|
||||
init_Edge_alloc(&list->list[list->length].insertion_edges);
|
||||
init_Edge_alloc(&list->list[list->length].mismatch_edges);
|
||||
|
||||
list->length++;
|
||||
list->list[list->length].ID = list->length;
|
||||
list->list[list->length].base = base;
|
||||
list->list[list->length].weight = 1;
|
||||
list->list[list->length].num_insertions = 0;
|
||||
init_Edge_alloc(&list->list[list->length].deletion_edges);
|
||||
init_Edge_alloc(&list->list[list->length].insertion_edges);
|
||||
init_Edge_alloc(&list->list[list->length].mismatch_edges);
|
||||
|
||||
return list->length - 1;
|
||||
list->length++;
|
||||
|
||||
return list->length - 1;
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
/*********
|
||||
* Graph *
|
||||
*********/
|
||||
|
||||
void init_Graph(Graph* g)
|
||||
{
|
||||
@@ -310,12 +266,6 @@ void clear_Graph(Graph* g)
|
||||
clear_Queue(&(g->node_q));
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
void addUnmatchedSeqToGraph(Graph* g, char* g_read_seq, long long g_read_length, long long* startID, long long* endID)
|
||||
{
|
||||
long long firstID, lastID, nodeID, i;
|
||||
@@ -356,8 +306,6 @@ void addUnmatchedSeqToGraph(Graph* g, char* g_read_seq, long long g_read_length,
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
void addmatchedSeqToGraph(Graph* backbone, long long currentNodeID, char* x_string, long long x_length,
|
||||
char* y_string, long long y_length, CIGAR* cigar, long long backbone_start, long long backbone_end)
|
||||
{
|
||||
@@ -422,9 +370,3 @@ void addmatchedSeqToGraph(Graph* backbone, long long currentNodeID, char* x_stri
|
||||
cigar_i++;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
536
Process_Read.cpp
536
Process_Read.cpp
@@ -1,25 +1,8 @@
|
||||
#include "Process_Read.h"
|
||||
#include <stdlib.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <fcntl.h>
|
||||
#include <pthread.h>
|
||||
|
||||
|
||||
gz_files fps;
|
||||
|
||||
R_buffer RDB;
|
||||
static uint64_t total_reads;
|
||||
|
||||
pthread_mutex_t i_readinputMutex;
|
||||
pthread_mutex_t i_queueMutex;
|
||||
pthread_mutex_t i_terminateMutex;
|
||||
pthread_cond_t i_flushCond;
|
||||
pthread_cond_t i_readinputflushCond;
|
||||
pthread_cond_t i_stallCond;
|
||||
pthread_cond_t i_readinputstallCond;
|
||||
pthread_mutex_t i_doneMutex;
|
||||
|
||||
#include "Process_Read.h"
|
||||
|
||||
uint8_t seq_nt6_table[256] = {
|
||||
5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5, 5,
|
||||
@@ -45,41 +28,35 @@ char bit_t_seq_table_rc[256][4] = {{0}};
|
||||
char s_H[5] = {'A', 'C', 'G', 'T', 'N'};
|
||||
char rc_Table[5] = {'T', 'G', 'C', 'A', 'N'};
|
||||
|
||||
|
||||
void init_All_reads(All_reads* r)
|
||||
{
|
||||
memset(r, 0, sizeof(All_reads));
|
||||
r->index_size = READ_INIT_NUMBER;
|
||||
r->read_length = (uint64_t*)malloc(sizeof(uint64_t)*r->index_size);
|
||||
r->read_sperate = NULL;
|
||||
r->N_site = NULL;
|
||||
r->total_reads_bases = 0;
|
||||
r->name_index_size = READ_INIT_NUMBER;
|
||||
r->name_index = (uint64_t*)malloc(sizeof(uint64_t)*r->name_index_size);
|
||||
r->name_index[0] = 0;
|
||||
r->name = NULL;
|
||||
r->total_name_length = 0;
|
||||
r->total_reads = 0;
|
||||
}
|
||||
|
||||
void destory_All_reads(All_reads* r)
|
||||
{
|
||||
uint64_t i = 0;
|
||||
for (i = 0; i < r->total_reads; i++)
|
||||
{
|
||||
if (r->N_site[i] != NULL)
|
||||
{
|
||||
free(r->N_site[i]);
|
||||
}
|
||||
free(r->read_sperate[i]);
|
||||
for (i = 0; i < r->total_reads; i++) {
|
||||
if (r->N_site[i]) free(r->N_site[i]);
|
||||
if (r->read_sperate[i]) free(r->read_sperate[i]);
|
||||
if (r->paf && r->paf[i].buffer) free(r->paf[i].buffer);
|
||||
if (r->reverse_paf && r->reverse_paf[i].buffer) free(r->reverse_paf[i].buffer);
|
||||
}
|
||||
free(r->paf);
|
||||
free(r->reverse_paf);
|
||||
free(r->N_site);
|
||||
free(r->read_sperate);
|
||||
free(r->name);
|
||||
free(r->name_index);
|
||||
free(r->read_length);
|
||||
free(r->trio_flag);
|
||||
}
|
||||
|
||||
|
||||
void write_All_reads(All_reads* r, char* read_file_name)
|
||||
{
|
||||
fprintf(stderr, "Writing reads to disk... \n");
|
||||
@@ -110,9 +87,6 @@ void write_All_reads(All_reads* r, char* read_file_name)
|
||||
{
|
||||
fwrite(&zero, sizeof(zero), 1, fp);
|
||||
}
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
fwrite(r->read_length, sizeof(uint64_t), r->total_reads, fp);
|
||||
@@ -123,22 +97,23 @@ void write_All_reads(All_reads* r, char* read_file_name)
|
||||
|
||||
fwrite(r->name, sizeof(char), r->total_name_length, fp);
|
||||
fwrite(r->name_index, sizeof(uint64_t), r->name_index_size, fp);
|
||||
fwrite(r->trio_flag, sizeof(uint8_t), r->total_reads, fp);
|
||||
fwrite(&(asm_opt.hom_cov), sizeof(asm_opt.hom_cov), 1, fp);
|
||||
fwrite(&(asm_opt.het_cov), sizeof(asm_opt.het_cov), 1, fp);
|
||||
|
||||
free(index_name);
|
||||
fflush(fp);
|
||||
fclose(fp);
|
||||
fprintf(stderr, "Reads has been written.\n");
|
||||
}
|
||||
|
||||
|
||||
|
||||
int load_All_reads(All_reads* r, char* read_file_name)
|
||||
{
|
||||
fprintf(stderr, "Loading reads from disk... \n");
|
||||
char* index_name = (char*)malloc(strlen(read_file_name)+15);
|
||||
sprintf(index_name, "%s.bin", read_file_name);
|
||||
FILE* fp = fopen(index_name, "r");
|
||||
if (!fp)
|
||||
{
|
||||
if (!fp) {
|
||||
free(index_name);
|
||||
return 0;
|
||||
}
|
||||
int local_adapterLen;
|
||||
@@ -161,12 +136,10 @@ int load_All_reads(All_reads* r, char* read_file_name)
|
||||
r->N_site = (uint64_t**)malloc(sizeof(uint64_t*)*r->total_reads);
|
||||
for (i = 0; i < r->total_reads; i++)
|
||||
{
|
||||
|
||||
f_flag += fread(&zero, sizeof(zero), 1, fp);
|
||||
|
||||
if (zero)
|
||||
{
|
||||
|
||||
r->N_site[i] = (uint64_t*)malloc(sizeof(uint64_t)*(zero + 1));
|
||||
r->N_site[i][0] = zero;
|
||||
if (r->N_site[i][0])
|
||||
@@ -178,7 +151,6 @@ int load_All_reads(All_reads* r, char* read_file_name)
|
||||
{
|
||||
r->N_site[i] = NULL;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
r->read_length = (uint64_t*)malloc(sizeof(uint64_t)*r->total_reads);
|
||||
@@ -201,11 +173,15 @@ int load_All_reads(All_reads* r, char* read_file_name)
|
||||
r->name_index = (uint64_t*)malloc(sizeof(uint64_t)*r->name_index_size);
|
||||
f_flag += fread(r->name_index, sizeof(uint64_t), r->name_index_size, fp);
|
||||
|
||||
/****************************may have bugs********************************/
|
||||
r->trio_flag = (uint8_t*)malloc(sizeof(uint8_t)*r->total_reads);
|
||||
f_flag += fread(r->trio_flag, sizeof(uint8_t), r->total_reads, fp);
|
||||
f_flag += fread(&(asm_opt.hom_cov), sizeof(asm_opt.hom_cov), 1, fp);
|
||||
f_flag += fread(&(asm_opt.het_cov), sizeof(asm_opt.het_cov), 1, fp);
|
||||
/****************************may have bugs********************************/
|
||||
|
||||
r->cigars = (Compressed_Cigar_record*)malloc(sizeof(Compressed_Cigar_record)*r->total_reads);
|
||||
r->second_round_cigar = (Compressed_Cigar_record*)malloc(sizeof(Compressed_Cigar_record)*r->total_reads);
|
||||
r->paf = (ma_hit_t_alloc*)malloc(sizeof(ma_hit_t_alloc)*r->total_reads);
|
||||
r->reverse_paf = (ma_hit_t_alloc*)malloc(sizeof(ma_hit_t_alloc)*r->total_reads);
|
||||
for (i = 0; i < r->total_reads; i++)
|
||||
{
|
||||
r->second_round_cigar[i].size = r->cigars[i].size = 0;
|
||||
@@ -215,8 +191,6 @@ int load_All_reads(All_reads* r, char* read_file_name)
|
||||
r->second_round_cigar[i].lost_base_size = r->cigars[i].lost_base_size = 0;
|
||||
r->second_round_cigar[i].lost_base_length = r->cigars[i].lost_base_length = 0;
|
||||
r->second_round_cigar[i].lost_base = r->cigars[i].lost_base = NULL;
|
||||
init_ma_hit_t_alloc(&(r->paf[i]));
|
||||
init_ma_hit_t_alloc(&(r->reverse_paf[i]));
|
||||
}
|
||||
|
||||
free(index_name);
|
||||
@@ -227,31 +201,56 @@ int load_All_reads(All_reads* r, char* read_file_name)
|
||||
}
|
||||
|
||||
|
||||
|
||||
inline void insert_read(All_reads* r, kstring_t* read, kstring_t* name)
|
||||
int destory_read_bin(All_reads* r)
|
||||
{
|
||||
r->total_reads++;
|
||||
r->total_reads_bases = r->total_reads_bases + read->l;
|
||||
r->total_name_length = r->total_name_length + name->l;
|
||||
|
||||
///must +1
|
||||
if (r->index_size < r->total_reads + 2)
|
||||
uint64_t i = 0;
|
||||
for (i = 0; i < r->total_reads; i++)
|
||||
{
|
||||
r->index_size = r->index_size * 2 + 2;
|
||||
r->read_length = (uint64_t*)realloc(r->read_length,sizeof(uint64_t)*(r->index_size));
|
||||
r->name_index_size = r->name_index_size * 2 + 2;
|
||||
r->name_index = (uint64_t*)realloc(r->name_index,sizeof(uint64_t)*(r->name_index_size));
|
||||
if (r->N_site[i]) free(r->N_site[i]);
|
||||
if (r->read_sperate[i]) free(r->read_sperate[i]);
|
||||
if (r->cigars[i].record) free(r->cigars[i].record);
|
||||
if (r->cigars[i].lost_base) free(r->cigars[i].lost_base);
|
||||
if (r->second_round_cigar[i].record) free(r->second_round_cigar[i].record);
|
||||
if (r->second_round_cigar[i].lost_base) free(r->second_round_cigar[i].lost_base);
|
||||
}
|
||||
|
||||
r->read_length[r->total_reads - 1] = read->l;
|
||||
r->name_index[r->total_reads] = r->name_index[r->total_reads-1] + name->l;
|
||||
free(r->N_site);
|
||||
free(r->read_length);
|
||||
free(r->read_size);
|
||||
free(r->read_sperate);
|
||||
free(r->name);
|
||||
free(r->name_index);
|
||||
free(r->trio_flag);
|
||||
free(r->cigars);
|
||||
free(r->second_round_cigar);
|
||||
return 1;
|
||||
}
|
||||
|
||||
|
||||
|
||||
void ha_insert_read_len(All_reads *r, int read_len, int name_len)
|
||||
{
|
||||
r->total_reads++;
|
||||
r->total_reads_bases += (uint64_t)read_len;
|
||||
r->total_name_length += (uint64_t)name_len;
|
||||
|
||||
// must +1
|
||||
if (r->index_size < r->total_reads + 2) {
|
||||
r->index_size = r->index_size * 2 + 2;
|
||||
r->read_length = (uint64_t*)realloc(r->read_length, sizeof(uint64_t) * r->index_size);
|
||||
r->name_index_size = r->name_index_size * 2 + 2;
|
||||
r->name_index = (uint64_t*)realloc(r->name_index, sizeof(uint64_t) * r->name_index_size);
|
||||
}
|
||||
|
||||
r->read_length[r->total_reads - 1] = read_len;
|
||||
r->name_index[r->total_reads] = r->name_index[r->total_reads - 1] + name_len;
|
||||
}
|
||||
|
||||
void malloc_All_reads(All_reads* r)
|
||||
{
|
||||
|
||||
r->read_size = (uint64_t*)malloc(sizeof(uint64_t)*r->total_reads);
|
||||
memcpy (r->read_size, r->read_length, sizeof(uint64_t)*r->total_reads);
|
||||
memcpy(r->read_size, r->read_length, sizeof(uint64_t)*r->total_reads);
|
||||
|
||||
r->read_sperate = (uint8_t**)malloc(sizeof(uint8_t*)*r->total_reads);
|
||||
long long i = 0;
|
||||
@@ -279,7 +278,8 @@ void malloc_All_reads(All_reads* r)
|
||||
|
||||
r->name = (char*)malloc(sizeof(char)*r->total_name_length);
|
||||
r->N_site = (uint64_t**)calloc(r->total_reads, sizeof(uint64_t*));
|
||||
|
||||
r->trio_flag = (uint8_t*)malloc(r->total_reads*sizeof(uint8_t));
|
||||
memset(r->trio_flag, AMBIGU, r->total_reads*sizeof(uint8_t));
|
||||
}
|
||||
|
||||
void destory_UC_Read(UC_Read* r)
|
||||
@@ -305,7 +305,6 @@ void init_aux_table()
|
||||
bit_t_seq_table_rc[i][2] = RC_CHAR(bit_t_seq_table[i][1]);
|
||||
bit_t_seq_table_rc[i][3] = RC_CHAR(bit_t_seq_table[i][0]);
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
@@ -330,18 +329,12 @@ void init_UC_Read(UC_Read* r)
|
||||
bit_t_seq_table_rc[i][2] = RC_CHAR(bit_t_seq_table[i][1]);
|
||||
bit_t_seq_table_rc[i][3] = RC_CHAR(bit_t_seq_table[i][0]);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
void recover_UC_Read_sub_region_begin_end
|
||||
(char* r, long long start_pos, long long length, uint8_t strand, All_reads* R_INF, long long ID, int extra_begin, int extra_end)
|
||||
void recover_UC_Read_sub_region_begin_end(char* r, long long start_pos, long long length, uint8_t strand,
|
||||
All_reads* R_INF, long long ID, int extra_begin, int extra_end)
|
||||
{
|
||||
|
||||
|
||||
|
||||
long long readLen = Get_READ_LENGTH((*R_INF), ID);
|
||||
uint8_t* src = Get_READ((*R_INF), ID);
|
||||
|
||||
@@ -349,17 +342,11 @@ void recover_UC_Read_sub_region_begin_end
|
||||
long long copyLen;
|
||||
long long end_pos = start_pos + length - 1;
|
||||
|
||||
|
||||
|
||||
|
||||
if (strand == 0)
|
||||
{
|
||||
|
||||
i = start_pos;
|
||||
copyLen = 0;
|
||||
|
||||
|
||||
|
||||
long long initLen = start_pos % 4;
|
||||
|
||||
if (initLen != 0)
|
||||
@@ -368,8 +355,6 @@ void recover_UC_Read_sub_region_begin_end
|
||||
copyLen = copyLen + 4 - initLen;
|
||||
i = i + copyLen;
|
||||
}
|
||||
|
||||
|
||||
while (copyLen < length)
|
||||
{
|
||||
memcpy(r+copyLen, bit_t_seq_table[src[i>>2]], 4);
|
||||
@@ -377,7 +362,6 @@ void recover_UC_Read_sub_region_begin_end
|
||||
i = i + 4;
|
||||
}
|
||||
|
||||
|
||||
if (R_INF->N_site[ID])
|
||||
{
|
||||
for (i = 1; i <= (long long)R_INF->N_site[ID][0]; i++)
|
||||
@@ -392,17 +376,12 @@ void recover_UC_Read_sub_region_begin_end
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
start_pos = readLen - start_pos - 1;
|
||||
end_pos = readLen - end_pos - 1;
|
||||
|
||||
|
||||
|
||||
///start_pos > end_pos
|
||||
i = start_pos;
|
||||
copyLen = 0;
|
||||
@@ -428,7 +407,6 @@ void recover_UC_Read_sub_region_begin_end
|
||||
|
||||
for (i = 1; i <= (long long)R_INF->N_site[ID][0]; i++)
|
||||
{
|
||||
|
||||
if ((long long)R_INF->N_site[ID][i] >= end_pos && (long long)R_INF->N_site[ID][i] <= start_pos)
|
||||
{
|
||||
r[readLen - R_INF->N_site[ID][i] - 1 - offset] = 'N';
|
||||
@@ -439,21 +417,11 @@ void recover_UC_Read_sub_region_begin_end
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
void recover_UC_Read_sub_region(char* r, long long start_pos, long long length, uint8_t strand, All_reads* R_INF, long long ID)
|
||||
{
|
||||
|
||||
|
||||
|
||||
long long readLen = Get_READ_LENGTH((*R_INF), ID);
|
||||
uint8_t* src = Get_READ((*R_INF), ID);
|
||||
|
||||
@@ -463,7 +431,6 @@ void recover_UC_Read_sub_region(char* r, long long start_pos, long long length,
|
||||
|
||||
if (strand == 0)
|
||||
{
|
||||
|
||||
i = start_pos;
|
||||
copyLen = 0;
|
||||
|
||||
@@ -475,8 +442,7 @@ void recover_UC_Read_sub_region(char* r, long long start_pos, long long length,
|
||||
copyLen = copyLen + 4 - initLen;
|
||||
i = i + copyLen;
|
||||
}
|
||||
|
||||
|
||||
|
||||
while (copyLen < length)
|
||||
{
|
||||
memcpy(r+copyLen, bit_t_seq_table[src[i>>2]], 4);
|
||||
@@ -484,7 +450,6 @@ void recover_UC_Read_sub_region(char* r, long long start_pos, long long length,
|
||||
i = i + 4;
|
||||
}
|
||||
|
||||
|
||||
if (R_INF->N_site[ID])
|
||||
{
|
||||
for (i = 1; i <= (long long)R_INF->N_site[ID][0]; i++)
|
||||
@@ -499,17 +464,12 @@ void recover_UC_Read_sub_region(char* r, long long start_pos, long long length,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
start_pos = readLen - start_pos - 1;
|
||||
end_pos = readLen - end_pos - 1;
|
||||
|
||||
|
||||
|
||||
///start_pos > end_pos
|
||||
i = start_pos;
|
||||
copyLen = 0;
|
||||
@@ -535,7 +495,6 @@ void recover_UC_Read_sub_region(char* r, long long start_pos, long long length,
|
||||
|
||||
for (i = 1; i <= (long long)R_INF->N_site[ID][0]; i++)
|
||||
{
|
||||
|
||||
if ((long long)R_INF->N_site[ID][i] >= end_pos && (long long)R_INF->N_site[ID][i] <= start_pos)
|
||||
{
|
||||
r[readLen - R_INF->N_site[ID][i] - 1 - offset] = 'N';
|
||||
@@ -546,15 +505,11 @@ void recover_UC_Read_sub_region(char* r, long long start_pos, long long length,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
void recover_UC_Read(UC_Read* r, All_reads* R_INF, uint64_t ID)
|
||||
void recover_UC_Read(UC_Read* r, const All_reads *R_INF, uint64_t ID)
|
||||
{
|
||||
r->length = Get_READ_LENGTH((*R_INF), ID);
|
||||
uint8_t* src = Get_READ((*R_INF), ID);
|
||||
@@ -615,7 +570,6 @@ void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID)
|
||||
index = index + 4;
|
||||
}
|
||||
|
||||
|
||||
if (R_INF->N_site[ID])
|
||||
{
|
||||
for (i = 1; i <= (long long)R_INF->N_site[ID][0]; i++)
|
||||
@@ -623,11 +577,8 @@ void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID)
|
||||
r->seq[r->length - R_INF->N_site[ID][i] - 1] = 'N';
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
#define COMPRESS_BASE {c = seq_nt6_table[(uint8_t)src[i]];\
|
||||
if (c >= 4)\
|
||||
{\
|
||||
@@ -637,9 +588,8 @@ void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID)
|
||||
}\
|
||||
i++;}\
|
||||
|
||||
void compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_lis, uint64_t N_site_occ)
|
||||
void ha_compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_lis, uint64_t N_site_occ)
|
||||
{
|
||||
|
||||
///N_site_lis saves the pos of all Ns in this read
|
||||
///N_site_lis[0] is the number of Ns
|
||||
if (N_site_occ)
|
||||
@@ -658,10 +608,8 @@ void compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_l
|
||||
uint8_t tmp = 0;
|
||||
uint8_t c = 0;
|
||||
|
||||
|
||||
while (i + 4 <= src_l)
|
||||
{
|
||||
|
||||
tmp = 0;
|
||||
|
||||
COMPRESS_BASE;
|
||||
@@ -697,351 +645,8 @@ void compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_l
|
||||
dest[dest_i] = tmp;
|
||||
dest_i++;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
void open_file(gz_files* nfps, char* name)
|
||||
{
|
||||
nfps->fp = gzopen(name, "r");
|
||||
if(nfps->fp == 0)
|
||||
{
|
||||
fprintf(stderr, "[ERROR] Cannot find the input file: %s\n", name);
|
||||
exit(0);
|
||||
}
|
||||
nfps->seq = kseq_init(nfps->fp);
|
||||
}
|
||||
|
||||
|
||||
void close_file(gz_files* nfps)
|
||||
{
|
||||
kseq_destroy(nfps->seq);
|
||||
gzclose(nfps->fp);
|
||||
}
|
||||
|
||||
void init_gz_files(hifiasm_opt_t* asm_opt)
|
||||
{
|
||||
fps.idx = 0;
|
||||
fps.num_reads = asm_opt->num_reads;
|
||||
fps.reads = asm_opt->read_file_names;
|
||||
fps.seq = NULL;
|
||||
fps.fp = NULL;
|
||||
if(fps.num_reads > 0)
|
||||
{
|
||||
open_file(&fps, fps.reads[fps.idx]);
|
||||
fps.idx++;
|
||||
}
|
||||
}
|
||||
|
||||
void destory_gz_files()
|
||||
{
|
||||
close_file(&fps);
|
||||
}
|
||||
|
||||
int read_item()
|
||||
{
|
||||
int l = kseq_read(fps.seq);
|
||||
if(l >= 0 || (l < 0 && fps.idx >= fps.num_reads))
|
||||
{
|
||||
return l;
|
||||
}
|
||||
///l < 0 && fps.idx < fps.num_reads
|
||||
close_file(&fps);
|
||||
open_file(&fps, fps.reads[fps.idx]);
|
||||
fps.idx++;
|
||||
return read_item();
|
||||
}
|
||||
|
||||
|
||||
inline void exchage_kstring_t(kstring_t* a, kstring_t* b)
|
||||
{
|
||||
kstring_t tmp;
|
||||
tmp = *a;
|
||||
*a = *b;
|
||||
*b = tmp;
|
||||
}
|
||||
|
||||
int get_read(kseq_t *s, int adapterLen)
|
||||
{
|
||||
int l;
|
||||
|
||||
///if ((l = kseq_read(seq)) >= 0)
|
||||
if ((l = read_item()) >= 0)
|
||||
{
|
||||
|
||||
exchage_kstring_t(&(fps.seq->comment), &s->comment);
|
||||
exchage_kstring_t(&(fps.seq->name), &s->name);
|
||||
exchage_kstring_t(&(fps.seq->qual), &s->qual);
|
||||
exchage_kstring_t(&(fps.seq->seq), &s->seq);
|
||||
|
||||
if(adapterLen > 0)
|
||||
{
|
||||
if((int)s->seq.l <= adapterLen*2)
|
||||
{
|
||||
s->seq.l = 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
long long i;
|
||||
for (i = 0; i < ((int)s->seq.l - adapterLen*2); i++)
|
||||
{
|
||||
s->seq.s[i] = s->seq.s[i + adapterLen];
|
||||
}
|
||||
s->seq.l -= adapterLen*2;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
return 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
void init_R_buffer_block(R_buffer_block* curr_sub_block)
|
||||
{
|
||||
curr_sub_block->read = (kseq_t*)calloc(RDB.block_inner_size, sizeof(kseq_t));
|
||||
curr_sub_block->num = 0;
|
||||
}
|
||||
|
||||
void clear_R_buffer()
|
||||
{
|
||||
RDB.all_read_end = 0;
|
||||
RDB.num = 0;
|
||||
}
|
||||
void init_R_buffer(int thread_num)
|
||||
{
|
||||
RDB.all_read_end = 0;
|
||||
RDB.num = 0;
|
||||
RDB.block_inner_size = READ_BLOCK_SIZE;
|
||||
RDB.size = thread_num*READ_BLOCK_NUM_PRE_THR;
|
||||
|
||||
RDB.sub_block = (R_buffer_block*)malloc(sizeof(R_buffer_block)*RDB.size);
|
||||
|
||||
int i = 0;
|
||||
|
||||
for (i = 0; i < RDB.size; i++)
|
||||
{
|
||||
init_R_buffer_block(&RDB.sub_block[i]);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
void destory_R_buffer_block(R_buffer_block* curr_sub_block)
|
||||
{
|
||||
kseq_destroy(curr_sub_block->read);
|
||||
}
|
||||
|
||||
|
||||
void destory_R_buffer()
|
||||
{
|
||||
int i = 0;
|
||||
|
||||
for (i = 0; i < RDB.size; i++)
|
||||
{
|
||||
destory_R_buffer_block(&RDB.sub_block[i]);
|
||||
}
|
||||
|
||||
free(RDB.sub_block);
|
||||
|
||||
}
|
||||
|
||||
|
||||
inline void load_read_block(R_buffer_block* read_batch, int batch_read_size,
|
||||
int* return_file_flag, int is_insert, int adapterLen)
|
||||
{
|
||||
int inner_i = 0;
|
||||
int file_flag = 1;
|
||||
|
||||
|
||||
|
||||
|
||||
while (inner_i<batch_read_size)
|
||||
{
|
||||
|
||||
file_flag = get_read(&read_batch->read[inner_i], adapterLen);
|
||||
|
||||
if (file_flag == 1)
|
||||
{
|
||||
read_batch->read[inner_i].ID = total_reads;
|
||||
total_reads++;
|
||||
|
||||
|
||||
if (is_insert)
|
||||
{
|
||||
insert_read(&R_INF, &read_batch->read[inner_i].seq,
|
||||
&read_batch->read[inner_i].name);
|
||||
}
|
||||
|
||||
inner_i++;
|
||||
}
|
||||
else if (file_flag == 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (inner_i || file_flag)
|
||||
{
|
||||
file_flag = 1;
|
||||
}
|
||||
|
||||
*return_file_flag = file_flag;
|
||||
read_batch->num = inner_i;
|
||||
|
||||
}
|
||||
|
||||
|
||||
inline void push_R_block(R_buffer_block* tmp_sub_block)
|
||||
{
|
||||
|
||||
|
||||
///only exchange pointers
|
||||
kseq_t *k1;
|
||||
k1 = RDB.sub_block[RDB.num].read;
|
||||
|
||||
RDB.sub_block[RDB.num].read = tmp_sub_block->read;
|
||||
|
||||
tmp_sub_block->read = k1;
|
||||
|
||||
RDB.sub_block[RDB.num].num = tmp_sub_block->num;
|
||||
tmp_sub_block->num = 0;
|
||||
|
||||
RDB.num++;
|
||||
}
|
||||
|
||||
|
||||
inline void pop_R_block(R_buffer_block* curr_sub_block)
|
||||
{
|
||||
RDB.num--;
|
||||
|
||||
///only exchange pointers
|
||||
kseq_t *k1;
|
||||
k1 = RDB.sub_block[RDB.num].read;
|
||||
|
||||
RDB.sub_block[RDB.num].read = curr_sub_block->read;
|
||||
|
||||
curr_sub_block->read = k1;
|
||||
|
||||
curr_sub_block->num = RDB.sub_block[RDB.num].num;
|
||||
RDB.sub_block[RDB.num].num = 0;
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
void* input_reads_muti_threads(void* arg)
|
||||
{
|
||||
int is_insert = *((int*)arg);
|
||||
|
||||
|
||||
total_reads = 0;
|
||||
|
||||
|
||||
int file_flag = 1;
|
||||
|
||||
R_buffer_block tmp_buf;
|
||||
|
||||
init_R_buffer_block(&tmp_buf);
|
||||
|
||||
|
||||
|
||||
while (1)
|
||||
{
|
||||
load_read_block(&tmp_buf, RDB.block_inner_size, &file_flag, is_insert, asm_opt.adapterLen);
|
||||
|
||||
if (file_flag == 0)
|
||||
{
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
pthread_mutex_lock(&i_readinputMutex);
|
||||
while (IS_FULL(RDB))
|
||||
{
|
||||
|
||||
pthread_cond_signal(&i_readinputstallCond);
|
||||
pthread_cond_wait(&i_readinputflushCond, &i_readinputMutex);
|
||||
}
|
||||
|
||||
|
||||
push_R_block(&tmp_buf);
|
||||
|
||||
pthread_cond_signal(&i_readinputstallCond);
|
||||
pthread_mutex_unlock(&i_readinputMutex);
|
||||
}
|
||||
|
||||
|
||||
pthread_mutex_lock(&i_readinputMutex);
|
||||
RDB.all_read_end = 1;
|
||||
pthread_cond_signal(&i_readinputstallCond); //important
|
||||
pthread_mutex_unlock(&i_readinputMutex);
|
||||
|
||||
destory_R_buffer_block(&tmp_buf);
|
||||
|
||||
fprintf(stderr, "Reads #: %lu\n", (unsigned long)total_reads);
|
||||
fprintf(stderr, "Bases #: %lu\n", (unsigned long)R_INF.total_reads_bases);
|
||||
|
||||
|
||||
return NULL;
|
||||
}
|
||||
|
||||
|
||||
|
||||
int get_reads_mul_thread(R_buffer_block* curr_sub_block)
|
||||
{
|
||||
|
||||
|
||||
pthread_mutex_lock(&i_readinputMutex);
|
||||
|
||||
|
||||
while (IS_EMPTY(RDB) && RDB.all_read_end == 0)
|
||||
{
|
||||
|
||||
pthread_cond_signal(&i_readinputflushCond);
|
||||
pthread_cond_wait(&i_readinputstallCond, &i_readinputMutex);
|
||||
}
|
||||
|
||||
|
||||
if (!IS_EMPTY(RDB))
|
||||
{
|
||||
pop_R_block(curr_sub_block);
|
||||
pthread_cond_signal(&i_readinputflushCond);
|
||||
pthread_mutex_unlock(&i_readinputMutex);
|
||||
|
||||
|
||||
return 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
curr_sub_block->num = 0;
|
||||
|
||||
pthread_cond_signal(&i_readinputstallCond); //important
|
||||
|
||||
pthread_mutex_unlock(&i_readinputMutex);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
void reverse_complement(char* pattern, uint64_t length)
|
||||
{
|
||||
uint64_t i = 0;
|
||||
@@ -1061,7 +666,4 @@ void reverse_complement(char* pattern, uint64_t length)
|
||||
{
|
||||
pattern[end] = RC_CHAR(pattern[end]);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -5,7 +5,6 @@
|
||||
#include <string.h>
|
||||
#include <stdlib.h>
|
||||
#include <zlib.h>
|
||||
#include "kseq.h"
|
||||
#include "Overlaps.h"
|
||||
#include "CommandLines.h"
|
||||
///#include "Hash_Table.h"
|
||||
@@ -18,16 +17,11 @@
|
||||
#define IS_FULL(buffer) ((buffer.num >= buffer.size)?1:0)
|
||||
#define IS_EMPTY(buffer) ((buffer.num == 0)?1:0)
|
||||
///#define Get_READ_LENGTH(R_INF, ID) (R_INF.index[ID+1] - R_INF.index[ID])
|
||||
#define Get_READ_LENGTH(R_INF, ID) R_INF.read_length[(ID)]
|
||||
#define Get_NAME_LENGTH(R_INF, ID) (R_INF.name_index[(ID)+1] - R_INF.name_index[(ID)])
|
||||
#define Get_READ_LENGTH(R_INF, ID) (R_INF).read_length[(ID)]
|
||||
#define Get_NAME_LENGTH(R_INF, ID) ((R_INF).name_index[(ID)+1] - (R_INF).name_index[(ID)])
|
||||
///#define Get_READ(R_INF, ID) R_INF.read + (R_INF.index[ID]>>2) + ID
|
||||
#define Get_READ(R_INF, ID) R_INF.read_sperate[(ID)]
|
||||
#define Get_NAME(R_INF, ID) R_INF.name + R_INF.name_index[(ID)]
|
||||
|
||||
|
||||
|
||||
KSEQ_INIT(gzFile, gzread)
|
||||
|
||||
#define Get_READ(R_INF, ID) (R_INF).read_sperate[(ID)]
|
||||
#define Get_NAME(R_INF, ID) ((R_INF).name + (R_INF).name_index[(ID)])
|
||||
|
||||
|
||||
extern uint8_t seq_nt6_table[256];
|
||||
@@ -37,12 +31,9 @@ extern char s_H[5];
|
||||
extern char rc_Table[5];
|
||||
|
||||
|
||||
|
||||
#define RC_CHAR(x) rc_Table[seq_nt6_table[(uint8_t)x]]
|
||||
|
||||
void init_aux_table();
|
||||
int get_read(kseq_t *s, int adapterLen);
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
@@ -76,7 +67,6 @@ inline void init_PAF_alloc(PAF_alloc* list)
|
||||
list->list = (PAF*)malloc(sizeof(PAF)*list->size);
|
||||
}
|
||||
|
||||
|
||||
inline void append_PAF_alloc(PAF_alloc* list, PAF* e)
|
||||
{
|
||||
if(list->length+1 > list->size)
|
||||
@@ -89,9 +79,6 @@ inline void append_PAF_alloc(PAF_alloc* list, PAF* e)
|
||||
list->length++;
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
/**[0-1] bits are type:**/
|
||||
@@ -104,8 +91,14 @@ typedef struct
|
||||
uint32_t lost_base_length;
|
||||
uint32_t lost_base_size;
|
||||
uint32_t new_length;
|
||||
}Compressed_Cigar_record;
|
||||
} Compressed_Cigar_record;
|
||||
|
||||
#define AMBIGU 0
|
||||
#define FATHER 1
|
||||
#define MOTHER 2
|
||||
#define MIX_TRIO 3
|
||||
#define NON_TRIO 4
|
||||
#define DROP 5
|
||||
|
||||
typedef struct
|
||||
{
|
||||
@@ -113,17 +106,16 @@ typedef struct
|
||||
///uint8_t* read;
|
||||
char* name;
|
||||
|
||||
|
||||
uint8_t** read_sperate;
|
||||
uint64_t* read_length;
|
||||
uint64_t* read_size;
|
||||
uint8_t* trio_flag;
|
||||
|
||||
///seq start pos in uint8_t* read
|
||||
///do not need it
|
||||
///uint64_t* index;
|
||||
uint64_t index_size;
|
||||
|
||||
|
||||
///name start pos in char* name
|
||||
uint64_t* name_index;
|
||||
uint64_t name_index_size;
|
||||
@@ -136,32 +128,10 @@ typedef struct
|
||||
|
||||
ma_hit_t_alloc* paf;
|
||||
ma_hit_t_alloc* reverse_paf;
|
||||
ma_sub_t* coverage_cut;
|
||||
|
||||
} All_reads;
|
||||
|
||||
extern All_reads R_INF;
|
||||
|
||||
void malloc_All_reads(All_reads* r);
|
||||
|
||||
typedef struct
|
||||
{
|
||||
kseq_t* read;
|
||||
long long num;
|
||||
|
||||
} R_buffer_block;
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
R_buffer_block* sub_block;
|
||||
long long block_inner_size;
|
||||
long long size;
|
||||
long long num;
|
||||
int all_read_end;
|
||||
} R_buffer;
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
char* seq;
|
||||
@@ -170,23 +140,12 @@ typedef struct
|
||||
long long RID;
|
||||
} UC_Read;
|
||||
|
||||
typedef struct
|
||||
{
|
||||
gzFile fp;
|
||||
kseq_t *seq;
|
||||
char** reads;
|
||||
int num_reads;
|
||||
int idx;
|
||||
} gz_files;
|
||||
|
||||
void init_R_buffer(int thread_num);
|
||||
void init_All_reads(All_reads* r);
|
||||
void* input_reads_muti_threads(void*);
|
||||
void init_R_buffer_block(R_buffer_block* curr_sub_block);
|
||||
int get_reads_mul_thread(R_buffer_block* curr_sub_block);
|
||||
void compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_lis, uint64_t N_site_occ);
|
||||
void malloc_All_reads(All_reads* r);
|
||||
void ha_insert_read_len(All_reads *r, int read_len, int name_len);
|
||||
void ha_compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_lis, uint64_t N_site_occ);
|
||||
void init_UC_Read(UC_Read* r);
|
||||
void recover_UC_Read(UC_Read* r, All_reads* R_INF, uint64_t ID);
|
||||
void recover_UC_Read(UC_Read* r, const All_reads *R_INF, uint64_t ID);
|
||||
void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID);
|
||||
void recover_UC_Read_sub_region(char* r, long long start_pos, long long length, uint8_t strand, All_reads* R_INF, long long ID);
|
||||
void destory_UC_Read(UC_Read* r);
|
||||
@@ -194,13 +153,6 @@ void reverse_complement(char* pattern, uint64_t length);
|
||||
void write_All_reads(All_reads* r, char* read_file_name);
|
||||
int load_All_reads(All_reads* r, char* read_file_name);
|
||||
void destory_All_reads(All_reads* r);
|
||||
|
||||
void destory_R_buffer_block(R_buffer_block* curr_sub_block);
|
||||
void destory_R_buffer();
|
||||
void clear_R_buffer();
|
||||
|
||||
void init_gz_files(hifiasm_opt_t* asm_opt);
|
||||
void destory_gz_files();
|
||||
|
||||
int destory_read_bin(All_reads* r);
|
||||
|
||||
#endif
|
||||
|
||||
4396
Purge_Dups.cpp
Normal file
4396
Purge_Dups.cpp
Normal file
File diff suppressed because it is too large
Load Diff
24
Purge_Dups.h
Normal file
24
Purge_Dups.h
Normal file
@@ -0,0 +1,24 @@
|
||||
#ifndef __PURGEDUPS__
|
||||
#define __PURGEDUPS__
|
||||
#include <stdio.h>
|
||||
#include <stdint.h>
|
||||
#include "kvec.h"
|
||||
#include "kdq.h"
|
||||
#include "Overlaps.h"
|
||||
#include "Hash_Table.h"
|
||||
#define COV_COUNT 1024
|
||||
#define HOM_PEAK_RATE 1.25
|
||||
#define HET_PEAK_RATE (HOM_PEAK_RATE*2)
|
||||
#define ALTER_COV_THRES 0.9
|
||||
#define REAL_ALTER_THRES 0.1
|
||||
|
||||
void purge_dups(ma_ug_t *ug, asg_t *read_g, ma_sub_t* coverage_cut, ma_hit_t_alloc* sources,
|
||||
ma_hit_t_alloc* reverse_sources, R_to_U* ruIndex, kvec_asg_arc_t_warp* edge, float density,
|
||||
uint32_t purege_minLen, int max_hang, int min_ovlp, long long bubble_dist, float drop_ratio,
|
||||
uint32_t just_contain, uint32_t just_coverage);
|
||||
void fill_unitig(uint64_t* buffer, uint32_t bufferLen, asg_t* read_g, kvec_asg_arc_t_warp* edge,
|
||||
uint32_t is_circle, uint64_t* rLen);
|
||||
void get_contig_length(ma_ug_t *ug, asg_t *g, uint64_t* primaryLen, uint64_t* alterLen);
|
||||
void enable_debug_mode(uint32_t mode);
|
||||
|
||||
#endif
|
||||
211
README.md
211
README.md
@@ -4,106 +4,169 @@
|
||||
# Install hifiasm (requiring g++ and zlib)
|
||||
git clone https://github.com/chhylp123/hifiasm
|
||||
cd hifiasm && make
|
||||
# Assembly
|
||||
./hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
||||
|
||||
# Run on test data (use -f0 for small datasets)
|
||||
wget https://github.com/chhylp123/hifiasm/releases/download/v0.7/chr11-2M.fa.gz
|
||||
./hifiasm -o test -t4 -f0 chr11-2M.fa.gz 2> test.log
|
||||
awk '/^S/{print ">"$1;print $2}' test.p_ctg.gfa > test.p_ctg.fa # get primary contigs in FASTA
|
||||
|
||||
# Assemble inbred/homozygous genomes (-l0 disables duplication purging)
|
||||
hifiasm -o CHM13.asm -t32 -l0 CHM13-HiFi.fa.gz 2> CHM13.asm.log
|
||||
# Assemble heterozygous with built-in duplication purging
|
||||
hifiasm -o HG002.asm -t32 HG002-file1.fq.gz HG002-file2.fq.gz
|
||||
|
||||
# Trio binning assembly (requiring https://github.com/lh3/yak)
|
||||
yak count -b37 -t16 -o pat.yak <(cat pat_1.fq.gz pat_2.fq.gz) <(cat pat_1.fq.gz pat_2.fq.gz)
|
||||
yak count -b37 -t16 -o mat.yak <(cat mat_1.fq.gz mat_2.fq.gz) <(cat mat_1.fq.gz mat_2.fq.gz)
|
||||
hifiasm -o HG002.asm -t32 -1 pat.yak -2 mat.yak HG002-HiFi.fa.gz
|
||||
```
|
||||
|
||||
## Introduction
|
||||
|
||||
Hifiasm is a fast haplotype-reserved de novo assembler for PacBio
|
||||
Hifi reads. Unlike most existing assemblers, hifiasm starts from uncollapsed
|
||||
genome. Thus, it is able to keep the haplotype information as much as possible.
|
||||
The input of hifiasm is the PacBio Hifi reads in fasta/fastq format, and its
|
||||
outputs consist of:
|
||||
Hifiasm is a fast haplotype-resolved de novo assembler for PacBio Hifi reads.
|
||||
It can assemble a human genome in several hours and works with the California
|
||||
redwood genome, one of the most complex genomes sequenced so far. Hifiasm can
|
||||
produce primary/alternate assemblies of quality competitive with the best
|
||||
assemblers. It also introduces a new graph binning algorithm and achieves
|
||||
the best haplotype-resolved assembly given trio data.
|
||||
|
||||
## Usage
|
||||
|
||||
A typical hifiasm command line looks like:
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
||||
```
|
||||
where `NA12878.fq.gz` provides the input reads, `-t` sets the number of CPUs in
|
||||
use and `-o` specifies the prefix of output files. For this example, the
|
||||
primary contigs are written to `NA12878.asm.p_ctg.gfa` and alternate contigs to
|
||||
`NA12878.asm.a_ctg.gfa`. At the first run, hifiasm saves corrected reads and
|
||||
overlaps to disk as `NA12878.asm.*.bin`. It reuses the saved results to avoid
|
||||
the time-consuming all-vs-all overlap calculation next time. You may specify
|
||||
`-i` to ignore precomputed overlaps and redo overlapping from raw reads.
|
||||
|
||||
Hifiasm purges haplotig duplications by default. For inbred or homozygous
|
||||
genomes, you may disable purging with option `-l0`. Old HiFi reads may contain
|
||||
short adapter sequences at the ends of reads. You can specify `-z20` to trim
|
||||
both ends of reads by 20bp. For small genomes, use `-f0` to disable the initial
|
||||
bloom filter which takes 16GB memory at the beginning. For genomes much larger
|
||||
than human, applying `-f38` or even `-f39` is preferred to save memory on k-mer
|
||||
counting.
|
||||
|
||||
When parental short reads are available, hifiasm can generate a pair of
|
||||
haplotype-resolved assemblies with trio binning. To perform such assembly, you
|
||||
need to count k-mers first with [yak][yak] first and then do assembly:
|
||||
```sh
|
||||
yak count -k31 -b37 -t16 -o pat.yak paternal.fq.gz
|
||||
yak count -k31 -b37 -t16 -o mat.yak maternal.fq.gz
|
||||
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak NA12878.fq.gz
|
||||
```
|
||||
Here `NA12878.asm.hap1.p_ctg.gfa` and `NA12878.asm.hap2.p_ctg.gfa` give the two
|
||||
haplotype assemblies. In the binning mode, hifiasm does not purge haplotig
|
||||
duplications by default. Because hifiasm reuses saved overlaps, you can
|
||||
generate both primary/alternate assemblies and trio binning assemblies with
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t 32 NA12878.fq.gz 2> NA12878.asm.pri.log
|
||||
hifiasm -o NA12878.asm -t 32 -1 pat.yak -2 mat.yak /dev/null 2> NA12878.asm.trio.log
|
||||
```
|
||||
The second command line will run much faster than the first. You can also dump
|
||||
error corrected in FASTA and/or overlaps in PAF with
|
||||
```sh
|
||||
hifiasm -o NA12878.asm -t 32 --write-paf --write-ec /dev/null
|
||||
```
|
||||
|
||||
## Output files
|
||||
|
||||
For non-trio assembly, hifiasm generates the following files:
|
||||
|
||||
1. Haplotype-resolved raw [unitig][unitig] graph in [GFA][gfa] format
|
||||
(*prefix*.r\_utg.gfa). This graph keeps all haplotype information, including
|
||||
somatic mutations and recurrent sequencing errors.
|
||||
2. Haplotype-resolved processed unitig graph without small bubbles
|
||||
(*prefix*.p\_utg.gfa). This is usually the preferred output for highly
|
||||
heterozygous genomes.
|
||||
3. Primary assembly [contig][unitig] graph (*prefix*.p\_ctg.gfa). This is the
|
||||
preferred output for inbred strains or human. For highly heterozygous
|
||||
genomes, this graph may represent multiple haplotypes. We plan to change
|
||||
this to represent one set of haplotypes.
|
||||
4. Alternate assembly contig graph (*prefix*.a\_ctg.gfa).
|
||||
5. Haplotype-aware error corrected reads in fasta format (*prefix*.ec.fa).
|
||||
6. All-to-all overlaps in the [PAF][paf] format (*prefix*.ovlp.paf).
|
||||
(*prefix*.p\_utg.gfa). Small bubbles might be caused by somatic mutations or noise in data,
|
||||
which are not the real haplotype information.
|
||||
3. Primary assembly [contig][unitig] graph (*prefix*.p\_ctg.gfa). This graph collapses different
|
||||
haplotypes.
|
||||
4. Alternate assembly contig graph (*prefix*.a\_ctg.gfa). This graph consists of all assemblies that
|
||||
are discarded in primary contig graph.
|
||||
|
||||
So far hifiasm is still in early development stage, it will output phased
|
||||
chromosome-level high-quality assembly in the near future. In addition, hifiasm
|
||||
also outputs three binary files that save all overlap inforamtion
|
||||
(hifiasm.asm.ovlp, hifiasm.asm.ovlp.source, hifiasm.asm.ovlp.reverse in default). With these files, hifiasm can avoid the time-consuming all-to-all overlap calculation step, and do the assembly
|
||||
directly and quickly. This might be helpful when you want to get an optimized
|
||||
assembly by multiple rounds of experiments with different parameters.
|
||||
For trio assembly, hifiasm generates the following files:
|
||||
|
||||
Hifiasm is a standalone and lightweight assembler, which does not need external
|
||||
libraries (except zlib). For large genomes, it can generate high-quality
|
||||
assembly in a few hours. Hifiasm has been tested on the following datasets:
|
||||
1. Haplotype-resolved raw [unitig][unitig] graph in [GFA][gfa] format
|
||||
(*prefix*.r\_utg.gfa). This graph keeps all haplotype information.
|
||||
|
||||
|<sub>Dataset<sub>|<sub>GSize<sub>|<sub>Cov<sub>|<sub>Asm options<sub>|<sub>CPU time<sub>|<sub>Wall time<sub>|<sub>RAM<sub>|<sub>[unitig][unitig]/[contig][unitig] N50<sup>[1]</sup><sub>|
|
||||
2. Phased paternal/haplotype1 contig graph (*prefix*.hap1.p\_ctg.gfa). This graph keeps the phased
|
||||
paternal/haplotype1 assembly.
|
||||
|
||||
3. Phased maternal/haplotype2 contig graph (*prefix*.hap2.p\_ctg.gfa). This graph keeps the phased
|
||||
maternal/haplotype2 assembly.
|
||||
|
||||
Hifiasm writes error corrected reads to the *prefix*.ec.bin binary file and
|
||||
writes overlaps to *prefix*.ovlp.source.bin and *prefix*.ovlp.reverse.bin.
|
||||
|
||||
## Results
|
||||
|
||||
The following table shows the statistics of several hifiasm primary assemblies:
|
||||
|
||||
|<sub>Dataset<sub>|<sub>Size<sub>|<sub>Cov.<sub>|<sub>Asm options<sub>|<sub>CPU time<sub>|<sub>Wall time<sub>|<sub>RAM<sub>|<sub> N50<sub>|
|
||||
|:---------------|-----:|-----:|:---------------------|-------:|--------:|----:|----------------:|
|
||||
|<sub>[Human NA12878]<sub>|<sub>3Gb<sub>|<sub>x28<sub>|<sub>-k 40 -t 42 -r 2<sub>|<sub>200h<sub>| <sub>5h32m<sub>|<sub>114G<sub>|<sub>93.5Kb/18.6Mb<sub>|
|
||||
|<sub>[Human HG002]<sub>|<sub>3Gb<sub>|<sub>x43<sub>|<sub>-k 40 -t 42 -r 2<sub>|<sub>405h10m<sub>|<sub>12h7m<sub>|<sub>146G<sub>|<sub>320kb/29.3Mb<sub>|
|
||||
|<sub>[Human CHM13]<sub>|<sub>3Gb<sub>|<sub>x27<sub>|<sub>-k 40 -t 42 -r 2<sub>|<sub>157h28m<sub>|<sub>5h10m<sub>|<sub>85.8G<sub>|<sub>NA<sup>[2]</sup>/39.8Mb<sub>|
|
||||
|<sub>[Butterfly]<sub>|<sub>358Mb<sub>|<sub>x35<sub>|<sub>-k 40 -t 42 -r 2 -z 20<sub>|<sub>17h6m<sub>|<sub>36m<sub>|<sub>16G<sub>|<sub>7.5Mb/NA<sup>[3]</sup><sub>|
|
||||
|<sub>[Mouse (C57/BL6J)][mouse-data]</sub>|<sub>2.6Gb</sub> |<sub>×25</sub>|<sub>-t48 -l0</sub> |<sub>172.9h</sub> |<sub>4.8h</sub> |<sub>76G</sub> |<sub>21.1Mb</sub>|
|
||||
|<sub>[Maize (B73)][maize-data]</sub> |<sub>2.2Gb</sub> |<sub>×22</sub>|<sub>-t48 -l0</sub> |<sub>203.2h</sub> |<sub>5.1h</sub> |<sub>68G</sub> |<sub>36.7Mb</sub>|
|
||||
|<sub>[Strawberry][strawberry-data]</sub> |<sub>0.8Gb</sub> |<sub>×36</sub>|<sub>-t48 -D10</sub>|<sub>152.7h</sub> |<sub>3.7h</sub> |<sub>91G</sub> |<sub>17.8Mb</sub>|
|
||||
|<sub>[Frog][frog-data]</sub> |<sub>9.5Gb</sub> |<sub>×29</sub>|<sub>-t48</sub> |<sub>2834.3h</sub>|<sub>69.0h</sub>|<sub>463G</sub>|<sub>9.3Mb</sub>|
|
||||
|<sub>[Redwood][redwood-data]</sub> |<sub>35.6Gb</sub>|<sub>×28</sub>|<sub>-t80</sub> |<sub>3890.3h</sub>|<sub>65.5h</sub>|<sub>699G</sub>|<sub>5.4Mb</sub>|
|
||||
|<sub>[Human (CHM13)][CHM13-data]</sub> |<sub>3.1Gb</sub> |<sub>×32</sub>|<sub>-t48 -l0</sub> |<sub>310.7h</sub> |<sub>8.2h</sub> |<sub>114G</sub>|<sub>88.9Mb</sub>|
|
||||
|<sub>[Human (HG00733)][HG00733-data]</sub>|<sub>3.1Gb</sub>|<sub>×33</sub>|<sub>-t48</sub> |<sub>269.1h</sub> |<sub>6.9h</sub> |<sub>135G</sub>|<sub>69.9Mb</sub>|
|
||||
|<sub>[Human (HG002)][NA24385-data]</sub> |<sub>3.1Gb</sub> |<sub>×36</sub>|<sub>-t48</sub> |<sub>305.4h</sub> |<sub>7.7h</sub> |<sub>137G</sub>|<sub>98.7Mb</sub>|
|
||||
|
||||
<sub>[1] unitig N50 is the N50 of assembly graph with haplotype information (i.e., bubbles), while the contig N50 is the N50 of haplotype collapsed assembly (i.e., without bubbles).
|
||||
[2] CHM13 is a homozygous sample, so that unitig N50 makes no sense.
|
||||
[3] Butterfly has high heterozygous rate, so that most chromosomes have been fully separated into two haplotypes. In this case, contig N50 makes no sense.<sub>
|
||||
[mouse-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRR11606870
|
||||
[maize-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRR11606869
|
||||
[strawberry-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRR11606867
|
||||
[frog-data]: https://www.ncbi.nlm.nih.gov/sra?term=(SRR11606868)%20OR%20SRR12048570
|
||||
[redwood-data]: https://www.ncbi.nlm.nih.gov/sra/?term=SRP251156
|
||||
[CHM13-data]: https://www.ncbi.nlm.nih.gov/sra?term=(((SRR11292120)%20OR%20SRR11292121)%20OR%20SRR11292122)%20OR%20SRR11292123
|
||||
|
||||
Note that different species need different assembly graphs. For homozygous genomes (i.e., Human CHM13), the primary assembly contig graph is the best choice.
|
||||
For species with high heterozygous rate (i.e., Butterfly), different haplotypes can be fully separated. It is important to remove small bubbles from the haplotype-resolved unitig graph. The
|
||||
reason is that some small bubbles are caused by somatic mutations or noise in data, which are not
|
||||
the real haplotype information. In this case, haplotype-resolved processed unitig graph
|
||||
without small bubbles should be better. For ordinary human genome (i.e., Human NA12878 and HG002), different haplotypes cannot be fully separated due to the low heterozygous rate. There are many small bubbles including haplotype information, which cannot be simply removed. Thus, it is necessary to use the haplotype-resolved raw unitig graph. **Hifiasm will generate a universal haplotype-resolved contig graph for all species in the near future.**
|
||||
Hifiasm can assemble a 3.1Gb human genome in several hours or a ~30Gb hexaploid
|
||||
redwood genome in a few days on a single machine. For trio binning assembly:
|
||||
|
||||
## Usage
|
||||
|<sub>Dataset<sub>|<sub>Cov.<sub>|<sub>CPU time<sub>|<sub>Elapsed time<sub>|<sub>RAM<sub>|<sub> N50<sub>|
|
||||
|:---------------|-----:|-------:|--------:|----:|----------------:|
|
||||
|<sub>[HG00733][HG00733-data], [\[father\]][HG00731-data], [\[mother\]][HG00732-data]</sub>|<sub>×33</sub>|<sub>269.1h</sub>|<sub>6.9h</sub>|<sub>135G</sub>|<sub>35.1Mb (paternal), 34.9Mb (maternal)</sub>|
|
||||
|<sub>[HG002][NA24385-data], [\[father\]][NA24149-data], [\[mother\]][NA24143-data]</sup>|<sub>×36</sub>|<sub>305.4h</sub>|<sub>7.7h</sub>|<sub>137G</sub>|<sub>41.0Mb (paternal), 40.8Mb (maternal)</sub>|
|
||||
|<sub>[NA12878][NA12878-data], [\[father\]][NA12891-data], [\[mother\]][NA12892-data]</sub>|<sub>×30</sub>|<sub>180.8h</sub>|<sub>4.9h</sub>|<sub>123G</sub>|<sub>27.7Mb (paternal), 27.0Mb (maternal)</sub>|
|
||||
|
||||
For Hifi reads assembly, a typical command line looks like:
|
||||
[HG00733-data]: https://www.ebi.ac.uk/ena/data/view/ERX3831682
|
||||
[HG00731-data]: https://www.ebi.ac.uk/ena/data/view/ERR3241754
|
||||
[HG00732-data]: https://www.ebi.ac.uk/ena/data/view/ERR3241755
|
||||
[NA24385-data]: https://www.ncbi.nlm.nih.gov/sra?term=(((SRR10382244)%20OR%20SRR10382245)%20OR%20SRR10382248)%20OR%20SRR10382249
|
||||
[NA24149-data]: https://ftp-trace.ncbi.nlm.nih.gov/giab/ftp/data/AshkenazimTrio/HG003_NA24149_father/NIST_HiSeq_HG003_Homogeneity-12389378/HG003Run01-13262252/
|
||||
[NA24143-data]: https://ftp-trace.ncbi.nlm.nih.gov/giab/ftp/data/AshkenazimTrio/HG004_NA24143_mother/NIST_HiSeq_HG004_Homogeneity-14572558/HG004Run01-15133132/
|
||||
[NA12878-data]: https://ftp-trace.ncbi.nlm.nih.gov/giab/ftp/data/NA12878/PacBio_SequelII_CCS_11kb/
|
||||
[NA12891-data]: https://www.ebi.ac.uk/ena/data/view/ERR194160
|
||||
[NA12892-data]: https://www.ebi.ac.uk/ena/data/view/ERR194161
|
||||
|
||||
```sh
|
||||
./hifiasm -o NA12878.asm -t 32 NA12878.fq.gz
|
||||
Except NA12878, the assemblies above were produced by hifiasm v0.7 and can be
|
||||
downloaded at
|
||||
```txt
|
||||
ftp://ftp.dfci.harvard.edu/pub/hli/hifiasm/submission/v0.7/
|
||||
```
|
||||
NA12878 was assembled with a more recent version of hifiasm and is available at
|
||||
```txt
|
||||
ftp://ftp.dfci.harvard.edu/pub/hli/hifiasm/NA12878-r253/
|
||||
```
|
||||
|
||||
where `NA12878.fq.gz` is the input reads and `-o` specifies the output files.
|
||||
In this example, all output files can be found at `NA12878.asm.*`. `-k`, `-t`
|
||||
and `-r` specify the length of k-mer, the number of CPU threads, and the number
|
||||
of correction rounds, respectively. Note that at first run, hifiasm will save
|
||||
all overlaps to disk, which can avoid the time-consuming all-to-all overlap
|
||||
calculation next time. For hifiasm, once the overlap information has been
|
||||
obtained during the previous run in advance, it is able to load all overlaps
|
||||
from disk and then directly do assembly. If you want to ignore the pre-computed
|
||||
overlap information, please specify `-i`.
|
||||
|
||||
Please note that some old Hifi reads may consist of short adapters. To improve
|
||||
the assembly quality, adapters should be removed by `-z` as follow:
|
||||
|
||||
```sh
|
||||
./hifiasm -o butterfly.asm -t 42 -z 20 butterfly.fq.gz
|
||||
```
|
||||
|
||||
In this example, hifiasm will remove 20 bases from both ends of each read.
|
||||
|
||||
[unitig]: http://wgs-assembler.sourceforge.net/wiki/index.php/Celera_Assembler_Terminology
|
||||
[gfa]: https://github.com/pmelsted/GFA-spec/blob/master/GFA-spec.md
|
||||
[paf]: https://github.com/lh3/miniasm/blob/master/PAF.md
|
||||
[yak]: https://github.com/lh3/yak
|
||||
|
||||
## Getting Help
|
||||
|
||||
For detailed description of options, please see `man ./hifiasm.1`.
|
||||
The `-h` option of hifiasm also provides simple description of options. If you
|
||||
have further questions, please raise an issue at the issue page.
|
||||
For detailed description of options, please see `man ./hifiasm.1`. The `-h`
|
||||
option of hifiasm also provides brief description of options. If you have
|
||||
further questions, please raise an issue at the [issue
|
||||
page](https://github.com/chhylp123/hifiasm/issues).
|
||||
|
||||
## Limitations and future works
|
||||
## Limitations
|
||||
|
||||
1. For genome with low heterozygous rate, hifiasm only outputs
|
||||
haplotype-resolved assembly graph, instead of the phased chromosome-level
|
||||
assembly (will support such output in future).
|
||||
|
||||
2. For different species, hifiasm outputs different assembly graphs, which are not easy to use.
|
||||
Hifiasm will generate a universal haplotype-resolved contig graph for all species in future.
|
||||
|
||||
3. The running time and memory usage should be further reduced.
|
||||
|
||||
4. The N50 should be further improved.
|
||||
1. Purging haplotig duplications may introduce misassemblies.
|
||||
|
||||
356
Trio.cpp
Normal file
356
Trio.cpp
Normal file
@@ -0,0 +1,356 @@
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <stdarg.h>
|
||||
#include <string.h>
|
||||
#include <assert.h>
|
||||
#include <zlib.h>
|
||||
#include "khashl.h" // hash table
|
||||
#include "kthread.h"
|
||||
#include "kseq.h"
|
||||
#include "Process_Read.h"
|
||||
#include "htab.h"
|
||||
#include "CommandLines.h"
|
||||
|
||||
#define YAK_MAX_KMER 31
|
||||
#define YAK_COUNTER_BITS 10 // yak uses 10, but hifiasm uses 12; we have to copy over some yak code here due to this
|
||||
#define YAK_N_COUNTS (1<<YAK_COUNTER_BITS)
|
||||
#define YAK_MAX_COUNT ((1<<YAK_COUNTER_BITS)-1)
|
||||
|
||||
#define YAK_LOAD_ALL 1
|
||||
#define YAK_LOAD_TRIOBIN1 2
|
||||
#define YAK_LOAD_TRIOBIN2 3
|
||||
|
||||
#define YAK_MAGIC "YAK\2"
|
||||
|
||||
#define yak_ch_eq(a, b) ((a)>>YAK_COUNTER_BITS == (b)>>YAK_COUNTER_BITS) // lower 8 bits for counts; higher bits for k-mer
|
||||
#define yak_ch_hash(a) ((a)>>YAK_COUNTER_BITS)
|
||||
KHASHL_SET_INIT(static klib_unused, yak_ht_t, yak_ht, uint64_t, yak_ch_hash, yak_ch_eq)
|
||||
|
||||
typedef const char *ha_cstr_t;
|
||||
KHASHL_MAP_INIT(static klib_unused, cstr_ht_t, cstr_ht, ha_cstr_t, int64_t, kh_hash_str, kh_eq_str)
|
||||
|
||||
KSTREAM_INIT(gzFile, gzread, 65536)
|
||||
|
||||
typedef struct {
|
||||
struct yak_ht_t *h;
|
||||
} yak_ch1_t;
|
||||
|
||||
typedef struct {
|
||||
int k, pre, n_hash, n_shift;
|
||||
uint64_t tot;
|
||||
yak_ch1_t *h;
|
||||
} yak_ch_t;
|
||||
|
||||
static int yak_ch_get(const yak_ch_t *h, uint64_t x)
|
||||
{
|
||||
int mask = (1<<h->pre) - 1;
|
||||
yak_ht_t *g = h->h[x&mask].h;
|
||||
khint_t k;
|
||||
k = yak_ht_get(g, x >> h->pre << YAK_COUNTER_BITS);
|
||||
return k == kh_end(g)? -1 : kh_key(g, k)&YAK_MAX_COUNT;
|
||||
}
|
||||
|
||||
static yak_ch_t *yak_ch_init(int k, int pre)
|
||||
{
|
||||
yak_ch_t *h;
|
||||
int i;
|
||||
if (pre < YAK_COUNTER_BITS) return 0;
|
||||
CALLOC(h, 1);
|
||||
h->k = k, h->pre = pre;
|
||||
CALLOC(h->h, 1<<h->pre);
|
||||
for (i = 0; i < 1<<h->pre; ++i)
|
||||
h->h[i].h = yak_ht_init();
|
||||
return h;
|
||||
}
|
||||
|
||||
static yak_ch_t *yak_ch_restore_core(yak_ch_t *ch0, const char *fn, int mode, ...)
|
||||
{
|
||||
va_list ap;
|
||||
FILE *fp;
|
||||
uint32_t t[3];
|
||||
char magic[4];
|
||||
int i, j, absent, min_cnt = 0, mid_cnt = 0, mode_err = 0;
|
||||
uint64_t mask = (1ULL<<YAK_COUNTER_BITS) - 1, n_ins = 0, n_new = 0;
|
||||
yak_ch_t *ch;
|
||||
|
||||
va_start(ap, mode);
|
||||
if (mode == YAK_LOAD_ALL) { // do nothing
|
||||
} else if (mode == YAK_LOAD_TRIOBIN1 || mode == YAK_LOAD_TRIOBIN2) {
|
||||
assert(YAK_COUNTER_BITS >= 4);
|
||||
min_cnt = va_arg(ap, int);
|
||||
mid_cnt = va_arg(ap, int);
|
||||
if (ch0 == 0 && mode == YAK_LOAD_TRIOBIN2)
|
||||
mode_err = 1;
|
||||
} else mode_err = 1;
|
||||
va_end(ap);
|
||||
if (mode_err) return 0;
|
||||
|
||||
if ((fp = fopen(fn, "rb")) == 0) return 0;
|
||||
if (fread(magic, 1, 4, fp) != 4) return 0;
|
||||
if (strncmp(magic, YAK_MAGIC, 4) != 0) {
|
||||
fprintf(stderr, "ERROR: wrong file magic.\n");
|
||||
fclose(fp);
|
||||
return 0;
|
||||
}
|
||||
fread(t, 4, 3, fp);
|
||||
if (t[2] != YAK_COUNTER_BITS) {
|
||||
fprintf(stderr, "ERROR: saved counter bits: %d; compile-time counter bits: %d\n", t[2], YAK_COUNTER_BITS);
|
||||
fclose(fp);
|
||||
return 0;
|
||||
}
|
||||
|
||||
ch = ch0 == 0? yak_ch_init(t[0], t[1]) : ch0;
|
||||
assert((int)t[0] == ch->k && (int)t[1] == ch->pre);
|
||||
for (i = 0; i < 1<<ch->pre; ++i) {
|
||||
yak_ht_t *h = ch->h[i].h;
|
||||
fread(t, 4, 2, fp);
|
||||
if (ch0 == 0) yak_ht_resize(h, t[0]);
|
||||
for (j = 0; j < (int)t[1]; ++j) {
|
||||
uint64_t key;
|
||||
fread(&key, 8, 1, fp);
|
||||
if (mode == YAK_LOAD_ALL) {
|
||||
++n_ins;
|
||||
yak_ht_put(h, key, &absent);
|
||||
if (absent) ++n_new;
|
||||
} else if (mode == YAK_LOAD_TRIOBIN1 || mode == YAK_LOAD_TRIOBIN2) {
|
||||
int cnt = key & mask, x, shift = mode == YAK_LOAD_TRIOBIN1? 0 : 2;
|
||||
if (cnt >= mid_cnt) x = 2<<shift;
|
||||
else if (cnt >= min_cnt) x = 1<<shift;
|
||||
else x = -1;
|
||||
if (x >= 0) {
|
||||
khint_t k;
|
||||
key = (key & ~mask) | x;
|
||||
++n_ins;
|
||||
k = yak_ht_put(h, key, &absent);
|
||||
if (absent) ++n_new;
|
||||
else kh_key(h, k) = kh_key(h, k) | x;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
fclose(fp);
|
||||
///fprintf(stderr, "[M::%s] inserted %ld k-mers, of which %ld are new\n", __func__, (long)n_ins, (long)n_new);
|
||||
return ch;
|
||||
}
|
||||
|
||||
static void yak_ch_destroy(yak_ch_t *h)
|
||||
{
|
||||
int i;
|
||||
if (h == 0) return;
|
||||
for (i = 0; i < 1<<h->pre; ++i)
|
||||
yak_ht_destroy(h->h[i].h);
|
||||
free(h->h); free(h);
|
||||
}
|
||||
|
||||
typedef struct {
|
||||
int max;
|
||||
uint32_t *s;
|
||||
} tb_buf_t;
|
||||
|
||||
typedef struct {
|
||||
int k, n_threads, print_diff;
|
||||
double ratio_thres;
|
||||
const yak_ch_t *ch;
|
||||
tb_buf_t *buf;
|
||||
UC_Read *bseq;
|
||||
All_reads* seq;
|
||||
} tb_shared_t;
|
||||
|
||||
typedef struct {
|
||||
int c[16];
|
||||
int sc[2];
|
||||
int nk;
|
||||
} tb_cnt_t;
|
||||
|
||||
typedef struct {
|
||||
int n_seq;
|
||||
tb_shared_t *aux;
|
||||
} tb_step_t;
|
||||
|
||||
static char tb_classify(const int sc[2], const int *c, int k, double ratio_thres)
|
||||
{
|
||||
char type;
|
||||
if (sc[0] == 0 && sc[1] == 0) {
|
||||
if (c[0<<2|2] == c[2<<2|0]) type = '0';
|
||||
else if (c[0<<2|2] >= k - 4 + c[2<<2|0] && (c[2<<2|0] <= 1 || c[0<<2|2] * 0.05 > c[2<<2|0])) type = 'p';
|
||||
else if (c[2<<2|0] >= k - 4 + c[0<<2|2] && (c[0<<2|2] <= 1 || c[2<<2|0] * 0.05 > c[0<<2|2])) type = 'm';
|
||||
else type = '0';
|
||||
} else if (sc[0] > k && sc[1] > k) {
|
||||
type = 'a';
|
||||
} else if (sc[0] >= k - 4 + sc[1] && sc[0] * 0.05 >= sc[1] && c[0<<2|2] * ratio_thres > c[2<<2|0]) {
|
||||
type = 'p';
|
||||
} else if (sc[1] >= k - 4 + sc[0] && sc[1] * 0.05 >= sc[0] && c[2<<2|0] * ratio_thres > c[0<<2|2]) {
|
||||
type = 'm';
|
||||
} else {
|
||||
type = 'a';
|
||||
}
|
||||
return type;
|
||||
}
|
||||
|
||||
static void tb_worker(void *_data, long k, int tid)
|
||||
{
|
||||
tb_shared_t *aux = (tb_shared_t*)_data;
|
||||
UC_Read *s = &aux->bseq[tid];
|
||||
recover_UC_Read(s, aux->seq, k);
|
||||
tb_buf_t *b = &aux->buf[tid];
|
||||
tb_cnt_t cnt; memset(&cnt, 0, sizeof(tb_cnt_t));
|
||||
uint64_t x[4], mask;
|
||||
int i, l, shift;
|
||||
if (aux->ch->k < 32) {
|
||||
mask = (1ULL<<2*aux->ch->k) - 1;
|
||||
shift = 2 * (aux->ch->k - 1);
|
||||
} else {
|
||||
mask = (1ULL<<aux->ch->k) - 1;
|
||||
shift = aux->ch->k - 1;
|
||||
}
|
||||
if (s->length > b->max) {
|
||||
b->max = s->length;
|
||||
kroundup32(b->max);
|
||||
b->s = (uint32_t*)realloc(b->s, b->max * sizeof(uint32_t));
|
||||
}
|
||||
memset(b->s, 0, s->length * sizeof(uint32_t));
|
||||
for (i = l = 0, x[0] = x[1] = x[2] = x[3] = 0; i < s->length; ++i) {
|
||||
int flag, c = seq_nt4_table[(uint8_t)s->seq[i]];
|
||||
if (c < 4) {
|
||||
if (aux->ch->k < 32) {
|
||||
x[0] = (x[0] << 2 | c) & mask;
|
||||
x[1] = x[1] >> 2 | (uint64_t)(3 - c) << shift;
|
||||
} else {
|
||||
x[0] = (x[0] << 1 | (c&1)) & mask;
|
||||
x[1] = (x[1] << 1 | (c>>1)) & mask;
|
||||
x[2] = x[2] >> 1 | (uint64_t)(1 - (c&1)) << shift;
|
||||
x[3] = x[3] >> 1 | (uint64_t)(1 - (c>>1)) << shift;
|
||||
}
|
||||
if (++l >= aux->k) {
|
||||
int type = 0, c1, c2;
|
||||
uint64_t y;
|
||||
++cnt.nk;
|
||||
if (aux->ch->k < 32)
|
||||
y = yak_hash64(x[0] < x[1]? x[0] : x[1], mask);
|
||||
else
|
||||
y = yak_hash_long(x);
|
||||
flag = yak_ch_get(aux->ch, y);
|
||||
if (flag < 0) flag = 0;
|
||||
c1 = flag&3, c2 = flag>>2&3;
|
||||
if (c1 == 2 && c2 == 0) type = 1;
|
||||
else if (c2 == 2 && c1 == 0) type = 2;
|
||||
b->s[i] = type;
|
||||
++cnt.c[flag];
|
||||
}
|
||||
} else l = 0, x[0] = x[1] = x[2] = x[3] = 0;
|
||||
}
|
||||
for (l = 0, i = 1; i <= s->length; ++i) {
|
||||
if (i == s->length || b->s[i] != b->s[l]) {
|
||||
if (b->s[l] > 0 && i - l >= aux->k - 4)
|
||||
cnt.sc[b->s[l] - 1] += i - l;
|
||||
l = i;
|
||||
}
|
||||
}
|
||||
|
||||
int *c = cnt.c;
|
||||
char type;
|
||||
type = tb_classify(cnt.sc, c, aux->k, aux->ratio_thres);
|
||||
aux->seq->trio_flag[k] = AMBIGU;
|
||||
if(type == 'p') aux->seq->trio_flag[k] = FATHER;
|
||||
if(type == 'm') aux->seq->trio_flag[k] = MOTHER;
|
||||
}
|
||||
|
||||
static void ha_triobin_yak(const hifiasm_opt_t *opt)
|
||||
{
|
||||
yak_ch_t *ch;
|
||||
int i /**, min_cnt = 2, mid_cnt = 5**/;
|
||||
tb_shared_t aux;
|
||||
memset(&aux, 0, sizeof(tb_shared_t));
|
||||
aux.n_threads = opt->thread_num, aux.print_diff = 0;
|
||||
aux.ratio_thres = 0.33;
|
||||
aux.seq = &R_INF;
|
||||
|
||||
ch = yak_ch_restore_core(0, opt->fn_bin_yak[0], YAK_LOAD_TRIOBIN1, opt->min_cnt, opt->mid_cnt);
|
||||
ch = yak_ch_restore_core(ch, opt->fn_bin_yak[1], YAK_LOAD_TRIOBIN2, opt->min_cnt, opt->mid_cnt);
|
||||
|
||||
aux.k = ch->k;
|
||||
aux.ch = ch;
|
||||
aux.buf = (tb_buf_t*)calloc(aux.n_threads, sizeof(tb_buf_t));
|
||||
aux.bseq = (UC_Read*)calloc(aux.n_threads, sizeof(UC_Read));
|
||||
for (i = 0; i < aux.n_threads; ++i)
|
||||
init_UC_Read(&aux.bseq[i]);
|
||||
|
||||
kt_for(aux.n_threads, tb_worker, &aux, aux.seq->total_reads);
|
||||
|
||||
for (i = 0; i < aux.n_threads; ++i) {
|
||||
free(aux.buf[i].s);
|
||||
destory_UC_Read(&aux.bseq[i]);
|
||||
}
|
||||
free(aux.buf);
|
||||
free(aux.bseq);
|
||||
yak_ch_destroy(ch);
|
||||
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> partitioned reads using yak dumps\n", __func__, yak_realtime(), yak_cpu_usage());
|
||||
}
|
||||
|
||||
static int ha_triobin_set_list(const cstr_ht_t *h, const char *fn, int flag)
|
||||
{
|
||||
gzFile fp;
|
||||
kstream_t *ks;
|
||||
kstring_t str = {0,0,0};
|
||||
int dret;
|
||||
int64_t n_tot = 0, n_bin = 0;
|
||||
fp = gzopen(fn, "r");
|
||||
if (fp == 0) {
|
||||
fprintf(stderr, "ERROR: failed to open file '%s'\n", fn);
|
||||
return -1;
|
||||
}
|
||||
ks = ks_init(fp);
|
||||
while (ks_getuntil(ks, KS_SEP_LINE, &str, &dret) >= 0) {
|
||||
char *p;
|
||||
khint_t k;
|
||||
++n_tot;
|
||||
for (p = str.s; *p; ++p)
|
||||
if (*p == '\t' || *p == ' ')
|
||||
*p = 0;
|
||||
k = cstr_ht_get(h, str.s);
|
||||
if (k != kh_end(h)) {
|
||||
R_INF.trio_flag[kh_val(h, k)] = flag;
|
||||
++n_bin;
|
||||
}
|
||||
}
|
||||
free(str.s);
|
||||
ks_destroy(ks);
|
||||
gzclose(fp);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] flagged %ld reads, out of %ld lines in file '%s'\n",
|
||||
__func__, yak_realtime(), yak_cpu_usage(), (long)n_bin, (long)n_tot, fn);
|
||||
return 0;
|
||||
}
|
||||
|
||||
static void ha_triobin_list(const hifiasm_opt_t *opt)
|
||||
{
|
||||
int64_t i;
|
||||
khint_t k;
|
||||
cstr_ht_t *h;
|
||||
assert(R_INF.total_reads < (uint32_t)-1);
|
||||
h = cstr_ht_init();
|
||||
for (i = 0; i < (int64_t)R_INF.total_reads; ++i) {
|
||||
int absent;
|
||||
char *str = (char*)calloc(Get_NAME_LENGTH(R_INF, i) + 1, 1);
|
||||
strncpy(str, Get_NAME(R_INF, i), Get_NAME_LENGTH(R_INF, i));
|
||||
k = cstr_ht_put(h, str, &absent);
|
||||
if (absent) kh_val(h, k) = i;
|
||||
}
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] created the hash table for read names\n", __func__, yak_realtime(), yak_cpu_usage());
|
||||
ha_triobin_set_list(h, opt->fn_bin_list[0], FATHER);
|
||||
ha_triobin_set_list(h, opt->fn_bin_list[1], MOTHER);
|
||||
for (k = 0; k < kh_end(h); ++k)
|
||||
if (kh_exist(h, k))
|
||||
free((char*)kh_key(h, k));
|
||||
cstr_ht_destroy(h);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> partitioned reads with external lists\n", __func__, yak_realtime(), yak_cpu_usage());
|
||||
}
|
||||
|
||||
void ha_triobin(const hifiasm_opt_t *opt)
|
||||
{
|
||||
memset(R_INF.trio_flag, AMBIGU, R_INF.total_reads * sizeof(uint8_t));
|
||||
if (opt->fn_bin_list[0] && opt->fn_bin_list[1])
|
||||
ha_triobin_list(opt);
|
||||
if (opt->fn_bin_yak[0] && opt->fn_bin_yak[1])
|
||||
ha_triobin_yak(opt);
|
||||
}
|
||||
183
anchor.cpp
Normal file
183
anchor.cpp
Normal file
@@ -0,0 +1,183 @@
|
||||
#include <stdio.h>
|
||||
#include "htab.h"
|
||||
#include "ksort.h"
|
||||
#include "Hash_Table.h"
|
||||
|
||||
#define HA_KMER_GOOD_RATIO 0.333
|
||||
|
||||
typedef struct { // this struct is not strictly necessary; we can use k_mer_pos instead, with modifications
|
||||
uint64_t srt;
|
||||
uint32_t self_off:31, good:1;
|
||||
uint32_t other_off;
|
||||
} anchor1_t;
|
||||
|
||||
#define an_key1(a) ((a).srt)
|
||||
#define an_key2(a) ((a).self_off)
|
||||
KRADIX_SORT_INIT(ha_an1, anchor1_t, an_key1, 8)
|
||||
KRADIX_SORT_INIT(ha_an2, anchor1_t, an_key2, 4)
|
||||
|
||||
#define oreg_xs_lt(a, b) (((uint64_t)(a).x_pos_s<<32|(a).x_pos_e) < ((uint64_t)(b).x_pos_s<<32|(b).x_pos_e))
|
||||
KSORT_INIT(or_xs, overlap_region, oreg_xs_lt)
|
||||
|
||||
#define oreg_ss_lt(a, b) ((a).shared_seed > (b).shared_seed) // in the decending order
|
||||
KSORT_INIT(or_ss, overlap_region, oreg_ss_lt)
|
||||
|
||||
typedef struct {
|
||||
int n, good;
|
||||
const ha_idxpos_t *a;
|
||||
} seed1_t;
|
||||
|
||||
struct ha_abuf_s {
|
||||
uint64_t n_a, m_a;
|
||||
uint32_t old_mz_m;
|
||||
ha_mz1_v mz;
|
||||
seed1_t *seed;
|
||||
anchor1_t *a;
|
||||
};
|
||||
|
||||
ha_abuf_t *ha_abuf_init(void)
|
||||
{
|
||||
return (ha_abuf_t*)calloc(1, sizeof(ha_abuf_t));
|
||||
}
|
||||
|
||||
void ha_abuf_destroy(ha_abuf_t *ab)
|
||||
{
|
||||
free(ab->seed); free(ab->a); free(ab->mz.a); free(ab);
|
||||
}
|
||||
|
||||
uint64_t ha_abuf_mem(const ha_abuf_t *ab)
|
||||
{
|
||||
return ab->m_a * sizeof(anchor1_t) + ab->mz.m * (sizeof(ha_mz1_t) + sizeof(seed1_t)) + sizeof(ha_abuf_t);
|
||||
}
|
||||
|
||||
static int ha_ov_type(const overlap_region *r, uint32_t len)
|
||||
{
|
||||
if (r->x_pos_s == 0 && r->x_pos_e == len - 1) return 2; // contained in a longer read
|
||||
else if (r->x_pos_s > 0 && r->x_pos_e < len - 1) return 3; // containing a shorter read
|
||||
else return r->x_pos_s == 0? 0 : 1;
|
||||
}
|
||||
|
||||
void ha_get_new_candidates(ha_abuf_t *ab, int64_t rid, UC_Read *ucr, overlap_region_alloc *overlap_list, Candidates_list *cl, double bw_thres, int max_n_chain, int keep_whole_chain)
|
||||
{
|
||||
extern void *ha_flt_tab;
|
||||
extern ha_pt_t *ha_idx;
|
||||
uint32_t i, rlen;
|
||||
uint64_t k, l;
|
||||
double low_occ = asm_opt.hom_cov * HA_KMER_GOOD_RATIO;
|
||||
double high_occ = asm_opt.hom_cov * (2.0 - HA_KMER_GOOD_RATIO);
|
||||
|
||||
// prepare
|
||||
clear_Candidates_list(cl);
|
||||
clear_overlap_region_alloc(overlap_list);
|
||||
recover_UC_Read(ucr, &R_INF, rid);
|
||||
ab->mz.n = 0, ab->n_a = 0;
|
||||
rlen = Get_READ_LENGTH(R_INF, rid); // read length
|
||||
|
||||
// get the list of anchors
|
||||
ha_sketch(ucr->seq, ucr->length, asm_opt.mz_win, asm_opt.k_mer_length, 0, !(asm_opt.flag & HA_F_NO_HPC), &ab->mz, ha_flt_tab);
|
||||
if (ab->mz.m > ab->old_mz_m) {
|
||||
ab->old_mz_m = ab->mz.m;
|
||||
REALLOC(ab->seed, ab->old_mz_m);
|
||||
}
|
||||
for (i = 0, ab->n_a = 0; i < ab->mz.n; ++i) {
|
||||
int n;
|
||||
ab->seed[i].a = ha_pt_get(ha_idx, ab->mz.a[i].x, &n);
|
||||
ab->seed[i].n = n;
|
||||
ab->seed[i].good = (n > low_occ && n < high_occ);
|
||||
ab->n_a += n;
|
||||
}
|
||||
if (ab->n_a > ab->m_a) {
|
||||
ab->m_a = ab->n_a;
|
||||
kroundup64(ab->m_a);
|
||||
REALLOC(ab->a, ab->m_a);
|
||||
}
|
||||
for (i = 0, k = 0; i < ab->mz.n; ++i) {
|
||||
int j;
|
||||
ha_mz1_t *z = &ab->mz.a[i];
|
||||
seed1_t *s = &ab->seed[i];
|
||||
for (j = 0; j < s->n; ++j) {
|
||||
const ha_idxpos_t *y = &s->a[j];
|
||||
anchor1_t *an = &ab->a[k++];
|
||||
uint8_t rev = z->rev == y->rev? 0 : 1;
|
||||
an->other_off = y->pos;
|
||||
an->self_off = rev? ucr->length - 1 - (z->pos + 1 - z->span) : z->pos;
|
||||
an->good = s->good;
|
||||
an->srt = (uint64_t)y->rid<<33 | (uint64_t)rev<<32 | an->other_off;
|
||||
}
|
||||
}
|
||||
|
||||
// sort anchors
|
||||
radix_sort_ha_an1(ab->a, ab->a + ab->n_a);
|
||||
for (k = 1, l = 0; k <= ab->n_a; ++k) {
|
||||
if (k == ab->n_a || ab->a[k].srt != ab->a[l].srt) {
|
||||
if (k - l > 1)
|
||||
radix_sort_ha_an2(ab->a + l, ab->a + k);
|
||||
l = k;
|
||||
}
|
||||
}
|
||||
|
||||
// copy over to _cl_
|
||||
if (ab->m_a >= (uint64_t)cl->size) {
|
||||
cl->size = ab->m_a;
|
||||
REALLOC(cl->list, cl->size);
|
||||
}
|
||||
for (k = 0; k < ab->n_a; ++k) {
|
||||
k_mer_hit *p = &cl->list[k];
|
||||
p->readID = ab->a[k].srt >> 33;
|
||||
p->strand = ab->a[k].srt >> 32 & 1;
|
||||
p->offset = ab->a[k].other_off;
|
||||
p->self_offset = ab->a[k].self_off;
|
||||
p->good = ab->a[k].good;
|
||||
}
|
||||
cl->length = ab->n_a;
|
||||
|
||||
calculate_overlap_region_by_chaining(cl, overlap_list, rid, ucr->length, &R_INF, bw_thres, keep_whole_chain);
|
||||
|
||||
#if 0
|
||||
if (overlap_list->length > 0) {
|
||||
fprintf(stderr, "B\t%ld\t%ld\t%d\n", (long)rid, (long)overlap_list->length, rlen);
|
||||
for (int i = 0; i < (int)overlap_list->length; ++i) {
|
||||
overlap_region *r = &overlap_list->list[i];
|
||||
fprintf(stderr, "C\t%d\t%d\t%d\t%c\t%d\t%ld\t%d\t%d\t%c\t%d\t%d\n", (int)r->x_id, (int)r->x_pos_s, (int)r->x_pos_e, "+-"[r->x_pos_strand],
|
||||
(int)r->y_id, (long)Get_READ_LENGTH(R_INF, r->y_id), (int)r->y_pos_s, (int)r->y_pos_e, "+-"[r->y_pos_strand], (int)r->shared_seed, ha_ov_type(r, rlen));
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
if ((int)overlap_list->length > max_n_chain) {
|
||||
int32_t w, n[4], s[4];
|
||||
n[0] = n[1] = n[2] = n[3] = 0, s[0] = s[1] = s[2] = s[3] = 0;
|
||||
ks_introsort_or_ss(overlap_list->length, overlap_list->list);
|
||||
for (i = 0; i < (uint32_t)overlap_list->length; ++i) {
|
||||
const overlap_region *r = &overlap_list->list[i];
|
||||
w = ha_ov_type(r, rlen);
|
||||
++n[w];
|
||||
if ((int)n[w] == max_n_chain) s[w] = r->shared_seed;
|
||||
}
|
||||
if (s[0] > 0 || s[1] > 0 || s[2] > 0 || s[3] > 0) {
|
||||
for (i = 0, k = 0; i < (uint32_t)overlap_list->length; ++i) {
|
||||
overlap_region *r = &overlap_list->list[i];
|
||||
w = ha_ov_type(r, rlen);
|
||||
if (r->shared_seed >= s[w]) {
|
||||
if ((uint32_t)k != i) {
|
||||
overlap_region t;
|
||||
t = overlap_list->list[k];
|
||||
overlap_list->list[k] = overlap_list->list[i];
|
||||
overlap_list->list[i] = t;
|
||||
}
|
||||
++k;
|
||||
}
|
||||
}
|
||||
overlap_list->length = k;
|
||||
}
|
||||
}
|
||||
|
||||
ks_introsort_or_xs(overlap_list->length, overlap_list->list);
|
||||
}
|
||||
|
||||
|
||||
|
||||
void ha_sort_list_by_anchor(overlap_region_alloc *overlap_list)
|
||||
{
|
||||
ks_introsort_or_xs(overlap_list->length, overlap_list->list);
|
||||
}
|
||||
173
extract.cpp
Normal file
173
extract.cpp
Normal file
@@ -0,0 +1,173 @@
|
||||
#include <zlib.h>
|
||||
#include <string.h>
|
||||
#include "Process_Read.h"
|
||||
#include "khashl.h"
|
||||
#include "kseq.h"
|
||||
|
||||
typedef const char *cstr_t;
|
||||
KHASHL_CSET_INIT(KH_LOCAL, strset_t, ss, cstr_t, kh_hash_str, kh_eq_str)
|
||||
KHASHL_MAP_INIT(KH_LOCAL, hm64_t, h64, uint64_t, int, kh_hash_uint64, kh_eq_generic)
|
||||
KSTREAM_INIT(gzFile, gzread, 65536)
|
||||
|
||||
#define GFA_MALLOC(ptr, len) ((ptr) = (__typeof__(ptr))malloc((len) * sizeof(*(ptr))))
|
||||
#define GFA_REALLOC(ptr, len) ((ptr) = (__typeof__(ptr))realloc((ptr), (len) * sizeof(*(ptr))))
|
||||
|
||||
char *gfa_strdup(const char *src)
|
||||
{
|
||||
int32_t len;
|
||||
char *dst;
|
||||
len = strlen(src);
|
||||
GFA_MALLOC(dst, len + 1);
|
||||
memcpy(dst, src, len + 1);
|
||||
return dst;
|
||||
}
|
||||
|
||||
char *gfa_strndup(const char *src, size_t n)
|
||||
{
|
||||
char *dst;
|
||||
GFA_MALLOC(dst, n + 1);
|
||||
strncpy(dst, src, n);
|
||||
dst[n] = 0;
|
||||
return dst;
|
||||
}
|
||||
|
||||
char **gv_read_list(const char *o, int *n_)
|
||||
{
|
||||
int n = 0, m = 0;
|
||||
char **s = 0;
|
||||
*n_ = 0;
|
||||
if (*o != '@') {
|
||||
const char *q = o, *p;
|
||||
for (p = q;; ++p) {
|
||||
if (*p == ',' || *p == 0) {
|
||||
if (n == m) {
|
||||
m = m? m<<1 : 16;
|
||||
GFA_REALLOC(s, m);
|
||||
}
|
||||
s[n++] = gfa_strndup(q, p - q);
|
||||
if (*p == 0) break;
|
||||
q = p + 1;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
gzFile fp;
|
||||
kstream_t *ks;
|
||||
kstring_t str = {0,0,0};
|
||||
int dret;
|
||||
|
||||
fp = gzopen(o + 1, "r");
|
||||
if (fp == 0) return 0;
|
||||
ks = ks_init(fp);
|
||||
while (ks_getuntil(ks, KS_SEP_LINE, &str, &dret) >= 0) {
|
||||
char *p;
|
||||
for (p = str.s; *p && !isspace(*p); ++p);
|
||||
if (n == m) {
|
||||
m = m? m<<1 : 16;
|
||||
GFA_REALLOC(s, m);
|
||||
}
|
||||
s[n++] = gfa_strndup(str.s, p - str.s);
|
||||
}
|
||||
ks_destroy(ks);
|
||||
gzclose(fp);
|
||||
}
|
||||
if (s) s = (char**)realloc(s, n * sizeof(char*));
|
||||
*n_ = n;
|
||||
return s;
|
||||
}
|
||||
|
||||
void ha_extract_print(const All_reads *rs, int n_rounds, int n, char **list)
|
||||
{
|
||||
hm64_t *h;
|
||||
khint_t k;
|
||||
int i, absent, m, l;
|
||||
uint64_t j;
|
||||
const ma_hit_t_alloc *ov[2] = { rs->paf, rs->reverse_paf };
|
||||
FILE *fp = stdout;
|
||||
|
||||
if (n > 0) {
|
||||
int max_len = 0;
|
||||
char *s = 0;
|
||||
strset_t *ss;
|
||||
ss = ss_init();
|
||||
for (i = 0; i < n; ++i)
|
||||
ss_put(ss, list[i], &absent);
|
||||
for (j = 0; j < rs->total_reads; ++j)
|
||||
if (max_len < (int)Get_NAME_LENGTH(*rs, j))
|
||||
max_len = Get_NAME_LENGTH(*rs, j);
|
||||
GFA_MALLOC(s, max_len + 1);
|
||||
h = h64_init();
|
||||
for (j = 0; j < rs->total_reads; ++j) {
|
||||
strncpy(s, Get_NAME(*rs, j), Get_NAME_LENGTH(*rs, j));
|
||||
s[Get_NAME_LENGTH(*rs, j)] = 0;
|
||||
if (ss_get(ss, s) != kh_end(ss)) {
|
||||
k = h64_put(h, j, &absent);
|
||||
kh_val(h, k) = -1;
|
||||
}
|
||||
}
|
||||
free(s);
|
||||
ss_destroy(ss);
|
||||
} else return;
|
||||
|
||||
for (m = 0; m < n_rounds; ++m) {
|
||||
for (j = 0; j < rs->total_reads; ++j) {
|
||||
for (l = 0; l < 2; ++l) {
|
||||
const ma_hit_t_alloc *o = &ov[l][j];
|
||||
for (i = 0; i < (int)o->length; ++i) {
|
||||
uint64_t q = Get_qn(o->buffer[i]);
|
||||
uint64_t t = Get_tn(o->buffer[i]);
|
||||
int q_hit = 0, t_hit = 0;
|
||||
k = h64_get(h, q);
|
||||
q_hit = (k < kh_end(h) && kh_val(h, k) < m);
|
||||
k = h64_get(h, t);
|
||||
t_hit = (k < kh_end(h) && kh_val(h, k) < m);
|
||||
if ((!q_hit && !t_hit) || (q_hit && t_hit)) continue;
|
||||
if (!q_hit) {
|
||||
k = h64_put(h, q, &absent);
|
||||
if (absent) kh_val(h, k) = m;
|
||||
}
|
||||
if (!t_hit) {
|
||||
k = h64_put(h, t, &absent);
|
||||
if (absent) kh_val(h, k) = m;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (j = 0; j < rs->total_reads; ++j) {
|
||||
for (l = 0; l < 2; ++l) {
|
||||
const ma_hit_t_alloc *o = &ov[l][j];
|
||||
for (i = 0; i < (int)o->length; ++i) {
|
||||
uint64_t q = Get_qn(o->buffer[i]);
|
||||
uint64_t t = Get_tn(o->buffer[i]);
|
||||
int q_hit = 0, t_hit = 0;
|
||||
q_hit = (h64_get(h, q) < kh_end(h));
|
||||
t_hit = (h64_get(h, t) < kh_end(h));
|
||||
if (!q_hit && !t_hit) continue;
|
||||
fwrite(Get_NAME(*rs, q), 1, Get_NAME_LENGTH(*rs, q), fp);
|
||||
fwrite("\t", 1, 1, fp);
|
||||
fprintf(fp, "%lu\t", (unsigned long)Get_READ_LENGTH(*rs, q));
|
||||
fprintf(fp, "%d\t", Get_qs(o->buffer[i]));
|
||||
fprintf(fp, "%d\t", Get_qe(o->buffer[i]));
|
||||
fputs(o->buffer[i].rev? "-\t" : "+\t", fp);
|
||||
fwrite(Get_NAME(*rs, t), 1, Get_NAME_LENGTH(*rs, t), fp);
|
||||
fwrite("\t", 1, 1, fp);
|
||||
fprintf(fp, "%lu\t", (unsigned long)Get_READ_LENGTH(*rs, t));
|
||||
fprintf(fp, "%d\t", Get_ts(o->buffer[i]));
|
||||
fprintf(fp, "%d\t%d\t%d\t%d\n", Get_te(o->buffer[i]), o->buffer[i].ml, o->buffer[i].bl, !l);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
h64_destroy(h);
|
||||
}
|
||||
|
||||
void ha_extract_print_list(const All_reads *rs, int n_rounds, const char *o)
|
||||
{
|
||||
int i, n;
|
||||
char **list;
|
||||
list = gv_read_list(o, &n);
|
||||
ha_extract_print(rs, n_rounds, n, list);
|
||||
for (i = 0; i < n; ++i) free(list[i]);
|
||||
free(list);
|
||||
}
|
||||
341
hifiasm.1
341
hifiasm.1
@@ -1,38 +1,60 @@
|
||||
.TH hifiasm 1 "3 Jan 2020" "hifiasm-0.1.0" "Bioinformatics tools"
|
||||
.TH hifiasm 1 "19 July 2020" "hifiasm-0.9 (r289)" "Bioinformatics tools"
|
||||
|
||||
.SH NAME
|
||||
.PP
|
||||
hifiasm - haplotype-resolved de novo assembler for PacBio Hifi reads.
|
||||
|
||||
.SH SYNOPSIS
|
||||
.PP
|
||||
hifiasm
|
||||
|
||||
* Assemble HiFi reads:
|
||||
.RS 4
|
||||
.B hifiasm
|
||||
.RB [ -o
|
||||
.IR outPrefix ]
|
||||
.IR prefix ]
|
||||
.RB [ -t
|
||||
.IR numThres ]
|
||||
.RB [ -r
|
||||
.IR roundCorrection ]
|
||||
.RB [ -a
|
||||
.IR roundGraphClean ]
|
||||
.IR nThreads ]
|
||||
.RB [ -z
|
||||
.IR endTrimLen ]
|
||||
.R [options]
|
||||
.I input1.fq
|
||||
.RI [ input2.fq
|
||||
.R [...]]
|
||||
.RE
|
||||
|
||||
* Trio binning assembly with yak dumps:
|
||||
.RS 4
|
||||
.B yak count
|
||||
.B -o
|
||||
.I paternal.yak
|
||||
.B -b37
|
||||
.RB [ -t
|
||||
.IR nThreads ]
|
||||
.RB [ -k
|
||||
.IR kmerLen ]
|
||||
.RB [ -z
|
||||
.IR adapterLen ]
|
||||
.RB [ -m
|
||||
.IR maxLargeBubbles ]
|
||||
.RB [ -p
|
||||
.IR maxSmallBubbles ]
|
||||
.RB [ -n
|
||||
.IR maxSmallUnitig ]
|
||||
.RB [ -x
|
||||
.IR maxDropRatio ]
|
||||
.RB [ -y
|
||||
.IR minDropRatio ]
|
||||
.RB [ -i ]
|
||||
.RB [ -v ]
|
||||
.RB [ -h ]
|
||||
.I <in_1.fq> <in_2.fq> <...>
|
||||
.I paternal.fq.gz
|
||||
.br
|
||||
.B yak count
|
||||
.B -o
|
||||
.I maternal.yak
|
||||
.B -b37
|
||||
.RB [ -t
|
||||
.IR nThreads ]
|
||||
.RB [ -k
|
||||
.IR kmerLen ]
|
||||
.I maternal.fq.gz
|
||||
.br
|
||||
.B hifiasm
|
||||
.RB [ -o
|
||||
.IR prefix ]
|
||||
.RB [ -t
|
||||
.IR nThreads ]
|
||||
.R [options]
|
||||
.B -1
|
||||
.I paternal.yak
|
||||
.B -2
|
||||
.I maternal.yak
|
||||
.I child.hifi.fq.gz
|
||||
.RE
|
||||
|
||||
.SH DESCRIPTION
|
||||
.PP
|
||||
@@ -49,49 +71,59 @@ outputs consist of multiple types of assembly graph in GFA format.
|
||||
|
||||
.TP 10
|
||||
.BI -o \ FILE
|
||||
Prefix of output files [hifiasm.asm]. The outputs of hifiasm include error corrected
|
||||
reads in fasta format, all-to-all overlaps in paf format, and four types of assembly
|
||||
graph in GFA format. For detailed description of all assembly graphs, please see
|
||||
.I 'Outputs'
|
||||
Prefix of output files [hifiasm.asm]. For detailed description of all assembly
|
||||
graphs, please see the
|
||||
.B OUTPUTS
|
||||
section of this man-page.
|
||||
|
||||
.TP 10
|
||||
.BI -t \ INT
|
||||
Number of CPU threads used by hifiasm [1].
|
||||
|
||||
.TP
|
||||
.BI -h
|
||||
Show help information.
|
||||
|
||||
.TP 10
|
||||
.BI -v
|
||||
.TP
|
||||
.BI --version
|
||||
Show version number.
|
||||
|
||||
.TP 10
|
||||
.BI -h
|
||||
Show help information.
|
||||
|
||||
.SS Error correction options
|
||||
|
||||
.TP 10
|
||||
.BI -k \ INT
|
||||
K-mer length [40]. This option must be less than 64.
|
||||
K-mer length [51]. This option must be less than 64.
|
||||
|
||||
.TP 10
|
||||
.TP
|
||||
.BI -w \ INT
|
||||
Minimizer window size [51].
|
||||
|
||||
.TP
|
||||
.BI -f \ INT
|
||||
Number of bits for bloom filter; 0 to disable [37]. This bloom filter is used
|
||||
to filter out singleton k-mers when counting all k-mers. It takes
|
||||
.RI 2^( INT -3)
|
||||
bytes of memory. A proper setting saves memory. 37 is recommended for human
|
||||
assembly.
|
||||
|
||||
.TP
|
||||
.BI -r \ INT
|
||||
Rounds of haplotype-aware error corrections [2]. This option affects all outputs of hifiasm.
|
||||
Rounds of haplotype-aware error corrections [3]. This option affects all outputs of hifiasm.
|
||||
|
||||
.SS Assembly options
|
||||
|
||||
.TP 10
|
||||
.TP
|
||||
.BI -a \ INT
|
||||
Rounds of assembly graph cleaning [4]. This option is used with
|
||||
.I [-x maxDropRatio]
|
||||
.B -x
|
||||
and
|
||||
.I [-y minDropRatio].
|
||||
.BR -y .
|
||||
Note that unlike
|
||||
.I [-r],
|
||||
.BR -r ,
|
||||
this option does not affect error corrected reads and all-to-all overlaps.
|
||||
|
||||
|
||||
.TP 10
|
||||
.TP
|
||||
.BI -z \ INT
|
||||
Length of adapters that should be removed [0]. This option remove
|
||||
.I INT
|
||||
@@ -100,40 +132,35 @@ Some old Hifi reads may consist of
|
||||
short adapters (e.g., 20bp adapter at one end). For such data, trimming short adapters would
|
||||
significantly improve the assembly quality.
|
||||
|
||||
|
||||
.TP 10
|
||||
.TP
|
||||
.BI -m \ INT
|
||||
Maximal probing distance for bubble popping when generating primary/alternate assembly
|
||||
Maximal probing distance for bubble popping when generating primary/alternate
|
||||
contig graphs [10000000]. Bubbles longer than
|
||||
.I INT
|
||||
bases will not be popped. For detailed description of these graphs, please see
|
||||
.I 'Outputs'
|
||||
bases will not be popped. For detailed description of these graphs, please see the
|
||||
.B OUTPUTS
|
||||
section of this man-page.
|
||||
|
||||
|
||||
.TP 10
|
||||
.TP
|
||||
.BI -p \ INT
|
||||
Maximal probing distance for bubble popping when generating haplotype-resolved processed unitig graph
|
||||
without small bubbles [100000]. Bubbles longer than
|
||||
.I INT
|
||||
bases will not be popped. Small bubbles might be caused by somatic mutations or noise in data, which
|
||||
are not the real haplotype information. For detailed description of this graph, please see
|
||||
.I 'Outputs'
|
||||
are not the real haplotype information. For detailed description of this graph, please see the
|
||||
.B OUTPUTS
|
||||
section of this man-page.
|
||||
|
||||
|
||||
.TP 10
|
||||
.TP
|
||||
.BI -n \ INT
|
||||
A unitig is considered small if it is composed of less than
|
||||
.I INT
|
||||
reads [3]. Hifiasm may try to remove small unitigs at various steps.
|
||||
|
||||
|
||||
|
||||
.TP 10
|
||||
.TP
|
||||
.BI -x \ FLOAT, -y \ FLOAT
|
||||
Max and min overlap drop ratio [0.8, 0.2]. This option is used with
|
||||
.I [-r roundCorrection].
|
||||
.BR -r .
|
||||
Given a node
|
||||
.I N
|
||||
in the assembly graph, let max(N)
|
||||
@@ -143,19 +170,21 @@ Hifiasm iteratively drops overlaps of
|
||||
.I N
|
||||
if their length / max(N)
|
||||
are below a threshold controlled by
|
||||
.I [-x maxDropRatio]
|
||||
.B -x
|
||||
and
|
||||
.I [-y minDropRatio].
|
||||
.BR -y .
|
||||
Hifiasm applies
|
||||
.I [-r roundCorrection]
|
||||
.B -r
|
||||
rounds of short overlap removal with an increasing threshold between
|
||||
.I [-x maxDropRatio]
|
||||
.B -x
|
||||
and
|
||||
.I [-y minDropRatio].
|
||||
.BR -y .
|
||||
|
||||
.TP 10
|
||||
.TP
|
||||
.BI -i
|
||||
Ignore saved overlaps in [*.ovlp*] files.
|
||||
Ignore error corrected reads and overlaps saved in
|
||||
.IR prefix .*.bin
|
||||
files.
|
||||
Apart from assembly graphs, hifiasm also outputs three binary files
|
||||
that save all overlap information during assembly step.
|
||||
With these files, hifiasm can avoid the time-consuming all-to-all overlap calculation step,
|
||||
@@ -163,71 +192,145 @@ and do the assembly directly and quickly.
|
||||
This might be helpful when users want to get an optimized assembly by multiple rounds of experiments
|
||||
with different parameters.
|
||||
|
||||
.TP
|
||||
.BI --pri-range \ INT1[,INT2]
|
||||
Min and max coverage cutoff of primary contigs.
|
||||
Keep contigs with coverage in this range at p_ctg.gfa.
|
||||
Inferred automatically in default.
|
||||
If INT2 is not specified, it is set to infinity.
|
||||
Set -1 to disable
|
||||
|
||||
.SH EXAMPLES
|
||||
.SS Trio-partition options
|
||||
|
||||
.TP 10
|
||||
.BI -1 \ FILE
|
||||
K-mer dump generated by
|
||||
.B yak count
|
||||
from the paternal/haplotype1 reads []
|
||||
|
||||
.TP
|
||||
.BR ./hifiasm " " \-o " " NA12878.asm " " \-t " " 32 " " NA12878_1.fq.gz " " NA12878_2.fq.gz
|
||||
In this example, hifiasm will be run with 32 CPU threads. The input read files are [NA12878_1.fq.gz]
|
||||
and [NA12878_2.fq.gz],
|
||||
while all output files can be found at [NA12878.asm.*].
|
||||
.BI -2 \ FILE
|
||||
K-mer dump generated by
|
||||
.B yak count
|
||||
from the maternal/haplotype2 reads []
|
||||
|
||||
.TP
|
||||
.BR ./hifiasm " " \-o " " butterfly.asm " " \-t " " 32 " " \-z " " 20 " " butterfly.fq.gz
|
||||
In this example, hifiasm will be run with 32 CPU threads. The input read file is [butterfly.fq.gz],
|
||||
while all output files can be found at [butterfly.asm.*].
|
||||
With
|
||||
.I [-z 20],
|
||||
hifiasm will remove 20 bases from both ends of each read.
|
||||
.BI -3 \ FILE
|
||||
List of paternal/haplotype1 read names []
|
||||
|
||||
.TP
|
||||
.BI -4 \ FILE
|
||||
List of maternal/haplotype2 read names []
|
||||
|
||||
.TP
|
||||
.BI -c \ INT
|
||||
Lower bound of the binned k-mer's frequency [2]. When doing trio binning,
|
||||
a k-mer is said to be differentiating if it occurs >=
|
||||
.B -d
|
||||
times in one sample
|
||||
but occurs <
|
||||
.B -c
|
||||
times in the other sample.
|
||||
|
||||
.TP
|
||||
.BI -d \ INT
|
||||
Upper bound of the binned k-mer's frequency [5]. When doing trio binning,
|
||||
a k-mer is said to be differentiating if it occurs >=
|
||||
.B -d
|
||||
times in one sample
|
||||
but occurs <
|
||||
.B -c
|
||||
times in the other sample.
|
||||
|
||||
|
||||
.SS Purge-dups options
|
||||
|
||||
.TP 10
|
||||
.BI -l \ INT
|
||||
Level of purge-dup. 0 to disable purge-dup, 1 to only purge contained haplotigs,
|
||||
2 to purge all types of haplotigs. In default, [2] for non-trio assembly, [0] for trio assembly.
|
||||
For trio assembly, only level 0 and level 1 are allowed.
|
||||
|
||||
.TP
|
||||
.BI -s \ FLOAT
|
||||
Similarity threshold for duplicate haplotigs that should be purged [0.75].
|
||||
|
||||
.TP
|
||||
.BI -O \ FLOAT
|
||||
Min number of overlapped reads for duplicate haplotigs that should be purged [1].
|
||||
|
||||
.TP
|
||||
.BI --purge-cov \ INT
|
||||
Coverage upper bound of Purge-dups, which is inferred automatically in default.
|
||||
If the coverage of a contig is higher than this bound, don't apply Purge-dups.
|
||||
|
||||
.SS Debugging options
|
||||
|
||||
.TP 10
|
||||
.B --dbg-gfa
|
||||
Write additional files to speed up the debugging of graph cleaning.
|
||||
|
||||
|
||||
.SH OUTPUTS
|
||||
|
||||
.PP
|
||||
Without trio partition options
|
||||
.B -1
|
||||
and
|
||||
.BR -2 ,
|
||||
hifiasm generates the following assembly graphs in the GFA format:
|
||||
|
||||
.RS 2
|
||||
.TP 2
|
||||
*
|
||||
.IR prefix .r_utg.gfa:
|
||||
haplotype-resolved raw unitig graph. This graph keeps all haplotype information.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .p_utg.gfa:
|
||||
haplotype-resolved processed unitig graph without small bubbles. Small bubbles
|
||||
might be caused by somatic mutations or noise in data, which are not the real
|
||||
haplotype information. The size of popped small bubbles should be specified by
|
||||
.BR -p .
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .p_ctg.gfa:
|
||||
assembly graph of primary contigs. This graph collapses different haplotypes.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .a_ctg.gfa:
|
||||
assembly graph of alternate contigs. This graph consists of all assemblies that
|
||||
are discarded in primary contig graph.
|
||||
|
||||
.RE
|
||||
|
||||
.PP
|
||||
Consider the prefix of output files has been specified by
|
||||
.I [-o outPrefix].
|
||||
During the error correction step, hifiasm outputs the following two files:
|
||||
With trio partition, hifiasm outputs the following assembly graphs:
|
||||
|
||||
.IP
|
||||
1. Haplotype-aware error corrected reads in fasta format [outPrefix.ec.fa].
|
||||
.RS 2
|
||||
.TP 2
|
||||
*
|
||||
.IR prefix .dip.r_utg.gfa:
|
||||
haplotype-resolved raw unitig graph. This graph keeps all haplotype information.
|
||||
|
||||
2. All-to-all overlaps in paf format [outPrefix.ovlp.paf].
|
||||
.TP
|
||||
*
|
||||
.IR prefix .hap1.p_ctg.gfa:
|
||||
phased paternal/haplotype1 contig graph. This graph keeps the phased
|
||||
paternal/haplotype1 assembly.
|
||||
|
||||
.TP
|
||||
*
|
||||
.IR prefix .hap2.p_ctg.gfa:
|
||||
phased maternal/haplotype2 contig graph. This graph keeps the phased
|
||||
maternal/haplotype2 assembly.
|
||||
.RE
|
||||
|
||||
.PP
|
||||
During the assembly step, hifiasm outputs the following four assembly graphs in GFA format:
|
||||
|
||||
|
||||
.IP
|
||||
1. Haplotype-resolved raw unitig graph [outPrefix.r_utg.gfa].
|
||||
This graph keeps all haplotype information.
|
||||
|
||||
|
||||
2. Haplotype-resolved processed unitig graph without small bubbles [outPrefix.p_utg.gfa].
|
||||
Small bubbles might be caused by somatic mutations or noise in data, which are not the real haplotype information.
|
||||
The size of popped small bubbles should be specified by
|
||||
.I [-p maxSmallBubbles].
|
||||
|
||||
|
||||
3. Primary assembly contig graph [outPrefix.p_ctg.gfa].
|
||||
This graph collapses different haplotypes.
|
||||
|
||||
4. Alternate assembly contig graph [outPrefix.a_ctg.gfa].
|
||||
This graph consists of all assemblies that are discarded in primary assembly contig graph.
|
||||
|
||||
.PP
|
||||
For each graph, hifiasm also outputs a simplified version without sequences. These simplified
|
||||
graphs can be easily visualized.
|
||||
|
||||
.PP
|
||||
Note that different species need different assembly graphs. For homozygous genomes,
|
||||
the primary assembly contig graph is the best choice.
|
||||
For species with high heterozygous rate, different haplotypes can be fully separated.
|
||||
It is important to remove small bubbles from the haplotype-resolved unitig graph. The
|
||||
reason is that some small bubbles are caused by somatic mutations or noise in data,
|
||||
which are not the real haplotype information. In this case, haplotype-resolved processed
|
||||
unitig graph without small bubbles should be better.
|
||||
For ordinary human genome, different haplotypes cannot be fully separated due to the low
|
||||
heterozygous rate. There are many small bubbles including haplotype information,
|
||||
which cannot be simply removed. Thus, it is necessary to use the haplotype-resolved raw
|
||||
unitig graph.
|
||||
|
||||
For each graph, hifiasm also outputs a simplified version without sequences for
|
||||
the ease of visualization. Hifiasm keeps corrected reads and overlaps in three
|
||||
binary files such as it can regenerate assembly graphs from the binary files
|
||||
without redoing error correction.
|
||||
|
||||
92
hist.cpp
Normal file
92
hist.cpp
Normal file
@@ -0,0 +1,92 @@
|
||||
#include <stdio.h>
|
||||
#include "htab.h"
|
||||
|
||||
static void ha_hist_line(int c, int x, int exceed, int64_t cnt)
|
||||
{
|
||||
int j;
|
||||
if (c >= 0) fprintf(stderr, "[M::%s] %5d: ", __func__, c);
|
||||
else fprintf(stderr, "[M::%s] %5s: ", __func__, "rest");
|
||||
for (j = 0; j < x; ++j) fputc('*', stderr);
|
||||
if (exceed) fputc('>', stderr);
|
||||
fprintf(stderr, " %lld\n", (long long)cnt);
|
||||
}
|
||||
|
||||
int ha_analyze_count(int n_cnt, const int64_t *cnt, int *peak_het)
|
||||
{
|
||||
const int hist_max = 100;
|
||||
int i, start, low_i, max_i, max2_i, max3_i;
|
||||
int64_t max, max2, max3, min;
|
||||
|
||||
// find the low point from the left
|
||||
*peak_het = -1;
|
||||
start = cnt[1] > 0? 1 : 2;
|
||||
low_i = start;
|
||||
for (i = low_i + 1; i < n_cnt; ++i)
|
||||
if (cnt[i] > cnt[i-1]) break;
|
||||
low_i = i - 1;
|
||||
fprintf(stderr, "[M::%s] lowest: count[%d] = %ld\n", __func__, low_i, (long)cnt[low_i]);
|
||||
if (low_i == n_cnt - 1) return -1; // low coverage
|
||||
|
||||
// find the highest peak
|
||||
max_i = low_i + 1, max = cnt[max_i];
|
||||
for (i = low_i + 1; i < n_cnt; ++i)
|
||||
if (cnt[i] > max)
|
||||
max = cnt[i], max_i = i;
|
||||
fprintf(stderr, "[M::%s] highest: count[%d] = %ld\n", __func__, max_i, (long)cnt[max_i]);
|
||||
|
||||
// print histogram
|
||||
for (i = start; i < n_cnt; ++i) {
|
||||
int x, exceed = 0;
|
||||
x = (int)((double)hist_max * cnt[i] / cnt[max_i] + .499);
|
||||
if (x > hist_max) exceed = 1, x = hist_max; // may happen if cnt[2] is higher
|
||||
if (i > max_i && x == 0) break;
|
||||
ha_hist_line(i, x, exceed, cnt[i]);
|
||||
}
|
||||
{
|
||||
int x, exceed = 0;
|
||||
int64_t rest = 0;
|
||||
for (; i < n_cnt; ++i) rest += cnt[i];
|
||||
x = (int)((double)hist_max * rest / cnt[max_i] + .499);
|
||||
if (x > hist_max) exceed = 1, x = hist_max;
|
||||
ha_hist_line(-1, x, exceed, rest);
|
||||
}
|
||||
|
||||
// look for smaller peak on the low end
|
||||
max2 = -1; max2_i = -1;
|
||||
for (i = max_i - 1; i > low_i; --i) {
|
||||
if (cnt[i] >= cnt[i-1] && cnt[i] >= cnt[i+1]) {
|
||||
if (cnt[i] > max2) max2 = cnt[i], max2_i = i;
|
||||
}
|
||||
}
|
||||
if (max2_i > low_i && max2_i < max_i) {
|
||||
for (i = max2_i + 1, min = max; i < max_i; ++i)
|
||||
if (cnt[i] < min) min = cnt[i];
|
||||
if (max2 < max * 0.05 || min > max2 * 0.95)
|
||||
max2 = -1, max2_i = -1;
|
||||
}
|
||||
if (max2 > 0) fprintf(stderr, "[M::%s] left: count[%d] = %ld\n", __func__, max2_i, (long)cnt[max2_i]);
|
||||
else fprintf(stderr, "[M::%s] left: none\n", __func__);
|
||||
|
||||
// look for smaller peak on the high end
|
||||
max3 = -1; max3_i = -1;
|
||||
for (i = max_i + 1; i < n_cnt - 1; ++i) {
|
||||
if (cnt[i] >= cnt[i-1] && cnt[i] >= cnt[i+1]) {
|
||||
if (cnt[i] > max3) max3 = cnt[i], max3_i = i;
|
||||
}
|
||||
}
|
||||
if (max3_i > max_i) {
|
||||
for (i = max_i + 1, min = max; i < max3_i; ++i)
|
||||
if (cnt[i] < min) min = cnt[i];
|
||||
if (max3 < max * 0.05 || min > max3 * 0.95 || max3_i > max_i * 2.5)
|
||||
max3 = -1, max3_i = -1;
|
||||
}
|
||||
if (max3 > 0) fprintf(stderr, "[M::%s] right: count[%d] = %ld\n", __func__, max3_i, (long)cnt[max3_i]);
|
||||
else fprintf(stderr, "[M::%s] right: none\n", __func__);
|
||||
if (max3_i > 0) {
|
||||
*peak_het = max_i;
|
||||
return max3_i;
|
||||
} else {
|
||||
if (max2_i > 0) *peak_het = max2_i;
|
||||
return max_i;
|
||||
}
|
||||
}
|
||||
839
htab.cpp
Normal file
839
htab.cpp
Normal file
@@ -0,0 +1,839 @@
|
||||
#include <stdint.h>
|
||||
#include <zlib.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <stdlib.h>
|
||||
#include <assert.h>
|
||||
#include "kthread.h"
|
||||
#include "khashl.h"
|
||||
#include "kseq.h"
|
||||
#include "ksort.h"
|
||||
#include "htab.h"
|
||||
|
||||
#define YAK_COUNTER_BITS 12
|
||||
#define YAK_N_COUNTS (1<<YAK_COUNTER_BITS)
|
||||
#define YAK_MAX_COUNT ((1<<YAK_COUNTER_BITS)-1)
|
||||
|
||||
const unsigned char seq_nt4_table[256] = { // translate ACGT to 0123
|
||||
0, 1, 2, 3, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 0, 4, 1, 4, 4, 4, 2, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 3, 3, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 0, 4, 1, 4, 4, 4, 2, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 3, 3, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4,
|
||||
4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4, 4
|
||||
};
|
||||
|
||||
void *ha_flt_tab;
|
||||
ha_pt_t *ha_idx;
|
||||
|
||||
/***************************
|
||||
* Yak specific parameters *
|
||||
***************************/
|
||||
|
||||
typedef struct {
|
||||
int32_t bf_shift, bf_n_hash;
|
||||
int32_t k, w, is_HPC;
|
||||
int32_t pre;
|
||||
int32_t n_thread;
|
||||
int64_t chunk_size;
|
||||
} yak_copt_t;
|
||||
|
||||
void yak_copt_init(yak_copt_t *o)
|
||||
{
|
||||
memset(o, 0, sizeof(yak_copt_t));
|
||||
o->bf_shift = 0;
|
||||
o->bf_n_hash = 4;
|
||||
o->k = 31;
|
||||
o->w = 1;
|
||||
o->pre = YAK_COUNTER_BITS;
|
||||
o->n_thread = 4;
|
||||
o->chunk_size = 20000000;
|
||||
}
|
||||
|
||||
/************************
|
||||
* Blocked bloom filter *
|
||||
************************/
|
||||
|
||||
#define YAK_BLK_SHIFT 9 // 64 bytes, the size of a cache line
|
||||
#define YAK_BLK_MASK ((1<<(YAK_BLK_SHIFT)) - 1)
|
||||
|
||||
typedef struct {
|
||||
int n_shift, n_hashes;
|
||||
uint8_t *b;
|
||||
} yak_bf_t;
|
||||
|
||||
yak_bf_t *yak_bf_init(int n_shift, int n_hashes)
|
||||
{
|
||||
yak_bf_t *b;
|
||||
void *ptr = 0;
|
||||
if (n_shift + YAK_BLK_SHIFT > 64 || n_shift < YAK_BLK_SHIFT) return 0;
|
||||
CALLOC(b, 1);
|
||||
b->n_shift = n_shift;
|
||||
b->n_hashes = n_hashes;
|
||||
posix_memalign(&ptr, 1<<(YAK_BLK_SHIFT-3), 1ULL<<(n_shift-3));
|
||||
b->b = (uint8_t*)ptr;
|
||||
bzero(b->b, 1ULL<<(n_shift-3));
|
||||
return b;
|
||||
}
|
||||
|
||||
void yak_bf_destroy(yak_bf_t *b)
|
||||
{
|
||||
if (b == 0) return;
|
||||
free(b->b); free(b);
|
||||
}
|
||||
|
||||
int yak_bf_insert(yak_bf_t *b, uint64_t hash)
|
||||
{
|
||||
int x = b->n_shift - YAK_BLK_SHIFT;
|
||||
uint64_t y = hash & ((1ULL<<x) - 1);
|
||||
int h1 = hash >> x & YAK_BLK_MASK;
|
||||
int h2 = hash >> b->n_shift & YAK_BLK_MASK;
|
||||
uint8_t *p = &b->b[y<<(YAK_BLK_SHIFT-3)];
|
||||
int i, z = h1, cnt = 0;
|
||||
if ((h2&31) == 0) h2 = (h2 + 1) & YAK_BLK_MASK; // otherwise we may repeatedly use a few bits
|
||||
for (i = 0; i < b->n_hashes; z = (z + h2) & YAK_BLK_MASK) {
|
||||
uint8_t *q = &p[z>>3], u;
|
||||
u = 1<<(z&7);
|
||||
cnt += !!(*q & u);
|
||||
*q |= u;
|
||||
++i;
|
||||
}
|
||||
return cnt;
|
||||
}
|
||||
|
||||
/********************
|
||||
* Count hash table *
|
||||
********************/
|
||||
|
||||
#define yak_ct_eq(a, b) ((a)>>YAK_COUNTER_BITS == (b)>>YAK_COUNTER_BITS) // lower 8 bits for counts; higher bits for k-mer
|
||||
#define yak_ct_hash(a) ((a)>>YAK_COUNTER_BITS)
|
||||
KHASHL_SET_INIT(static klib_unused, yak_ct_t, yak_ct, uint64_t, yak_ct_hash, yak_ct_eq)
|
||||
|
||||
typedef struct {
|
||||
yak_ct_t *h;
|
||||
yak_bf_t *b;
|
||||
} ha_ct1_t;
|
||||
|
||||
typedef struct {
|
||||
int k, pre, n_hash, n_shift;
|
||||
uint64_t tot;
|
||||
ha_ct1_t *h;
|
||||
} ha_ct_t;
|
||||
|
||||
static ha_ct_t *ha_ct_init(int k, int pre, int n_hash, int n_shift)
|
||||
{
|
||||
ha_ct_t *h;
|
||||
int i;
|
||||
if (pre < YAK_COUNTER_BITS) return 0;
|
||||
CALLOC(h, 1);
|
||||
h->k = k, h->pre = pre;
|
||||
CALLOC(h->h, 1<<h->pre);
|
||||
for (i = 0; i < 1<<h->pre; ++i)
|
||||
h->h[i].h = yak_ct_init();
|
||||
if (n_hash > 0 && n_shift > h->pre) {
|
||||
h->n_hash = n_hash, h->n_shift = n_shift;
|
||||
for (i = 0; i < 1<<h->pre; ++i)
|
||||
h->h[i].b = yak_bf_init(h->n_shift - h->pre, h->n_hash);
|
||||
}
|
||||
return h;
|
||||
}
|
||||
|
||||
static void ha_ct_destroy_bf(ha_ct_t *h)
|
||||
{
|
||||
int i;
|
||||
for (i = 0; i < 1<<h->pre; ++i) {
|
||||
if (h->h[i].b)
|
||||
yak_bf_destroy(h->h[i].b);
|
||||
h->h[i].b = 0;
|
||||
}
|
||||
}
|
||||
|
||||
static void ha_ct_destroy(ha_ct_t *h)
|
||||
{
|
||||
int i;
|
||||
if (h == 0) return;
|
||||
ha_ct_destroy_bf(h);
|
||||
for (i = 0; i < 1<<h->pre; ++i)
|
||||
yak_ct_destroy(h->h[i].h);
|
||||
free(h->h); free(h);
|
||||
}
|
||||
|
||||
static int ha_ct_insert_list(ha_ct_t *h, int create_new, int n, const uint64_t *a)
|
||||
{
|
||||
int j, mask = (1<<h->pre) - 1, n_ins = 0;
|
||||
ha_ct1_t *g;
|
||||
if (n == 0) return 0;
|
||||
g = &h->h[a[0]&mask];
|
||||
for (j = 0; j < n; ++j) {
|
||||
int ins = 1, absent;
|
||||
uint64_t x = a[j] >> h->pre;
|
||||
khint_t k;
|
||||
if ((a[j]&mask) != (a[0]&mask)) continue;
|
||||
if (create_new) {
|
||||
if (g->b)
|
||||
ins = (yak_bf_insert(g->b, x) == h->n_hash);
|
||||
if (ins) {
|
||||
k = yak_ct_put(g->h, x << YAK_COUNTER_BITS | (g->b? 1 : 0), &absent);
|
||||
if (absent) ++n_ins;
|
||||
if ((kh_key(g->h, k)&YAK_MAX_COUNT) < YAK_MAX_COUNT)
|
||||
++kh_key(g->h, k);
|
||||
}
|
||||
} else {
|
||||
k = yak_ct_get(g->h, x<<YAK_COUNTER_BITS);
|
||||
if (k != kh_end(g->h) && (kh_key(g->h, k)&YAK_MAX_COUNT) < YAK_MAX_COUNT)
|
||||
++kh_key(g->h, k);
|
||||
}
|
||||
}
|
||||
return n_ins;
|
||||
}
|
||||
|
||||
/*** generate histogram ***/
|
||||
|
||||
typedef struct {
|
||||
uint64_t c[YAK_N_COUNTS];
|
||||
} buf_cnt_t;
|
||||
|
||||
typedef struct {
|
||||
const ha_ct_t *h;
|
||||
buf_cnt_t *cnt;
|
||||
} hist_aux_t;
|
||||
|
||||
static void worker_ct_hist(void *data, long i, int tid) // callback for kt_for()
|
||||
{
|
||||
hist_aux_t *a = (hist_aux_t*)data;
|
||||
uint64_t *cnt = a->cnt[tid].c;
|
||||
yak_ct_t *g = a->h->h[i].h;
|
||||
khint_t k;
|
||||
for (k = 0; k < kh_end(g); ++k)
|
||||
if (kh_exist(g, k))
|
||||
++cnt[kh_key(g, k)&YAK_MAX_COUNT];
|
||||
}
|
||||
|
||||
static void ha_ct_hist(const ha_ct_t *h, int64_t cnt[YAK_N_COUNTS], int n_thread)
|
||||
{
|
||||
hist_aux_t a;
|
||||
int i, j;
|
||||
a.h = h;
|
||||
memset(cnt, 0, YAK_N_COUNTS * sizeof(uint64_t));
|
||||
CALLOC(a.cnt, n_thread);
|
||||
kt_for(n_thread, worker_ct_hist, &a, 1<<h->pre);
|
||||
for (i = 0; i < YAK_N_COUNTS; ++i) cnt[i] = 0;
|
||||
for (j = 0; j < n_thread; ++j)
|
||||
for (i = 0; i < YAK_N_COUNTS; ++i)
|
||||
cnt[i] += a.cnt[j].c[i];
|
||||
free(a.cnt);
|
||||
}
|
||||
|
||||
/*** shrink a hash table ***/
|
||||
|
||||
typedef struct {
|
||||
int min, max;
|
||||
ha_ct_t *h;
|
||||
} shrink_aux_t;
|
||||
|
||||
static void worker_ct_shrink(void *data, long i, int tid) // callback for kt_for()
|
||||
{
|
||||
shrink_aux_t *a = (shrink_aux_t*)data;
|
||||
ha_ct_t *h = a->h;
|
||||
yak_ct_t *g = h->h[i].h, *f;
|
||||
khint_t k;
|
||||
f = yak_ct_init();
|
||||
yak_ct_resize(f, kh_size(g));
|
||||
for (k = 0; k < kh_end(g); ++k) {
|
||||
if (kh_exist(g, k)) {
|
||||
int absent, c = kh_key(g, k) & YAK_MAX_COUNT;
|
||||
if (c >= a->min && c <= a->max)
|
||||
yak_ct_put(f, kh_key(g, k), &absent);
|
||||
}
|
||||
}
|
||||
yak_ct_destroy(g);
|
||||
h->h[i].h = f;
|
||||
}
|
||||
|
||||
static void ha_ct_shrink(ha_ct_t *h, int min, int max, int n_thread)
|
||||
{
|
||||
int i;
|
||||
shrink_aux_t a;
|
||||
a.h = h, a.min = min, a.max = max;
|
||||
kt_for(n_thread, worker_ct_shrink, &a, 1<<h->pre);
|
||||
for (i = 0, h->tot = 0; i < 1<<h->pre; ++i)
|
||||
h->tot += kh_size(h->h[i].h);
|
||||
}
|
||||
|
||||
/***********************
|
||||
* Position hash table *
|
||||
***********************/
|
||||
|
||||
KHASHL_MAP_INIT(static klib_unused, yak_pt_t, yak_pt, uint64_t, uint64_t, yak_ct_hash, yak_ct_eq)
|
||||
#define generic_key(x) (x)
|
||||
KRADIX_SORT_INIT(ha64, uint64_t, generic_key, 8)
|
||||
|
||||
typedef struct {
|
||||
yak_pt_t *h;
|
||||
uint64_t n;
|
||||
ha_idxpos_t *a;
|
||||
} ha_pt1_t;
|
||||
|
||||
struct ha_pt_s {
|
||||
int k, pre;
|
||||
uint64_t tot, tot_pos;
|
||||
ha_pt1_t *h;
|
||||
};
|
||||
|
||||
typedef struct {
|
||||
const ha_ct_t *ct;
|
||||
ha_pt_t *pt;
|
||||
} pt_gen_aux_t;
|
||||
|
||||
static void worker_pt_gen(void *data, long i, int tid) // callback for kt_for()
|
||||
{
|
||||
pt_gen_aux_t *a = (pt_gen_aux_t*)data;
|
||||
ha_pt1_t *b = &a->pt->h[i];
|
||||
yak_ct_t *g = a->ct->h[i].h;
|
||||
khint_t k;
|
||||
for (k = 0, b->n = 0; k != kh_end(g); ++k) {
|
||||
if (kh_exist(g, k)) {
|
||||
int absent;
|
||||
khint_t l;
|
||||
l = yak_pt_put(b->h, kh_key(g, k) >> a->ct->pre << YAK_COUNTER_BITS, &absent);
|
||||
kh_val(b->h, l) = b->n;
|
||||
b->n += kh_key(g, k) & YAK_MAX_COUNT;
|
||||
}
|
||||
}
|
||||
yak_ct_destroy(g);
|
||||
a->ct->h[i].h = 0;
|
||||
CALLOC(b->a, b->n);
|
||||
}
|
||||
|
||||
ha_pt_t *ha_pt_gen(ha_ct_t *ct, int n_thread)
|
||||
{
|
||||
pt_gen_aux_t a;
|
||||
int i;
|
||||
ha_pt_t *pt;
|
||||
ha_ct_destroy_bf(ct);
|
||||
CALLOC(pt, 1);
|
||||
pt->k = ct->k, pt->pre = ct->pre, pt->tot = ct->tot;
|
||||
CALLOC(pt->h, 1<<pt->pre);
|
||||
for (i = 0; i < 1<<pt->pre; ++i) {
|
||||
pt->h[i].h = yak_pt_init();
|
||||
yak_pt_resize(pt->h[i].h, kh_size(ct->h[i].h));
|
||||
}
|
||||
a.ct = ct, a.pt = pt;
|
||||
kt_for(n_thread, worker_pt_gen, &a, 1<<pt->pre);
|
||||
free(ct->h); free(ct);
|
||||
return pt;
|
||||
}
|
||||
|
||||
int ha_pt_insert_list(ha_pt_t *h, int n, const ha_mz1_t *a)
|
||||
{
|
||||
int j, mask = (1<<h->pre) - 1, n_ins = 0;
|
||||
ha_pt1_t *g;
|
||||
if (n == 0) return 0;
|
||||
g = &h->h[a[0].x&mask];
|
||||
for (j = 0; j < n; ++j) {
|
||||
uint64_t x = a[j].x >> h->pre;
|
||||
khint_t k;
|
||||
int n;
|
||||
ha_idxpos_t *p;
|
||||
if ((a[j].x&mask) != (a[0].x&mask)) continue;
|
||||
k = yak_pt_get(g->h, x<<YAK_COUNTER_BITS);
|
||||
if (k == kh_end(g->h)) continue;
|
||||
n = kh_key(g->h, k) & YAK_MAX_COUNT;
|
||||
assert(n < YAK_MAX_COUNT);
|
||||
p = &g->a[kh_val(g->h, k) + n];
|
||||
p->rid = a[j].rid, p->rev = a[j].rev, p->pos = a[j].pos, p->span = a[j].span;
|
||||
//(uint64_t)a[j].rid<<36 | (uint64_t)a[j].rev<<35 | (uint64_t)a[j].pos<<8 | (uint64_t)a[j].span;
|
||||
++kh_key(g->h, k);
|
||||
++n_ins;
|
||||
}
|
||||
return n_ins;
|
||||
}
|
||||
/*
|
||||
static void worker_pt_sort(void *data, long i, int tid)
|
||||
{
|
||||
ha_pt_t *h = (ha_pt_t*)data;
|
||||
ha_pt1_t *g = &h->h[i];
|
||||
khint_t k;
|
||||
for (k = 0; k < kh_end(g->h); ++k) {
|
||||
int n;
|
||||
uint64_t *p;
|
||||
if (!kh_exist(g->h, k)) continue;
|
||||
n = kh_key(g->h, k) & YAK_MAX_COUNT;
|
||||
p = &g->a[kh_val(g->h, k)];
|
||||
radix_sort_ha64(p, p + n);
|
||||
}
|
||||
}
|
||||
|
||||
void ha_pt_sort(ha_pt_t *h, int n_thread)
|
||||
{
|
||||
kt_for(n_thread, worker_pt_sort, h, 1<<h->pre);
|
||||
}
|
||||
*/
|
||||
void ha_pt_destroy(ha_pt_t *h)
|
||||
{
|
||||
int i;
|
||||
if (h == 0) return;
|
||||
for (i = 0; i < 1<<h->pre; ++i) {
|
||||
yak_pt_destroy(h->h[i].h);
|
||||
free(h->h[i].a);
|
||||
}
|
||||
free(h->h); free(h);
|
||||
}
|
||||
|
||||
const ha_idxpos_t *ha_pt_get(const ha_pt_t *h, uint64_t hash, int *n)
|
||||
{
|
||||
khint_t k;
|
||||
const ha_pt1_t *g = &h->h[hash & ((1ULL<<h->pre) - 1)];
|
||||
*n = 0;
|
||||
k = yak_pt_get(g->h, hash >> h->pre << YAK_COUNTER_BITS);
|
||||
if (k == kh_end(g->h)) return 0;
|
||||
*n = kh_key(g->h, k) & YAK_MAX_COUNT;
|
||||
return &g->a[kh_val(g->h, k)];
|
||||
}
|
||||
|
||||
/**********************************
|
||||
* Buffer for counting all k-mers *
|
||||
**********************************/
|
||||
|
||||
typedef struct {
|
||||
int n, m;
|
||||
uint64_t n_ins;
|
||||
uint64_t *a;
|
||||
ha_mz1_t *b;
|
||||
} ch_buf_t;
|
||||
|
||||
static inline void ct_insert_buf(ch_buf_t *buf, int p, uint64_t y) // insert a k-mer $y to a linear buffer
|
||||
{
|
||||
int pre = y & ((1<<p) - 1);
|
||||
ch_buf_t *b = &buf[pre];
|
||||
if (b->n == b->m) {
|
||||
b->m = b->m < 8? 8 : b->m + (b->m>>1);
|
||||
REALLOC(b->a, b->m);
|
||||
}
|
||||
b->a[b->n++] = y;
|
||||
}
|
||||
|
||||
static inline void pt_insert_buf(ch_buf_t *buf, int p, const ha_mz1_t *y)
|
||||
{
|
||||
int pre = y->x & ((1<<p) - 1);
|
||||
ch_buf_t *b = &buf[pre];
|
||||
if (b->n == b->m) {
|
||||
b->m = b->m < 8? 8 : b->m + (b->m>>1);
|
||||
REALLOC(b->b, b->m);
|
||||
}
|
||||
b->b[b->n++] = *y;
|
||||
}
|
||||
|
||||
static void count_seq_buf(ch_buf_t *buf, int k, int p, int len, const char *seq) // insert k-mers in $seq to linear buffer $buf
|
||||
{
|
||||
int i, l;
|
||||
uint64_t x[4], mask = (1ULL<<k) - 1, shift = k - 1;
|
||||
for (i = l = 0, x[0] = x[1] = x[2] = x[3] = 0; i < len; ++i) {
|
||||
int c = seq_nt4_table[(uint8_t)seq[i]];
|
||||
if (c < 4) { // not an "N" base
|
||||
x[0] = (x[0] << 1 | (c&1)) & mask;
|
||||
x[1] = (x[1] << 1 | (c>>1)) & mask;
|
||||
x[2] = x[2] >> 1 | (uint64_t)(1 - (c&1)) << shift;
|
||||
x[3] = x[3] >> 1 | (uint64_t)(1 - (c>>1)) << shift;
|
||||
if (++l >= k)
|
||||
ct_insert_buf(buf, p, yak_hash_long(x));
|
||||
} else l = 0, x[0] = x[1] = x[2] = x[3] = 0; // if there is an "N", restart
|
||||
}
|
||||
}
|
||||
|
||||
static void count_seq_buf_HPC(ch_buf_t *buf, int k, int p, int len, const char *seq) // insert k-mers in $seq to linear buffer $buf
|
||||
{
|
||||
int i, l, last = -1;
|
||||
uint64_t x[4], mask = (1ULL<<k) - 1, shift = k - 1;
|
||||
for (i = l = 0, x[0] = x[1] = x[2] = x[3] = 0; i < len; ++i) {
|
||||
int c = seq_nt4_table[(uint8_t)seq[i]];
|
||||
if (c < 4) { // not an "N" base
|
||||
if (c != last) {
|
||||
x[0] = (x[0] << 1 | (c&1)) & mask;
|
||||
x[1] = (x[1] << 1 | (c>>1)) & mask;
|
||||
x[2] = x[2] >> 1 | (uint64_t)(1 - (c&1)) << shift;
|
||||
x[3] = x[3] >> 1 | (uint64_t)(1 - (c>>1)) << shift;
|
||||
if (++l >= k)
|
||||
ct_insert_buf(buf, p, yak_hash_long(x));
|
||||
last = c;
|
||||
}
|
||||
} else l = 0, last = -1, x[0] = x[1] = x[2] = x[3] = 0; // if there is an "N", restart
|
||||
}
|
||||
}
|
||||
|
||||
/******************
|
||||
* K-mer counting *
|
||||
******************/
|
||||
|
||||
KSEQ_INIT(gzFile, gzread)
|
||||
|
||||
#define HAF_COUNT_EXACT 0x1
|
||||
#define HAF_COUNT_ALL 0x2
|
||||
#define HAF_RS_WRITE_LEN 0x4
|
||||
#define HAF_RS_WRITE_SEQ 0x8
|
||||
#define HAF_RS_READ 0x10
|
||||
#define HAF_CREATE_NEW 0x20
|
||||
|
||||
typedef struct { // global data structure for kt_pipeline()
|
||||
const yak_copt_t *opt;
|
||||
const void *flt_tab;
|
||||
int flag, create_new, is_store;
|
||||
uint64_t n_seq;
|
||||
kseq_t *ks;
|
||||
UC_Read ucr;
|
||||
ha_ct_t *ct;
|
||||
ha_pt_t *pt;
|
||||
const All_reads *rs_in;
|
||||
All_reads *rs_out;
|
||||
} pl_data_t;
|
||||
|
||||
typedef struct { // data structure for each step in kt_pipeline()
|
||||
pl_data_t *p;
|
||||
uint64_t n_seq0;
|
||||
int n_seq, m_seq, sum_len, nk;
|
||||
int *len;
|
||||
char **seq;
|
||||
ha_mz1_v *mz_buf;
|
||||
ha_mz1_v *mz;
|
||||
ch_buf_t *buf;
|
||||
} st_data_t;
|
||||
|
||||
static void worker_for_insert(void *data, long i, int tid) // callback for kt_for()
|
||||
{
|
||||
st_data_t *s = (st_data_t*)data;
|
||||
ch_buf_t *b = &s->buf[i];
|
||||
if (s->p->pt)
|
||||
b->n_ins += ha_pt_insert_list(s->p->pt, b->n, b->b);
|
||||
else
|
||||
b->n_ins += ha_ct_insert_list(s->p->ct, s->p->create_new, b->n, b->a);
|
||||
}
|
||||
|
||||
static void worker_for_mz(void *data, long i, int tid)
|
||||
{
|
||||
st_data_t *s = (st_data_t*)data;
|
||||
ha_mz1_v *b = &s->mz_buf[tid];
|
||||
s->mz_buf[tid].n = 0;
|
||||
ha_sketch(s->seq[i], s->len[i], s->p->opt->w, s->p->opt->k, s->n_seq0 + i, s->p->opt->is_HPC, b, s->p->flt_tab);
|
||||
s->mz[i].n = s->mz[i].m = b->n;
|
||||
MALLOC(s->mz[i].a, b->n);
|
||||
memcpy(s->mz[i].a, b->a, b->n * sizeof(ha_mz1_t));
|
||||
}
|
||||
|
||||
static void *worker_count(void *data, int step, void *in) // callback for kt_pipeline()
|
||||
{
|
||||
pl_data_t *p = (pl_data_t*)data;
|
||||
if (step == 0) { // step 1: read a block of sequences
|
||||
int ret;
|
||||
st_data_t *s;
|
||||
CALLOC(s, 1);
|
||||
s->p = p;
|
||||
s->n_seq0 = p->n_seq;
|
||||
if (p->rs_in && (p->flag & HAF_RS_READ)) {
|
||||
while (p->n_seq < p->rs_in->total_reads) {
|
||||
int l;
|
||||
recover_UC_Read(&p->ucr, p->rs_in, p->n_seq);
|
||||
l = p->ucr.length;
|
||||
if (s->n_seq == s->m_seq) {
|
||||
s->m_seq = s->m_seq < 16? 16 : s->m_seq + (s->m_seq>>1);
|
||||
REALLOC(s->len, s->m_seq);
|
||||
REALLOC(s->seq, s->m_seq);
|
||||
}
|
||||
MALLOC(s->seq[s->n_seq], l);
|
||||
memcpy(s->seq[s->n_seq], p->ucr.seq, l);
|
||||
s->len[s->n_seq++] = l;
|
||||
++p->n_seq;
|
||||
s->sum_len += l;
|
||||
s->nk += l >= p->opt->k? l - p->opt->k + 1 : 0;
|
||||
if (s->sum_len >= p->opt->chunk_size)
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
while ((ret = kseq_read(p->ks)) >= 0) {
|
||||
int l = p->ks->seq.l;
|
||||
if (p->n_seq >= 1<<28) {
|
||||
fprintf(stderr, "ERROR: this implementation supports no more than %d reads\n", 1<<28);
|
||||
exit(1);
|
||||
}
|
||||
if (p->rs_out) {
|
||||
if (p->flag & HAF_RS_WRITE_LEN) {
|
||||
assert(p->n_seq == p->rs_out->total_reads);
|
||||
ha_insert_read_len(p->rs_out, l, p->ks->name.l);
|
||||
} else if (p->flag & HAF_RS_WRITE_SEQ) {
|
||||
int i, n_N;
|
||||
assert(l == (int)p->rs_out->read_length[p->n_seq]);
|
||||
for (i = n_N = 0; i < l; ++i) // count number of ambiguous bases
|
||||
if (seq_nt4_table[(uint8_t)p->ks->seq.s[i]] >= 4)
|
||||
++n_N;
|
||||
ha_compress_base(Get_READ(*p->rs_out, p->n_seq), p->ks->seq.s, l, &p->rs_out->N_site[p->n_seq], n_N);
|
||||
memcpy(&p->rs_out->name[p->rs_out->name_index[p->n_seq]], p->ks->name.s, p->ks->name.l);
|
||||
}
|
||||
}
|
||||
if (s->n_seq == s->m_seq) {
|
||||
s->m_seq = s->m_seq < 16? 16 : s->m_seq + (s->m_seq>>1);
|
||||
REALLOC(s->len, s->m_seq);
|
||||
REALLOC(s->seq, s->m_seq);
|
||||
}
|
||||
MALLOC(s->seq[s->n_seq], l);
|
||||
memcpy(s->seq[s->n_seq], p->ks->seq.s, l);
|
||||
s->len[s->n_seq++] = l;
|
||||
++p->n_seq;
|
||||
s->sum_len += l;
|
||||
s->nk += l >= p->opt->k? l - p->opt->k + 1 : 0;
|
||||
if (s->sum_len >= p->opt->chunk_size)
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (s->sum_len == 0) free(s);
|
||||
else return s;
|
||||
} else if (step == 1) { // step 2: extract k-mers
|
||||
st_data_t *s = (st_data_t*)in;
|
||||
int i, n_pre = 1<<p->opt->pre, m;
|
||||
// allocate the k-mer buffer
|
||||
CALLOC(s->buf, n_pre);
|
||||
m = (int)(s->nk * 1.2 / n_pre) + 1;
|
||||
for (i = 0; i < n_pre; ++i) {
|
||||
s->buf[i].m = m;
|
||||
if (p->pt) MALLOC(s->buf[i].b, m);
|
||||
else MALLOC(s->buf[i].a, m);
|
||||
}
|
||||
// fill the buffer
|
||||
if (p->opt->w == 1) { // enumerate all k-mers
|
||||
for (i = 0; i < s->n_seq; ++i) {
|
||||
if (p->opt->is_HPC)
|
||||
count_seq_buf_HPC(s->buf, p->opt->k, p->opt->pre, s->len[i], s->seq[i]);
|
||||
else
|
||||
count_seq_buf(s->buf, p->opt->k, p->opt->pre, s->len[i], s->seq[i]);
|
||||
if (!p->is_store) free(s->seq[i]);
|
||||
}
|
||||
} else { // minimizers only
|
||||
uint32_t j;
|
||||
// compute minimizers
|
||||
CALLOC(s->mz, s->n_seq);
|
||||
CALLOC(s->mz_buf, p->opt->n_thread);
|
||||
kt_for(p->opt->n_thread, worker_for_mz, s, s->n_seq);
|
||||
for (i = 0; i < p->opt->n_thread; ++i)
|
||||
free(s->mz_buf[i].a);
|
||||
free(s->mz_buf);
|
||||
// insert minimizers
|
||||
if (p->pt) {
|
||||
for (i = 0; i < s->n_seq; ++i)
|
||||
for (j = 0; j < s->mz[i].n; ++j)
|
||||
pt_insert_buf(s->buf, p->opt->pre, &s->mz[i].a[j]);
|
||||
} else {
|
||||
for (i = 0; i < s->n_seq; ++i)
|
||||
for (j = 0; j < s->mz[i].n; ++j)
|
||||
ct_insert_buf(s->buf, p->opt->pre, s->mz[i].a[j].x);
|
||||
}
|
||||
for (i = 0; i < s->n_seq; ++i) {
|
||||
free(s->mz[i].a);
|
||||
if (!p->is_store) free(s->seq[i]);
|
||||
}
|
||||
free(s->mz);
|
||||
}
|
||||
free(s->seq); free(s->len);
|
||||
s->seq = 0, s->len = 0;
|
||||
return s;
|
||||
} else if (step == 2) { // step 3: insert k-mers to hash table
|
||||
st_data_t *s = (st_data_t*)in;
|
||||
int i, n = 1<<p->opt->pre;
|
||||
uint64_t n_ins = 0;
|
||||
kt_for(p->opt->n_thread, worker_for_insert, s, n);
|
||||
for (i = 0; i < n; ++i) {
|
||||
n_ins += s->buf[i].n_ins;
|
||||
if (p->pt) free(s->buf[i].b);
|
||||
else free(s->buf[i].a);
|
||||
}
|
||||
if (p->ct) p->ct->tot += n_ins;
|
||||
if (p->pt) p->pt->tot_pos += n_ins;
|
||||
free(s->buf);
|
||||
#if 0
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] processed %ld sequences; %ld %s in the hash table\n", __func__,
|
||||
yak_realtime(), yak_cpu_usage(), (long)s->n_seq0 + s->n_seq,
|
||||
(long)(p->pt? p->pt->tot_pos : p->ct->tot), p->pt? "positions" : "distinct k-mers");
|
||||
#endif
|
||||
free(s);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static ha_ct_t *yak_count(const yak_copt_t *opt, const char *fn, int flag, ha_pt_t *p0, ha_ct_t *c0, const void *flt_tab, All_reads *rs, int64_t *n_seq)
|
||||
{
|
||||
int read_rs = (rs && (flag & HAF_RS_READ));
|
||||
pl_data_t pl;
|
||||
gzFile fp = 0;
|
||||
memset(&pl, 0, sizeof(pl_data_t));
|
||||
pl.n_seq = *n_seq;
|
||||
if (read_rs) {
|
||||
pl.rs_in = rs;
|
||||
init_UC_Read(&pl.ucr);
|
||||
} else {
|
||||
if ((fp = gzopen(fn, "r")) == 0) return 0;
|
||||
pl.ks = kseq_init(fp);
|
||||
}
|
||||
if (rs && (flag & (HAF_RS_WRITE_LEN|HAF_RS_WRITE_SEQ)))
|
||||
pl.rs_out = rs;
|
||||
pl.flt_tab = flt_tab;
|
||||
pl.opt = opt;
|
||||
pl.flag = flag;
|
||||
if (p0) {
|
||||
pl.pt = p0, pl.create_new = 0; // never create new elements in a position table
|
||||
assert(p0->k == opt->k && p0->pre == opt->pre);
|
||||
} else if (c0) {
|
||||
pl.ct = c0, pl.create_new = !!(flag&HAF_CREATE_NEW);
|
||||
assert(c0->k == opt->k && c0->pre == opt->pre);
|
||||
} else {
|
||||
pl.create_new = 1; // alware create new elements if the count table is empty
|
||||
pl.ct = ha_ct_init(opt->k, opt->pre, opt->bf_n_hash, opt->bf_shift);
|
||||
}
|
||||
kt_pipeline(3, worker_count, &pl, 3);
|
||||
if (read_rs) {
|
||||
destory_UC_Read(&pl.ucr);
|
||||
} else {
|
||||
kseq_destroy(pl.ks);
|
||||
gzclose(fp);
|
||||
}
|
||||
*n_seq = pl.n_seq;
|
||||
return pl.ct;
|
||||
}
|
||||
|
||||
ha_ct_t *ha_count(const hifiasm_opt_t *asm_opt, int flag, ha_pt_t *p0, const void *flt_tab, All_reads *rs)
|
||||
{
|
||||
int i;
|
||||
int64_t n_seq = 0;
|
||||
yak_copt_t opt;
|
||||
ha_ct_t *h = 0;
|
||||
assert(!(flag & HAF_RS_WRITE_LEN) || !(flag & HAF_RS_WRITE_SEQ)); // not both
|
||||
if (rs) {
|
||||
if (flag & HAF_RS_WRITE_LEN)
|
||||
init_All_reads(rs);
|
||||
else if (flag & HAF_RS_WRITE_SEQ)
|
||||
malloc_All_reads(rs);
|
||||
}
|
||||
yak_copt_init(&opt);
|
||||
opt.k = asm_opt->k_mer_length;
|
||||
opt.is_HPC = !(asm_opt->flag&HA_F_NO_HPC);
|
||||
opt.w = flag & HAF_COUNT_ALL? 1 : asm_opt->mz_win;
|
||||
opt.bf_shift = flag & HAF_COUNT_EXACT? 0 : asm_opt->bf_shift;
|
||||
opt.n_thread = asm_opt->thread_num;
|
||||
for (i = 0; i < asm_opt->num_reads; ++i)
|
||||
h = yak_count(&opt, asm_opt->read_file_names[i], flag|HAF_CREATE_NEW, p0, h, flt_tab, rs, &n_seq);
|
||||
if (h && opt.bf_shift > 0)
|
||||
ha_ct_destroy_bf(h);
|
||||
return h;
|
||||
}
|
||||
|
||||
/***************************
|
||||
* High count filter table *
|
||||
***************************/
|
||||
|
||||
KHASHL_SET_INIT(static klib_unused, yak_ft_t, yak_ft, uint64_t, kh_hash_dummy, kh_eq_generic)
|
||||
|
||||
static yak_ft_t *gen_hh(const ha_ct_t *h)
|
||||
{
|
||||
int i;
|
||||
yak_ft_t *hh;
|
||||
hh = yak_ft_init();
|
||||
yak_ft_resize(hh, h->tot * 2);
|
||||
for (i = 0; i < 1<<h->pre; ++i) {
|
||||
yak_ct_t *ht = h->h[i].h;
|
||||
khint_t k;
|
||||
for (k = 0; k < kh_end(ht); ++k) {
|
||||
if (kh_exist(ht, k)) {
|
||||
uint64_t y = kh_key(ht, k) >> h->pre << YAK_COUNTER_BITS | i;
|
||||
int absent;
|
||||
yak_ft_put(hh, y, &absent);
|
||||
}
|
||||
}
|
||||
}
|
||||
return hh;
|
||||
}
|
||||
|
||||
int ha_ft_isflt(const void *hh, uint64_t y)
|
||||
{
|
||||
yak_ft_t *h = (yak_ft_t*)hh;
|
||||
khint_t k;
|
||||
k = yak_ft_get(h, y);
|
||||
return k == kh_end(h)? 0 : 1;
|
||||
}
|
||||
|
||||
void ha_ft_destroy(void *h)
|
||||
{
|
||||
if (h) yak_ft_destroy((yak_ft_t*)h);
|
||||
}
|
||||
|
||||
/*************************
|
||||
* High-level interfaces *
|
||||
*************************/
|
||||
|
||||
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov)
|
||||
{
|
||||
yak_ft_t *flt_tab;
|
||||
int64_t cnt[YAK_N_COUNTS];
|
||||
int peak_hom, peak_het, cutoff;
|
||||
ha_ct_t *h;
|
||||
h = ha_count(asm_opt, HAF_COUNT_ALL|HAF_RS_WRITE_LEN, NULL, NULL, rs);
|
||||
ha_ct_hist(h, cnt, asm_opt->thread_num);
|
||||
peak_hom = ha_analyze_count(YAK_N_COUNTS, cnt, &peak_het);
|
||||
if (hom_cov) *hom_cov = peak_hom;
|
||||
if (peak_hom > 0) fprintf(stderr, "[M::%s] peak_hom: %d; peak_het: %d\n", __func__, peak_hom, peak_het);
|
||||
cutoff = (int)(peak_hom * asm_opt->high_factor);
|
||||
if (cutoff > YAK_MAX_COUNT - 1) cutoff = YAK_MAX_COUNT - 1;
|
||||
ha_ct_shrink(h, cutoff, YAK_MAX_COUNT, asm_opt->thread_num);
|
||||
flt_tab = gen_hh(h);
|
||||
ha_ct_destroy(h);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f@%.3fGB] ==> filtered out %ld k-mers occurring %d or more times\n", __func__,
|
||||
yak_realtime(), yak_cpu_usage(), yak_peakrss_in_gb(), (long)kh_size(flt_tab), cutoff);
|
||||
return (void*)flt_tab;
|
||||
}
|
||||
|
||||
ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_from_store, All_reads *rs, int *hom_cov, int *het_cov)
|
||||
{
|
||||
int64_t cnt[YAK_N_COUNTS], tot_cnt;
|
||||
int peak_hom, peak_het, i, extra_flag1, extra_flag2;
|
||||
ha_ct_t *ct;
|
||||
ha_pt_t *pt;
|
||||
if (read_from_store) {
|
||||
extra_flag1 = extra_flag2 = HAF_RS_READ;
|
||||
} else if (rs->total_reads == 0) {
|
||||
extra_flag1 = HAF_RS_WRITE_LEN;
|
||||
extra_flag2 = HAF_RS_WRITE_SEQ;
|
||||
} else {
|
||||
extra_flag1 = HAF_RS_WRITE_SEQ;
|
||||
extra_flag2 = HAF_RS_READ;
|
||||
}
|
||||
ct = ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag1, NULL, flt_tab, rs);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> counted %ld distinct minimizer k-mers\n", __func__,
|
||||
yak_realtime(), yak_cpu_usage(), (long)ct->tot);
|
||||
ha_ct_hist(ct, cnt, asm_opt->thread_num);
|
||||
fprintf(stderr, "[M::%s] count[%d] = %ld (for sanity check)\n", __func__, YAK_MAX_COUNT, (long)cnt[YAK_MAX_COUNT]);
|
||||
peak_hom = ha_analyze_count(YAK_N_COUNTS, cnt, &peak_het);
|
||||
if (hom_cov) *hom_cov = peak_hom;
|
||||
if (het_cov) *het_cov = peak_het;
|
||||
if (peak_hom > 0) fprintf(stderr, "[M::%s] peak_hom: %d; peak_het: %d\n", __func__, peak_hom, peak_het);
|
||||
if (flt_tab == 0) {
|
||||
int cutoff = (int)(peak_hom * asm_opt->high_factor);
|
||||
if (cutoff > YAK_MAX_COUNT - 1) cutoff = YAK_MAX_COUNT - 1;
|
||||
ha_ct_shrink(ct, 2, cutoff, asm_opt->thread_num);
|
||||
for (i = 2, tot_cnt = 0; i <= cutoff; ++i) tot_cnt += cnt[i] * i;
|
||||
} else {
|
||||
ha_ct_shrink(ct, 2, YAK_MAX_COUNT - 1, asm_opt->thread_num);
|
||||
for (i = 2, tot_cnt = 0; i <= YAK_MAX_COUNT - 1; ++i) tot_cnt += cnt[i] * i;
|
||||
}
|
||||
pt = ha_pt_gen(ct, asm_opt->thread_num);
|
||||
ha_count(asm_opt, HAF_COUNT_EXACT|extra_flag2, pt, flt_tab, rs);
|
||||
assert((uint64_t)tot_cnt == pt->tot_pos);
|
||||
//ha_pt_sort(pt, asm_opt->thread_num);
|
||||
fprintf(stderr, "[M::%s::%.3f*%.2f] ==> indexed %ld positions\n", __func__,
|
||||
yak_realtime(), yak_cpu_usage(), (long)pt->tot_pos);
|
||||
return pt;
|
||||
}
|
||||
103
htab.h
Normal file
103
htab.h
Normal file
@@ -0,0 +1,103 @@
|
||||
#ifndef __HA_HTAB_H__
|
||||
#define __HA_HTAB_H__
|
||||
#define __STDC_LIMIT_MACROS
|
||||
#include <stdint.h>
|
||||
#include "Process_Read.h"
|
||||
#include "CommandLines.h"
|
||||
|
||||
typedef struct {
|
||||
uint64_t x;
|
||||
uint64_t rid:28, pos:27, rev:1, span:8;
|
||||
} ha_mz1_t;
|
||||
|
||||
typedef struct {
|
||||
uint64_t rid:28, pos:27, rev:1, span:8; // actually it is not necessary to keep span in the index
|
||||
} ha_idxpos_t;
|
||||
|
||||
typedef struct { uint32_t n, m; ha_mz1_t *a; } ha_mz1_v;
|
||||
|
||||
struct ha_pt_s;
|
||||
typedef struct ha_pt_s ha_pt_t;
|
||||
|
||||
struct ha_abuf_s;
|
||||
typedef struct ha_abuf_s ha_abuf_t;
|
||||
|
||||
extern const unsigned char seq_nt4_table[256];
|
||||
extern void *ha_flt_tab;
|
||||
extern ha_pt_t *ha_idx;
|
||||
|
||||
void *ha_ft_gen(const hifiasm_opt_t *asm_opt, All_reads *rs, int *hom_cov);
|
||||
int ha_ft_isflt(const void *hh, uint64_t y);
|
||||
void ha_ft_destroy(void *h);
|
||||
|
||||
ha_pt_t *ha_pt_gen(const hifiasm_opt_t *asm_opt, const void *flt_tab, int read_from_store, All_reads *rs, int *hom_cov, int *het_cov);
|
||||
void ha_pt_destroy(ha_pt_t *h);
|
||||
const ha_idxpos_t *ha_pt_get(const ha_pt_t *h, uint64_t hash, int *n);
|
||||
|
||||
ha_abuf_t *ha_abuf_init(void);
|
||||
void ha_abuf_destroy(ha_abuf_t *ab);
|
||||
uint64_t ha_abuf_mem(const ha_abuf_t *ab);
|
||||
|
||||
double yak_cputime(void);
|
||||
void yak_reset_realtime(void);
|
||||
double yak_realtime(void);
|
||||
long yak_peakrss(void);
|
||||
double yak_peakrss_in_gb(void);
|
||||
double yak_cpu_usage(void);
|
||||
|
||||
void ha_triobin(const hifiasm_opt_t *opt);
|
||||
|
||||
void ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mz1_v *p, const void *hf);
|
||||
int ha_analyze_count(int n_cnt, const int64_t *cnt, int *peak_het);
|
||||
|
||||
static inline uint64_t yak_hash64(uint64_t key, uint64_t mask) // invertible integer hash function
|
||||
{
|
||||
key = (~key + (key << 21)) & mask; // key = (key << 21) - key - 1;
|
||||
key = key ^ key >> 24;
|
||||
key = ((key + (key << 3)) + (key << 8)) & mask; // key * 265
|
||||
key = key ^ key >> 14;
|
||||
key = ((key + (key << 2)) + (key << 4)) & mask; // key * 21
|
||||
key = key ^ key >> 28;
|
||||
key = (key + (key << 31)) & mask;
|
||||
return key;
|
||||
}
|
||||
|
||||
static inline uint64_t yak_hash64_64(uint64_t key)
|
||||
{
|
||||
key = ~key + (key << 21);
|
||||
key = key ^ key >> 24;
|
||||
key = (key + (key << 3)) + (key << 8);
|
||||
key = key ^ key >> 14;
|
||||
key = (key + (key << 2)) + (key << 4);
|
||||
key = key ^ key >> 28;
|
||||
key = key + (key << 31);
|
||||
return key;
|
||||
}
|
||||
|
||||
static inline uint64_t yak_hash_long(uint64_t x[4])
|
||||
{
|
||||
int j = x[1] < x[3]? 0 : 1;
|
||||
return yak_hash64_64(x[j<<1|0]) + yak_hash64_64(x[j<<1|1]);
|
||||
}
|
||||
|
||||
#define CALLOC(ptr, len) ((ptr) = (__typeof__(ptr))calloc((len), sizeof(*(ptr))))
|
||||
#define MALLOC(ptr, len) ((ptr) = (__typeof__(ptr))malloc((len) * sizeof(*(ptr))))
|
||||
#define REALLOC(ptr, len) ((ptr) = (__typeof__(ptr))realloc((ptr), (len) * sizeof(*(ptr))))
|
||||
|
||||
#ifndef kroundup32
|
||||
#define kroundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
||||
#endif
|
||||
|
||||
#ifndef kroundup64
|
||||
#define kroundup64(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, x|=(x)>>32, ++(x))
|
||||
#endif
|
||||
|
||||
#ifndef klib_unused
|
||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
||||
#define klib_unused __attribute__ ((__unused__))
|
||||
#else
|
||||
#define klib_unused
|
||||
#endif
|
||||
#endif /* klib_unused */
|
||||
|
||||
#endif // __YAK_H__
|
||||
2
ketopt.h
2
ketopt.h
@@ -17,7 +17,7 @@ typedef struct {
|
||||
} ketopt_t;
|
||||
|
||||
typedef struct {
|
||||
char *name;
|
||||
const char *name;
|
||||
int has_arg;
|
||||
int val;
|
||||
} ko_longopt_t;
|
||||
|
||||
669
khash.h
669
khash.h
@@ -1,669 +0,0 @@
|
||||
/* The MIT License
|
||||
|
||||
Copyright (c) 2008, 2009, 2011 by Attractive Chaos <attractor@live.co.uk>
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
*/
|
||||
|
||||
/*
|
||||
An example:
|
||||
|
||||
#include "khash.h"
|
||||
KHASH_MAP_INIT_INT(32, char)
|
||||
int main() {
|
||||
int ret, is_missing;
|
||||
khiter_t k;
|
||||
khash_t(32) *h = kh_init(32);
|
||||
k = kh_put(32, h, 5, &ret);
|
||||
kh_value(h, k) = 10;
|
||||
k = kh_get(32, h, 10);
|
||||
is_missing = (k == kh_end(h));
|
||||
k = kh_get(32, h, 5);
|
||||
kh_del(32, h, k);
|
||||
for (k = kh_begin(h); k != kh_end(h); ++k)
|
||||
if (kh_exist(h, k)) kh_value(h, k) = 1;
|
||||
kh_destroy(32, h);
|
||||
return 0;
|
||||
}
|
||||
*/
|
||||
|
||||
/*
|
||||
2013-05-02 (0.2.8):
|
||||
|
||||
* Use quadratic probing. When the capacity is power of 2, stepping function
|
||||
i*(i+1)/2 guarantees to traverse each bucket. It is better than double
|
||||
hashing on cache performance and is more robust than linear probing.
|
||||
|
||||
In theory, double hashing should be more robust than quadratic probing.
|
||||
However, my implementation is probably not for large hash tables, because
|
||||
the second hash function is closely tied to the first hash function,
|
||||
which reduce the effectiveness of double hashing.
|
||||
|
||||
Reference: http://research.cs.vt.edu/AVresearch/hashing/quadratic.php
|
||||
|
||||
2011-12-29 (0.2.7):
|
||||
|
||||
* Minor code clean up; no actual effect.
|
||||
|
||||
2011-09-16 (0.2.6):
|
||||
|
||||
* The capacity is a power of 2. This seems to dramatically improve the
|
||||
speed for simple keys. Thank Zilong Tan for the suggestion. Reference:
|
||||
|
||||
- http://code.google.com/p/ulib/
|
||||
- http://nothings.org/computer/judy/
|
||||
|
||||
* Allow to optionally use linear probing which usually has better
|
||||
performance for random input. Double hashing is still the default as it
|
||||
is more robust to certain non-random input.
|
||||
|
||||
* Added Wang's integer hash function (not used by default). This hash
|
||||
function is more robust to certain non-random input.
|
||||
|
||||
2011-02-14 (0.2.5):
|
||||
|
||||
* Allow to declare global functions.
|
||||
|
||||
2009-09-26 (0.2.4):
|
||||
|
||||
* Improve portability
|
||||
|
||||
2008-09-19 (0.2.3):
|
||||
|
||||
* Corrected the example
|
||||
* Improved interfaces
|
||||
|
||||
2008-09-11 (0.2.2):
|
||||
|
||||
* Improved speed a little in kh_put()
|
||||
|
||||
2008-09-10 (0.2.1):
|
||||
|
||||
* Added kh_clear()
|
||||
* Fixed a compiling error
|
||||
|
||||
2008-09-02 (0.2.0):
|
||||
|
||||
* Changed to token concatenation which increases flexibility.
|
||||
|
||||
2008-08-31 (0.1.2):
|
||||
|
||||
* Fixed a bug in kh_get(), which has not been tested previously.
|
||||
|
||||
2008-08-31 (0.1.1):
|
||||
|
||||
* Added destructor
|
||||
*/
|
||||
|
||||
|
||||
#ifndef __AC_KHASH_H
|
||||
#define __AC_KHASH_H
|
||||
|
||||
/*!
|
||||
@header
|
||||
|
||||
Generic hash table library.
|
||||
*/
|
||||
|
||||
#define AC_VERSION_KHASH_H "0.2.8"
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
|
||||
|
||||
/* compiler specific configuration */
|
||||
|
||||
#if UINT_MAX == 0xffffffffu
|
||||
typedef unsigned int khint32_t;
|
||||
#elif ULONG_MAX == 0xffffffffu
|
||||
typedef unsigned long khint32_t;
|
||||
#endif
|
||||
|
||||
#if ULONG_MAX == ULLONG_MAX
|
||||
typedef unsigned long khint64_t;
|
||||
#else
|
||||
typedef unsigned long long khint64_t;
|
||||
#endif
|
||||
|
||||
#ifndef kh_inline
|
||||
#ifdef _MSC_VER
|
||||
#define kh_inline __inline
|
||||
#else
|
||||
#define kh_inline inline
|
||||
#endif
|
||||
#endif /* kh_inline */
|
||||
|
||||
#ifndef klib_unused
|
||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
||||
#define klib_unused __attribute__ ((__unused__))
|
||||
#else
|
||||
#define klib_unused
|
||||
#endif
|
||||
#endif /* klib_unused */
|
||||
|
||||
typedef khint32_t khint_t;
|
||||
typedef khint_t khiter_t;
|
||||
|
||||
#define __ac_isempty(flag, i) ((flag[i>>4]>>((i&0xfU)<<1))&2)
|
||||
#define __ac_isdel(flag, i) ((flag[i>>4]>>((i&0xfU)<<1))&1)
|
||||
#define __ac_iseither(flag, i) ((flag[i>>4]>>((i&0xfU)<<1))&3)
|
||||
#define __ac_set_isdel_false(flag, i) (flag[i>>4]&=~(1ul<<((i&0xfU)<<1)))
|
||||
#define __ac_set_isempty_false(flag, i) (flag[i>>4]&=~(2ul<<((i&0xfU)<<1)))
|
||||
#define __ac_set_isboth_false(flag, i) (flag[i>>4]&=~(3ul<<((i&0xfU)<<1)))
|
||||
#define __ac_set_isdel_true(flag, i) (flag[i>>4]|=1ul<<((i&0xfU)<<1))
|
||||
|
||||
#define __ac_fsize(m) ((m) < 16? 1 : (m)>>4)
|
||||
|
||||
#ifndef kroundup32
|
||||
#define kroundup32(x) (--(x), (x)|=(x)>>1, (x)|=(x)>>2, (x)|=(x)>>4, (x)|=(x)>>8, (x)|=(x)>>16, ++(x))
|
||||
#endif
|
||||
|
||||
#ifndef kcalloc
|
||||
#define kcalloc(N,Z) calloc(N,Z)
|
||||
#endif
|
||||
#ifndef kmalloc
|
||||
#define kmalloc(Z) malloc(Z)
|
||||
#endif
|
||||
#ifndef krealloc
|
||||
#define krealloc(P,Z) realloc(P,Z)
|
||||
#endif
|
||||
#ifndef kfree
|
||||
#define kfree(P) free(P)
|
||||
#endif
|
||||
|
||||
static const double __ac_HASH_UPPER = 0.77;
|
||||
|
||||
#define __KHASH_TYPE(name, khkey_t, khval_t) \
|
||||
typedef struct kh_##name##_s { \
|
||||
khint_t n_buckets, size, n_occupied, upper_bound; \
|
||||
khint32_t *flags; \
|
||||
khkey_t *keys; \
|
||||
khval_t *vals; \
|
||||
} kh_##name##_t;
|
||||
|
||||
#define __KHASH_PROTOTYPES(name, khkey_t, khval_t) \
|
||||
extern kh_##name##_t *kh_init_##name(void); \
|
||||
extern void kh_destroy_##name(kh_##name##_t *h); \
|
||||
extern void kh_clear_##name(kh_##name##_t *h); \
|
||||
extern khint_t kh_get_##name(const kh_##name##_t *h, khkey_t key); \
|
||||
extern int kh_resize_##name(kh_##name##_t *h, khint_t new_n_buckets); \
|
||||
extern khint_t kh_put_##name(kh_##name##_t *h, khkey_t key, int *ret); \
|
||||
extern void kh_del_##name(kh_##name##_t *h, khint_t x);\
|
||||
extern void kh_write_##name(kh_##name##_t *h, FILE* fp);\
|
||||
extern void kh_load_##name(kh_##name##_t *h, FILE* fp);
|
||||
|
||||
#define __KHASH_IMPL(name, SCOPE, khkey_t, khval_t, kh_is_map, __hash_func, __hash_equal) \
|
||||
SCOPE kh_##name##_t *kh_init_##name(void) { \
|
||||
return (kh_##name##_t*)kcalloc(1, sizeof(kh_##name##_t)); \
|
||||
} \
|
||||
SCOPE void kh_destroy_##name(kh_##name##_t *h) \
|
||||
{ \
|
||||
if (h) { \
|
||||
kfree((void *)h->keys); kfree(h->flags); \
|
||||
kfree((void *)h->vals); \
|
||||
kfree(h); \
|
||||
} \
|
||||
} \
|
||||
SCOPE void kh_clear_##name(kh_##name##_t *h) \
|
||||
{ \
|
||||
if (h && h->flags) { \
|
||||
memset(h->flags, 0xaa, __ac_fsize(h->n_buckets) * sizeof(khint32_t)); \
|
||||
h->size = h->n_occupied = 0; \
|
||||
} \
|
||||
} \
|
||||
SCOPE khint_t kh_get_##name(const kh_##name##_t *h, khkey_t key) \
|
||||
{ \
|
||||
if (h->n_buckets) { \
|
||||
khint_t k, i, last, mask, step = 0; \
|
||||
mask = h->n_buckets - 1; \
|
||||
k = __hash_func(key); i = k & mask; \
|
||||
last = i; \
|
||||
while (!__ac_isempty(h->flags, i) && (__ac_isdel(h->flags, i) || !__hash_equal(h->keys[i], key))) { \
|
||||
i = (i + (++step)) & mask; \
|
||||
if (i == last) return h->n_buckets; \
|
||||
} \
|
||||
return __ac_iseither(h->flags, i)? h->n_buckets : i; \
|
||||
} else return 0; \
|
||||
} \
|
||||
SCOPE int kh_resize_##name(kh_##name##_t *h, khint_t new_n_buckets) \
|
||||
{ /* This function uses 0.25*n_buckets bytes of working space instead of [sizeof(key_t+val_t)+.25]*n_buckets. */ \
|
||||
khint32_t *new_flags = 0; \
|
||||
khint_t j = 1; \
|
||||
{ \
|
||||
kroundup32(new_n_buckets); \
|
||||
if (new_n_buckets < 4) new_n_buckets = 4; \
|
||||
if (h->size >= (khint_t)(new_n_buckets * __ac_HASH_UPPER + 0.5)) j = 0; /* requested size is too small */ \
|
||||
else { /* hash table size to be changed (shrink or expand); rehash */ \
|
||||
new_flags = (khint32_t*)kmalloc(__ac_fsize(new_n_buckets) * sizeof(khint32_t)); \
|
||||
if (!new_flags) return -1; \
|
||||
memset(new_flags, 0xaa, __ac_fsize(new_n_buckets) * sizeof(khint32_t)); \
|
||||
if (h->n_buckets < new_n_buckets) { /* expand */ \
|
||||
khkey_t *new_keys = (khkey_t*)krealloc((void *)h->keys, new_n_buckets * sizeof(khkey_t)); \
|
||||
if (!new_keys) { kfree(new_flags); return -1; } \
|
||||
h->keys = new_keys; \
|
||||
if (kh_is_map) { \
|
||||
khval_t *new_vals = (khval_t*)krealloc((void *)h->vals, new_n_buckets * sizeof(khval_t)); \
|
||||
if (!new_vals) { kfree(new_flags); return -1; } \
|
||||
h->vals = new_vals; \
|
||||
} \
|
||||
} /* otherwise shrink */ \
|
||||
} \
|
||||
} \
|
||||
if (j) { /* rehashing is needed */ \
|
||||
for (j = 0; j != h->n_buckets; ++j) { \
|
||||
if (__ac_iseither(h->flags, j) == 0) { \
|
||||
khkey_t key = h->keys[j]; \
|
||||
khval_t val; \
|
||||
khint_t new_mask; \
|
||||
new_mask = new_n_buckets - 1; \
|
||||
if (kh_is_map) val = h->vals[j]; \
|
||||
__ac_set_isdel_true(h->flags, j); \
|
||||
while (1) { /* kick-out process; sort of like in Cuckoo hashing */ \
|
||||
khint_t k, i, step = 0; \
|
||||
k = __hash_func(key); \
|
||||
i = k & new_mask; \
|
||||
while (!__ac_isempty(new_flags, i)) i = (i + (++step)) & new_mask; \
|
||||
__ac_set_isempty_false(new_flags, i); \
|
||||
if (i < h->n_buckets && __ac_iseither(h->flags, i) == 0) { /* kick out the existing element */ \
|
||||
{ khkey_t tmp = h->keys[i]; h->keys[i] = key; key = tmp; } \
|
||||
if (kh_is_map) { khval_t tmp = h->vals[i]; h->vals[i] = val; val = tmp; } \
|
||||
__ac_set_isdel_true(h->flags, i); /* mark it as deleted in the old hash table */ \
|
||||
} else { /* write the element and jump out of the loop */ \
|
||||
h->keys[i] = key; \
|
||||
if (kh_is_map) h->vals[i] = val; \
|
||||
break; \
|
||||
} \
|
||||
} \
|
||||
} \
|
||||
} \
|
||||
if (h->n_buckets > new_n_buckets) { /* shrink the hash table */ \
|
||||
h->keys = (khkey_t*)krealloc((void *)h->keys, new_n_buckets * sizeof(khkey_t)); \
|
||||
if (kh_is_map) h->vals = (khval_t*)krealloc((void *)h->vals, new_n_buckets * sizeof(khval_t)); \
|
||||
} \
|
||||
kfree(h->flags); /* free the working space */ \
|
||||
h->flags = new_flags; \
|
||||
h->n_buckets = new_n_buckets; \
|
||||
h->n_occupied = h->size; \
|
||||
h->upper_bound = (khint_t)(h->n_buckets * __ac_HASH_UPPER + 0.5); \
|
||||
} \
|
||||
return 0; \
|
||||
} \
|
||||
SCOPE khint_t kh_put_##name(kh_##name##_t *h, khkey_t key, int *ret) \
|
||||
{ \
|
||||
khint_t x; \
|
||||
if (h->n_occupied >= h->upper_bound) { /* update the hash table */ \
|
||||
if (h->n_buckets > (h->size<<1)) { \
|
||||
if (kh_resize_##name(h, h->n_buckets - 1) < 0) { /* clear "deleted" elements */ \
|
||||
*ret = -1; return h->n_buckets; \
|
||||
} \
|
||||
} else if (kh_resize_##name(h, h->n_buckets + 1) < 0) { /* expand the hash table */ \
|
||||
*ret = -1; return h->n_buckets; \
|
||||
} \
|
||||
} /* TODO: to implement automatically shrinking; resize() already support shrinking */ \
|
||||
{ \
|
||||
khint_t k, i, site, last, mask = h->n_buckets - 1, step = 0; \
|
||||
x = site = h->n_buckets; k = __hash_func(key); i = k & mask; \
|
||||
if (__ac_isempty(h->flags, i)) x = i; /* for speed up */ \
|
||||
else { \
|
||||
last = i; \
|
||||
while (!__ac_isempty(h->flags, i) && (__ac_isdel(h->flags, i) || !__hash_equal(h->keys[i], key))) { \
|
||||
if (__ac_isdel(h->flags, i)) site = i; \
|
||||
i = (i + (++step)) & mask; \
|
||||
if (i == last) { x = site; break; } \
|
||||
} \
|
||||
if (x == h->n_buckets) { \
|
||||
if (__ac_isempty(h->flags, i) && site != h->n_buckets) x = site; \
|
||||
else x = i; \
|
||||
} \
|
||||
} \
|
||||
} \
|
||||
if (__ac_isempty(h->flags, x)) { /* not present at all */ \
|
||||
h->keys[x] = key; \
|
||||
__ac_set_isboth_false(h->flags, x); \
|
||||
++h->size; ++h->n_occupied; \
|
||||
*ret = 1; \
|
||||
} else if (__ac_isdel(h->flags, x)) { /* deleted */ \
|
||||
h->keys[x] = key; \
|
||||
__ac_set_isboth_false(h->flags, x); \
|
||||
++h->size; \
|
||||
*ret = 2; \
|
||||
} else *ret = 0; /* Don't touch h->keys[x] if present and not deleted */ \
|
||||
return x; \
|
||||
} \
|
||||
SCOPE void kh_del_##name(kh_##name##_t *h, khint_t x) \
|
||||
{ \
|
||||
if (x != h->n_buckets && !__ac_iseither(h->flags, x)) { \
|
||||
__ac_set_isdel_true(h->flags, x); \
|
||||
--h->size; \
|
||||
} \
|
||||
} \
|
||||
SCOPE void kh_write_##name(kh_##name##_t *h, FILE* fp)\
|
||||
{\
|
||||
fwrite(&(h->n_buckets), sizeof(khint_t), 1, fp);\
|
||||
fwrite(&(h->size), sizeof(khint_t), 1, fp);\
|
||||
fwrite(&(h->n_occupied), sizeof(khint_t), 1, fp);\
|
||||
fwrite(&(h->upper_bound), sizeof(khint_t), 1, fp);\
|
||||
if (h->n_buckets)\
|
||||
{\
|
||||
fwrite(h->flags, sizeof(khint32_t), __ac_fsize(h->n_buckets), fp);\
|
||||
fwrite(h->keys, sizeof(khkey_t), h->n_buckets, fp);\
|
||||
fwrite(h->vals, sizeof(khval_t), h->n_buckets, fp);\
|
||||
}\
|
||||
} \
|
||||
SCOPE void kh_load_##name(kh_##name##_t *h, FILE* fp)\
|
||||
{\
|
||||
int f_flag;\
|
||||
f_flag = fread(&(h->n_buckets), sizeof(khint_t), 1, fp);\
|
||||
f_flag += fread(&(h->size), sizeof(khint_t), 1, fp);\
|
||||
f_flag += fread(&(h->n_occupied), sizeof(khint_t), 1, fp);\
|
||||
f_flag += fread(&(h->upper_bound), sizeof(khint_t), 1, fp);\
|
||||
if (h->n_buckets)\
|
||||
{\
|
||||
h->flags = (khint32_t*)kmalloc(__ac_fsize(h->n_buckets) * sizeof(khint32_t));\
|
||||
f_flag += fread(h->flags, sizeof(khint32_t), __ac_fsize(h->n_buckets), fp);\
|
||||
h->keys = (khkey_t*)kmalloc(sizeof(khkey_t)*h->n_buckets);\
|
||||
f_flag += fread(h->keys, sizeof(khkey_t), h->n_buckets, fp);\
|
||||
h->vals = (khval_t*)kmalloc(sizeof(khval_t)*h->n_buckets);\
|
||||
f_flag += fread(h->vals, sizeof(khval_t), h->n_buckets, fp);\
|
||||
}\
|
||||
}
|
||||
|
||||
#define KHASH_DECLARE(name, khkey_t, khval_t) \
|
||||
__KHASH_TYPE(name, khkey_t, khval_t) \
|
||||
__KHASH_PROTOTYPES(name, khkey_t, khval_t)
|
||||
|
||||
#define KHASH_INIT2(name, SCOPE, khkey_t, khval_t, kh_is_map, __hash_func, __hash_equal) \
|
||||
__KHASH_TYPE(name, khkey_t, khval_t) \
|
||||
__KHASH_IMPL(name, SCOPE, khkey_t, khval_t, kh_is_map, __hash_func, __hash_equal)
|
||||
|
||||
#define KHASH_INIT(name, khkey_t, khval_t, kh_is_map, __hash_func, __hash_equal) \
|
||||
KHASH_INIT2(name, static kh_inline klib_unused, khkey_t, khval_t, kh_is_map, __hash_func, __hash_equal)
|
||||
|
||||
/* --- BEGIN OF HASH FUNCTIONS --- */
|
||||
|
||||
/*! @function
|
||||
@abstract Integer hash function
|
||||
@param key The integer [khint32_t]
|
||||
@return The hash value [khint_t]
|
||||
*/
|
||||
#define kh_int_hash_func(key) (khint32_t)(key)
|
||||
/*! @function
|
||||
@abstract Integer comparison function
|
||||
*/
|
||||
#define kh_int_hash_equal(a, b) ((a) == (b))
|
||||
/*! @function
|
||||
@abstract 64-bit integer hash function
|
||||
@param key The integer [khint64_t]
|
||||
@return The hash value [khint_t]
|
||||
*/
|
||||
#define kh_int64_hash_func(key) (khint32_t)((key)>>33^(key)^(key)<<11)
|
||||
/*! @function
|
||||
@abstract 64-bit integer comparison function
|
||||
*/
|
||||
#define kh_int64_hash_equal(a, b) ((a) == (b))
|
||||
/*! @function
|
||||
@abstract const char* hash function
|
||||
@param s Pointer to a null terminated string
|
||||
@return The hash value
|
||||
*/
|
||||
static kh_inline khint_t __ac_X31_hash_string(const char *s)
|
||||
{
|
||||
khint_t h = (khint_t)*s;
|
||||
if (h) for (++s ; *s; ++s) h = (h << 5) - h + (khint_t)*s;
|
||||
return h;
|
||||
}
|
||||
/*! @function
|
||||
@abstract Another interface to const char* hash function
|
||||
@param key Pointer to a null terminated string [const char*]
|
||||
@return The hash value [khint_t]
|
||||
*/
|
||||
#define kh_str_hash_func(key) __ac_X31_hash_string(key)
|
||||
/*! @function
|
||||
@abstract Const char* comparison function
|
||||
*/
|
||||
#define kh_str_hash_equal(a, b) (strcmp(a, b) == 0)
|
||||
|
||||
static kh_inline khint_t __ac_Wang_hash(khint_t key)
|
||||
{
|
||||
key += ~(key << 15);
|
||||
key ^= (key >> 10);
|
||||
key += (key << 3);
|
||||
key ^= (key >> 6);
|
||||
key += ~(key << 11);
|
||||
key ^= (key >> 16);
|
||||
return key;
|
||||
}
|
||||
#define kh_int_hash_func2(key) __ac_Wang_hash((khint_t)key)
|
||||
|
||||
/* --- END OF HASH FUNCTIONS --- */
|
||||
|
||||
/* Other convenient macros... */
|
||||
|
||||
/*!
|
||||
@abstract Type of the hash table.
|
||||
@param name Name of the hash table [symbol]
|
||||
*/
|
||||
#define khash_t(name) kh_##name##_t
|
||||
|
||||
/*! @function
|
||||
@abstract Initiate a hash table.
|
||||
@param name Name of the hash table [symbol]
|
||||
@return Pointer to the hash table [khash_t(name)*]
|
||||
*/
|
||||
#define kh_init(name) kh_init_##name()
|
||||
|
||||
/*! @function
|
||||
@abstract Destroy a hash table.
|
||||
@param name Name of the hash table [symbol]
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
*/
|
||||
#define kh_destroy(name, h) kh_destroy_##name(h)
|
||||
|
||||
/*! @function
|
||||
@abstract Reset a hash table without deallocating memory.
|
||||
@param name Name of the hash table [symbol]
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
*/
|
||||
#define kh_clear(name, h) kh_clear_##name(h)
|
||||
|
||||
/*! @function
|
||||
@abstract Resize a hash table.
|
||||
@param name Name of the hash table [symbol]
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@param s New size [khint_t]
|
||||
*/
|
||||
#define kh_resize(name, h, s) kh_resize_##name(h, s)
|
||||
|
||||
/*! @function
|
||||
@abstract Insert a key to the hash table.
|
||||
@param name Name of the hash table [symbol]
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@param k Key [type of keys]
|
||||
@param r Extra return code: -1 if the operation failed;
|
||||
0 if the key is present in the hash table;
|
||||
1 if the bucket is empty (never used); 2 if the element in
|
||||
the bucket has been deleted [int*]
|
||||
@return Iterator to the inserted element [khint_t]
|
||||
*/
|
||||
#define kh_put(name, h, k, r) kh_put_##name(h, k, r)
|
||||
|
||||
/*! @function
|
||||
@abstract Retrieve a key from the hash table.
|
||||
@param name Name of the hash table [symbol]
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@param k Key [type of keys]
|
||||
@return Iterator to the found element, or kh_end(h) if the element is absent [khint_t]
|
||||
*/
|
||||
#define kh_get(name, h, k) kh_get_##name(h, k)
|
||||
|
||||
/*! @function
|
||||
@abstract Remove a key from the hash table.
|
||||
@param name Name of the hash table [symbol]
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@param k Iterator to the element to be deleted [khint_t]
|
||||
*/
|
||||
#define kh_del(name, h, k) kh_del_##name(h, k)
|
||||
|
||||
/*! @function
|
||||
@abstract Test whether a bucket contains data.
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@param x Iterator to the bucket [khint_t]
|
||||
@return 1 if containing data; 0 otherwise [int]
|
||||
*/
|
||||
#define kh_exist(h, x) (!__ac_iseither((h)->flags, (x)))
|
||||
|
||||
/*! @function
|
||||
@abstract Get key given an iterator
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@param x Iterator to the bucket [khint_t]
|
||||
@return Key [type of keys]
|
||||
*/
|
||||
#define kh_key(h, x) ((h)->keys[x])
|
||||
|
||||
/*! @function
|
||||
@abstract Get value given an iterator
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@param x Iterator to the bucket [khint_t]
|
||||
@return Value [type of values]
|
||||
@discussion For hash sets, calling this results in segfault.
|
||||
*/
|
||||
#define kh_val(h, x) ((h)->vals[x])
|
||||
|
||||
/*! @function
|
||||
@abstract Alias of kh_val()
|
||||
*/
|
||||
#define kh_value(h, x) ((h)->vals[x])
|
||||
|
||||
/*! @function
|
||||
@abstract Get the start iterator
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@return The start iterator [khint_t]
|
||||
*/
|
||||
#define kh_begin(h) (khint_t)(0)
|
||||
|
||||
/*! @function
|
||||
@abstract Get the end iterator
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@return The end iterator [khint_t]
|
||||
*/
|
||||
#define kh_end(h) ((h)->n_buckets)
|
||||
|
||||
/*! @function
|
||||
@abstract Get the number of elements in the hash table
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@return Number of elements in the hash table [khint_t]
|
||||
*/
|
||||
#define kh_size(h) ((h)->size)
|
||||
|
||||
/*! @function
|
||||
@abstract Get the number of buckets in the hash table
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@return Number of buckets in the hash table [khint_t]
|
||||
*/
|
||||
#define kh_n_buckets(h) ((h)->n_buckets)
|
||||
|
||||
/*! @function
|
||||
@abstract Iterate over the entries in the hash table
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@param kvar Variable to which key will be assigned
|
||||
@param vvar Variable to which value will be assigned
|
||||
@param code Block of code to execute
|
||||
*/
|
||||
#define kh_foreach(h, kvar, vvar, code) { khint_t __i; \
|
||||
for (__i = kh_begin(h); __i != kh_end(h); ++__i) { \
|
||||
if (!kh_exist(h,__i)) continue; \
|
||||
(kvar) = kh_key(h,__i); \
|
||||
(vvar) = kh_val(h,__i); \
|
||||
code; \
|
||||
} }
|
||||
|
||||
/*! @function
|
||||
@abstract Iterate over the values in the hash table
|
||||
@param h Pointer to the hash table [khash_t(name)*]
|
||||
@param vvar Variable to which value will be assigned
|
||||
@param code Block of code to execute
|
||||
*/
|
||||
#define kh_foreach_value(h, vvar, code) { khint_t __i; \
|
||||
for (__i = kh_begin(h); __i != kh_end(h); ++__i) { \
|
||||
if (!kh_exist(h,__i)) continue; \
|
||||
(vvar) = kh_val(h,__i); \
|
||||
code; \
|
||||
} }
|
||||
|
||||
/* More convenient interfaces */
|
||||
|
||||
/*! @function
|
||||
@abstract Instantiate a hash set containing integer keys
|
||||
@param name Name of the hash table [symbol]
|
||||
*/
|
||||
#define KHASH_SET_INIT_INT(name) \
|
||||
KHASH_INIT(name, khint32_t, char, 0, kh_int_hash_func, kh_int_hash_equal)
|
||||
|
||||
/*! @function
|
||||
@abstract Instantiate a hash map containing integer keys
|
||||
@param name Name of the hash table [symbol]
|
||||
@param khval_t Type of values [type]
|
||||
*/
|
||||
#define KHASH_MAP_INIT_INT(name, khval_t) \
|
||||
KHASH_INIT(name, khint32_t, khval_t, 1, kh_int_hash_func, kh_int_hash_equal)
|
||||
|
||||
/*! @function
|
||||
@abstract Instantiate a hash set containing 64-bit integer keys
|
||||
@param name Name of the hash table [symbol]
|
||||
*/
|
||||
#define KHASH_SET_INIT_INT64(name) \
|
||||
KHASH_INIT(name, khint64_t, char, 0, kh_int64_hash_func, kh_int64_hash_equal)
|
||||
|
||||
/*! @function
|
||||
@abstract Instantiate a hash map containing 64-bit integer keys
|
||||
@param name Name of the hash table [symbol]
|
||||
@param khval_t Type of values [type]
|
||||
*/
|
||||
#define KHASH_MAP_INIT_INT64(name, khval_t) \
|
||||
KHASH_INIT(name, khint64_t, khval_t, 1, kh_int64_hash_func, kh_int64_hash_equal)
|
||||
|
||||
typedef const char *kh_cstr_t;
|
||||
/*! @function
|
||||
@abstract Instantiate a hash map containing const char* keys
|
||||
@param name Name of the hash table [symbol]
|
||||
*/
|
||||
#define KHASH_SET_INIT_STR(name) \
|
||||
KHASH_INIT(name, kh_cstr_t, char, 0, kh_str_hash_func, kh_str_hash_equal)
|
||||
|
||||
/*! @function
|
||||
@abstract Instantiate a hash map containing const char* keys
|
||||
@param name Name of the hash table [symbol]
|
||||
@param khval_t Type of values [type]
|
||||
*/
|
||||
#define KHASH_MAP_INIT_STR(name, khval_t) \
|
||||
KHASH_INIT(name, kh_cstr_t, khval_t, 1, kh_str_hash_func, kh_str_hash_equal)
|
||||
|
||||
|
||||
|
||||
|
||||
#define kh_write(name, h, fp) kh_write_##name(h, fp)
|
||||
|
||||
#define kh_load(name, h, fp) kh_load_##name(h, fp)
|
||||
|
||||
|
||||
#endif /* __AC_KHASH_H */
|
||||
352
khashl.h
Normal file
352
khashl.h
Normal file
@@ -0,0 +1,352 @@
|
||||
/* The MIT License
|
||||
|
||||
Copyright (c) 2019 by Attractive Chaos <attractor@live.co.uk>
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining
|
||||
a copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be
|
||||
included in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
|
||||
EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
|
||||
NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS
|
||||
BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN
|
||||
ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
|
||||
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
*/
|
||||
|
||||
#ifndef __AC_KHASHL_H
|
||||
#define __AC_KHASHL_H
|
||||
|
||||
#define AC_VERSION_KHASHL_H "0.1"
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <limits.h>
|
||||
|
||||
/************************************
|
||||
* Compiler specific configurations *
|
||||
************************************/
|
||||
|
||||
#if UINT_MAX == 0xffffffffu
|
||||
typedef unsigned int khint32_t;
|
||||
#elif ULONG_MAX == 0xffffffffu
|
||||
typedef unsigned long khint32_t;
|
||||
#endif
|
||||
|
||||
#if ULONG_MAX == ULLONG_MAX
|
||||
typedef unsigned long khint64_t;
|
||||
#else
|
||||
typedef unsigned long long khint64_t;
|
||||
#endif
|
||||
|
||||
#ifndef kh_inline
|
||||
#ifdef _MSC_VER
|
||||
#define kh_inline __inline
|
||||
#else
|
||||
#define kh_inline inline
|
||||
#endif
|
||||
#endif /* kh_inline */
|
||||
|
||||
#ifndef klib_unused
|
||||
#if (defined __clang__ && __clang_major__ >= 3) || (defined __GNUC__ && __GNUC__ >= 3)
|
||||
#define klib_unused __attribute__ ((__unused__))
|
||||
#else
|
||||
#define klib_unused
|
||||
#endif
|
||||
#endif /* klib_unused */
|
||||
|
||||
#define KH_LOCAL static kh_inline klib_unused
|
||||
|
||||
typedef khint32_t khint_t;
|
||||
|
||||
/******************
|
||||
* malloc aliases *
|
||||
******************/
|
||||
|
||||
#ifndef kcalloc
|
||||
#define kcalloc(N,Z) calloc(N,Z)
|
||||
#endif
|
||||
#ifndef kmalloc
|
||||
#define kmalloc(Z) malloc(Z)
|
||||
#endif
|
||||
#ifndef krealloc
|
||||
#define krealloc(P,Z) realloc(P,Z)
|
||||
#endif
|
||||
#ifndef kfree
|
||||
#define kfree(P) free(P)
|
||||
#endif
|
||||
|
||||
/****************************
|
||||
* Simple private functions *
|
||||
****************************/
|
||||
|
||||
#define __kh_used(flag, i) (flag[i>>5] >> (i&0x1fU) & 1U)
|
||||
#define __kh_set_used(flag, i) (flag[i>>5] |= 1U<<(i&0x1fU))
|
||||
#define __kh_set_unused(flag, i) (flag[i>>5] &= ~(1U<<(i&0x1fU)))
|
||||
|
||||
#define __kh_fsize(m) ((m) < 32? 1 : (m)>>5)
|
||||
|
||||
static kh_inline khint_t __kh_h2b(khint_t hash, khint_t bits) { return hash * 2654435769U >> (32 - bits); }
|
||||
|
||||
/*******************
|
||||
* Hash table base *
|
||||
*******************/
|
||||
|
||||
#define __KHASHL_TYPE(HType, khkey_t) \
|
||||
typedef struct HType { \
|
||||
khint_t bits, count; \
|
||||
khint32_t *used; \
|
||||
khkey_t *keys; \
|
||||
} HType;
|
||||
|
||||
#define __KHASHL_PROTOTYPES(HType, prefix, khkey_t) \
|
||||
extern HType *prefix##_init(void); \
|
||||
extern void prefix##_destroy(HType *h); \
|
||||
extern void prefix##_clear(HType *h); \
|
||||
extern khint_t prefix##_getp(const HType *h, const khkey_t *key); \
|
||||
extern int prefix##_resize(HType *h, khint_t new_n_buckets); \
|
||||
extern khint_t prefix##_putp(HType *h, const khkey_t *key, int *absent); \
|
||||
extern void prefix##_del(HType *h, khint_t k);
|
||||
|
||||
#define __KHASHL_IMPL_BASIC(SCOPE, HType, prefix) \
|
||||
SCOPE HType *prefix##_init(void) { \
|
||||
return (HType*)kcalloc(1, sizeof(HType)); \
|
||||
} \
|
||||
SCOPE void prefix##_destroy(HType *h) { \
|
||||
if (!h) return; \
|
||||
kfree((void *)h->keys); kfree(h->used); \
|
||||
kfree(h); \
|
||||
} \
|
||||
SCOPE void prefix##_clear(HType *h) { \
|
||||
if (h && h->used) { \
|
||||
uint32_t n_buckets = 1U << h->bits; \
|
||||
memset(h->used, 0, __kh_fsize(n_buckets) * sizeof(khint32_t)); \
|
||||
h->count = 0; \
|
||||
} \
|
||||
}
|
||||
|
||||
#define __KHASHL_IMPL_GET(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||
SCOPE khint_t prefix##_getp(const HType *h, const khkey_t *key) { \
|
||||
khint_t i, last, n_buckets, mask; \
|
||||
if (h->keys == 0) return 0; \
|
||||
n_buckets = 1U << h->bits; \
|
||||
mask = n_buckets - 1U; \
|
||||
i = last = __kh_h2b(__hash_fn(*key), h->bits); \
|
||||
while (__kh_used(h->used, i) && !__hash_eq(h->keys[i], *key)) { \
|
||||
i = (i + 1U) & mask; \
|
||||
if (i == last) return n_buckets; \
|
||||
} \
|
||||
return !__kh_used(h->used, i)? n_buckets : i; \
|
||||
} \
|
||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { return prefix##_getp(h, &key); }
|
||||
|
||||
#define __KHASHL_IMPL_RESIZE(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||
SCOPE int prefix##_resize(HType *h, khint_t new_n_buckets) { \
|
||||
khint32_t *new_used = 0; \
|
||||
khint_t j = 0, x = new_n_buckets, n_buckets, new_bits, new_mask; \
|
||||
while ((x >>= 1) != 0) ++j; \
|
||||
if (new_n_buckets & (new_n_buckets - 1)) ++j; \
|
||||
new_bits = j > 2? j : 2; \
|
||||
new_n_buckets = 1U << new_bits; \
|
||||
if (h->count > (new_n_buckets>>1) + (new_n_buckets>>2)) return 0; /* requested size is too small */ \
|
||||
new_used = (khint32_t*)kmalloc(__kh_fsize(new_n_buckets) * sizeof(khint32_t)); \
|
||||
memset(new_used, 0, __kh_fsize(new_n_buckets) * sizeof(khint32_t)); \
|
||||
if (!new_used) return -1; /* not enough memory */ \
|
||||
n_buckets = h->keys? 1U<<h->bits : 0U; \
|
||||
if (n_buckets < new_n_buckets) { /* expand */ \
|
||||
khkey_t *new_keys = (khkey_t*)krealloc((void*)h->keys, new_n_buckets * sizeof(khkey_t)); \
|
||||
if (!new_keys) { kfree(new_used); return -1; } \
|
||||
h->keys = new_keys; \
|
||||
} /* otherwise shrink */ \
|
||||
new_mask = new_n_buckets - 1; \
|
||||
for (j = 0; j != n_buckets; ++j) { \
|
||||
khkey_t key; \
|
||||
if (!__kh_used(h->used, j)) continue; \
|
||||
key = h->keys[j]; \
|
||||
__kh_set_unused(h->used, j); \
|
||||
while (1) { /* kick-out process; sort of like in Cuckoo hashing */ \
|
||||
khint_t i; \
|
||||
i = __kh_h2b(__hash_fn(key), new_bits); \
|
||||
while (__kh_used(new_used, i)) i = (i + 1) & new_mask; \
|
||||
__kh_set_used(new_used, i); \
|
||||
if (i < n_buckets && __kh_used(h->used, i)) { /* kick out the existing element */ \
|
||||
{ khkey_t tmp = h->keys[i]; h->keys[i] = key; key = tmp; } \
|
||||
__kh_set_unused(h->used, i); /* mark it as deleted in the old hash table */ \
|
||||
} else { /* write the element and jump out of the loop */ \
|
||||
h->keys[i] = key; \
|
||||
break; \
|
||||
} \
|
||||
} \
|
||||
} \
|
||||
if (n_buckets > new_n_buckets) /* shrink the hash table */ \
|
||||
h->keys = (khkey_t*)krealloc((void *)h->keys, new_n_buckets * sizeof(khkey_t)); \
|
||||
kfree(h->used); /* free the working space */ \
|
||||
h->used = new_used, h->bits = new_bits; \
|
||||
return 0; \
|
||||
}
|
||||
|
||||
#define __KHASHL_IMPL_PUT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||
SCOPE khint_t prefix##_putp(HType *h, const khkey_t *key, int *absent) { \
|
||||
khint_t n_buckets, i, last, mask; \
|
||||
n_buckets = h->keys? 1U<<h->bits : 0U; \
|
||||
*absent = -1; \
|
||||
if (h->count >= (n_buckets>>1) + (n_buckets>>2)) { /* rehashing */ \
|
||||
if (prefix##_resize(h, n_buckets + 1U) < 0) \
|
||||
return n_buckets; \
|
||||
n_buckets = 1U<<h->bits; \
|
||||
} /* TODO: to implement automatically shrinking; resize() already support shrinking */ \
|
||||
mask = n_buckets - 1; \
|
||||
i = last = __kh_h2b(__hash_fn(*key), h->bits); \
|
||||
while (__kh_used(h->used, i) && !__hash_eq(h->keys[i], *key)) { \
|
||||
i = (i + 1U) & mask; \
|
||||
if (i == last) break; \
|
||||
} \
|
||||
if (!__kh_used(h->used, i)) { /* not present at all */ \
|
||||
h->keys[i] = *key; \
|
||||
__kh_set_used(h->used, i); \
|
||||
++h->count; \
|
||||
*absent = 1; \
|
||||
} else *absent = 0; /* Don't touch h->keys[i] if present */ \
|
||||
return i; \
|
||||
} \
|
||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { return prefix##_putp(h, &key, absent); }
|
||||
|
||||
#define __KHASHL_IMPL_DEL(SCOPE, HType, prefix, khkey_t, __hash_fn) \
|
||||
SCOPE int prefix##_del(HType *h, khint_t i) { \
|
||||
khint_t j = i, k, mask, n_buckets; \
|
||||
if (h->keys == 0) return 0; \
|
||||
n_buckets = 1U<<h->bits; \
|
||||
mask = n_buckets - 1U; \
|
||||
while (1) { \
|
||||
j = (j + 1U) & mask; \
|
||||
if (j == i || !__kh_used(h->used, j)) break; /* j==i only when the table is completely full */ \
|
||||
k = __kh_h2b(__hash_fn(h->keys[j]), h->bits); \
|
||||
if ((j > i && (k <= i || k > j)) || (j < i && (k <= i && k > j))) \
|
||||
h->keys[i] = h->keys[j], i = j; \
|
||||
} \
|
||||
__kh_set_unused(h->used, i); \
|
||||
--h->count; \
|
||||
return 1; \
|
||||
}
|
||||
|
||||
#define KHASHL_DECLARE(HType, prefix, khkey_t) \
|
||||
__KHASHL_TYPE(HType, khkey_t) \
|
||||
__KHASHL_PROTOTYPES(HType, prefix, khkey_t)
|
||||
|
||||
#define KHASHL_INIT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||
__KHASHL_TYPE(HType, khkey_t) \
|
||||
__KHASHL_IMPL_BASIC(SCOPE, HType, prefix) \
|
||||
__KHASHL_IMPL_GET(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||
__KHASHL_IMPL_RESIZE(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||
__KHASHL_IMPL_PUT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||
__KHASHL_IMPL_DEL(SCOPE, HType, prefix, khkey_t, __hash_fn)
|
||||
|
||||
/*****************************
|
||||
* More convenient interface *
|
||||
*****************************/
|
||||
|
||||
#define __kh_packed __attribute__ ((__packed__))
|
||||
#define __kh_cached_hash(x) ((x).hash)
|
||||
|
||||
#define KHASHL_SET_INIT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||
typedef struct { khkey_t key; } __kh_packed HType##_s_bucket_t; \
|
||||
static kh_inline khint_t prefix##_s_hash(HType##_s_bucket_t x) { return __hash_fn(x.key); } \
|
||||
static kh_inline int prefix##_s_eq(HType##_s_bucket_t x, HType##_s_bucket_t y) { return __hash_eq(x.key, y.key); } \
|
||||
KHASHL_INIT(KH_LOCAL, HType, prefix##_s, HType##_s_bucket_t, prefix##_s_hash, prefix##_s_eq) \
|
||||
SCOPE HType *prefix##_init(void) { return prefix##_s_init(); } \
|
||||
SCOPE void prefix##_destroy(HType *h) { prefix##_s_destroy(h); } \
|
||||
SCOPE void prefix##_resize(HType *h, khint_t new_n_buckets) { prefix##_s_resize(h, new_n_buckets); } \
|
||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_s_bucket_t t; t.key = key; return prefix##_s_getp(h, &t); } \
|
||||
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_s_del(h, k); } \
|
||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_s_bucket_t t; t.key = key; return prefix##_s_putp(h, &t, absent); }
|
||||
|
||||
#define KHASHL_MAP_INIT(SCOPE, HType, prefix, khkey_t, kh_val_t, __hash_fn, __hash_eq) \
|
||||
typedef struct { khkey_t key; kh_val_t val; } __kh_packed HType##_m_bucket_t; \
|
||||
static kh_inline khint_t prefix##_m_hash(HType##_m_bucket_t x) { return __hash_fn(x.key); } \
|
||||
static kh_inline int prefix##_m_eq(HType##_m_bucket_t x, HType##_m_bucket_t y) { return __hash_eq(x.key, y.key); } \
|
||||
KHASHL_INIT(KH_LOCAL, HType, prefix##_m, HType##_m_bucket_t, prefix##_m_hash, prefix##_m_eq) \
|
||||
SCOPE HType *prefix##_init(void) { return prefix##_m_init(); } \
|
||||
SCOPE void prefix##_destroy(HType *h) { prefix##_m_destroy(h); } \
|
||||
SCOPE void prefix##_resize(HType *h, khint_t new_n_buckets) { prefix##_m_resize(h, new_n_buckets); } \
|
||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_m_bucket_t t; t.key = key; return prefix##_m_getp(h, &t); } \
|
||||
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_m_del(h, k); } \
|
||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_m_bucket_t t; t.key = key; return prefix##_m_putp(h, &t, absent); }
|
||||
|
||||
#define KHASHL_CSET_INIT(SCOPE, HType, prefix, khkey_t, __hash_fn, __hash_eq) \
|
||||
typedef struct { khkey_t key; khint_t hash; } __kh_packed HType##_cs_bucket_t; \
|
||||
static kh_inline int prefix##_cs_eq(HType##_cs_bucket_t x, HType##_cs_bucket_t y) { return x.hash == y.hash && __hash_eq(x.key, y.key); } \
|
||||
KHASHL_INIT(KH_LOCAL, HType, prefix##_cs, HType##_cs_bucket_t, __kh_cached_hash, prefix##_cs_eq) \
|
||||
SCOPE HType *prefix##_init(void) { return prefix##_cs_init(); } \
|
||||
SCOPE void prefix##_destroy(HType *h) { prefix##_cs_destroy(h); } \
|
||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_cs_bucket_t t; t.key = key; t.hash = __hash_fn(key); return prefix##_cs_getp(h, &t); } \
|
||||
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_cs_del(h, k); } \
|
||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_cs_bucket_t t; t.key = key, t.hash = __hash_fn(key); return prefix##_cs_putp(h, &t, absent); }
|
||||
|
||||
#define KHASHL_CMAP_INIT(SCOPE, HType, prefix, khkey_t, kh_val_t, __hash_fn, __hash_eq) \
|
||||
typedef struct { khkey_t key; kh_val_t val; khint_t hash; } __kh_packed HType##_cm_bucket_t; \
|
||||
static kh_inline int prefix##_cm_eq(HType##_cm_bucket_t x, HType##_cm_bucket_t y) { return x.hash == y.hash && __hash_eq(x.key, y.key); } \
|
||||
KHASHL_INIT(KH_LOCAL, HType, prefix##_cm, HType##_cm_bucket_t, __kh_cached_hash, prefix##_cm_eq) \
|
||||
SCOPE HType *prefix##_init(void) { return prefix##_cm_init(); } \
|
||||
SCOPE void prefix##_destroy(HType *h) { prefix##_cm_destroy(h); } \
|
||||
SCOPE khint_t prefix##_get(const HType *h, khkey_t key) { HType##_cm_bucket_t t; t.key = key; t.hash = __hash_fn(key); return prefix##_cm_getp(h, &t); } \
|
||||
SCOPE int prefix##_del(HType *h, khint_t k) { return prefix##_cm_del(h, k); } \
|
||||
SCOPE khint_t prefix##_put(HType *h, khkey_t key, int *absent) { HType##_cm_bucket_t t; t.key = key, t.hash = __hash_fn(key); return prefix##_cm_putp(h, &t, absent); }
|
||||
|
||||
/**************************
|
||||
* Public macro functions *
|
||||
**************************/
|
||||
|
||||
#define kh_bucket(h, x) ((h)->keys[x])
|
||||
#define kh_size(h) ((h)->count)
|
||||
#define kh_capacity(h) ((h)->keys? 1U<<(h)->bits : 0U)
|
||||
#define kh_end(h) kh_capacity(h)
|
||||
|
||||
#define kh_key(h, x) ((h)->keys[x].key)
|
||||
#define kh_val(h, x) ((h)->keys[x].val)
|
||||
#define kh_exist(h, x) __kh_used((h)->used, (x))
|
||||
|
||||
/**************************************
|
||||
* Common hash and equality functions *
|
||||
**************************************/
|
||||
|
||||
#define kh_eq_generic(a, b) ((a) == (b))
|
||||
#define kh_eq_str(a, b) (strcmp((a), (b)) == 0)
|
||||
#define kh_hash_dummy(x) ((khint_t)(x))
|
||||
|
||||
static kh_inline khint_t kh_hash_uint32(khint_t key) {
|
||||
key += ~(key << 15);
|
||||
key ^= (key >> 10);
|
||||
key += (key << 3);
|
||||
key ^= (key >> 6);
|
||||
key += ~(key << 11);
|
||||
key ^= (key >> 16);
|
||||
return key;
|
||||
}
|
||||
|
||||
static kh_inline khint_t kh_hash_uint64(khint64_t key) {
|
||||
key = ~key + (key << 21);
|
||||
key = key ^ key >> 24;
|
||||
key = (key + (key << 3)) + (key << 8);
|
||||
key = key ^ key >> 14;
|
||||
key = (key + (key << 2)) + (key << 4);
|
||||
key = key ^ key >> 28;
|
||||
key = key + (key << 31);
|
||||
return (khint_t)key;
|
||||
}
|
||||
|
||||
static kh_inline khint_t kh_hash_str(const char *s) {
|
||||
khint_t h = (khint_t)*s;
|
||||
if (h) for (++s ; *s; ++s) h = (h << 5) - h + (khint_t)*s;
|
||||
return h;
|
||||
}
|
||||
|
||||
#endif /* __AC_KHASHL_H */
|
||||
195
kmer.cpp
195
kmer.cpp
@@ -1,195 +0,0 @@
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include "kmer.h"
|
||||
|
||||
void init_HPC_seq(HPC_seq* seq, char* str, long long l)
|
||||
{
|
||||
seq->i = 0;
|
||||
seq->l = l;
|
||||
seq->N_occ = 0;
|
||||
seq->str = str;
|
||||
}
|
||||
|
||||
void init_Hash_code(Hash_code* code)
|
||||
{
|
||||
code->x[0] = 0;
|
||||
code->x[1] = 0;
|
||||
}
|
||||
|
||||
|
||||
|
||||
void init_small_hash_table(small_hash_table* x)
|
||||
{
|
||||
x->size = 0;
|
||||
x->buffer = NULL;
|
||||
x->length = 0;
|
||||
}
|
||||
|
||||
void clear_small_hash_table(small_hash_table* x)
|
||||
{
|
||||
x->length = 0;
|
||||
}
|
||||
|
||||
void resize_small_hash_table(small_hash_table* x, uint64_t size)
|
||||
{
|
||||
if(size > x->size)
|
||||
{
|
||||
x->size = size;
|
||||
x->buffer = (k_v*)realloc(x->buffer, x->size*sizeof(k_v));
|
||||
}
|
||||
}
|
||||
|
||||
void destory_small_hash_table(small_hash_table* x)
|
||||
{
|
||||
free(x->buffer);
|
||||
}
|
||||
|
||||
|
||||
void add_small_hash_table(small_hash_table* x, k_v* element)
|
||||
{
|
||||
if(x->length + 1 > x->size)
|
||||
{
|
||||
x->size = (x->length + 1) * 2;
|
||||
x->buffer = (k_v*)realloc(x->buffer, x->size*sizeof(k_v));
|
||||
}
|
||||
|
||||
x->buffer[x->length] = (*element);
|
||||
x->length++;
|
||||
}
|
||||
|
||||
//x > y, return 1; x < y, return -1, x == y, return 0
|
||||
int compare_k_mer(k_v* x, k_v* y)
|
||||
{
|
||||
if(x->key.x[1] != y->key.x[1])
|
||||
{
|
||||
return x->key.x[1] > y->key.x[1] ? 1: -1;
|
||||
}
|
||||
else
|
||||
{
|
||||
if(x->key.x[0] != y->key.x[0])
|
||||
{
|
||||
return x->key.x[0] > y->key.x[0] ? 1: -1;
|
||||
}
|
||||
else
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
int cmp_k_mer_kv(const void * a, const void * b)
|
||||
{
|
||||
int flag = compare_k_mer((k_v*)a, (k_v*)b);
|
||||
|
||||
if(flag == 0)
|
||||
{
|
||||
if ((*(k_v*)a).value != (*(k_v*)b).value)
|
||||
{
|
||||
return (*(k_v*)a).value > (*(k_v*)b).value ? 1: -1;
|
||||
}
|
||||
else
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
}
|
||||
else
|
||||
{
|
||||
return flag;
|
||||
}
|
||||
}
|
||||
|
||||
void sort_small_hash_table(small_hash_table* x)
|
||||
{
|
||||
qsort(x->buffer, x->length, sizeof(k_v), cmp_k_mer_kv);
|
||||
}
|
||||
|
||||
|
||||
inline long long firstEqual(k_v* arr, long long arrLen, k_v* key)
|
||||
{
|
||||
long long L = 0, R = arrLen - 1; //[L, R]
|
||||
long long mid;
|
||||
int flag;
|
||||
while( L <= R)
|
||||
{
|
||||
mid = L + (R - L)/2;
|
||||
|
||||
flag = compare_k_mer(&(arr[mid]), key);
|
||||
|
||||
///arr[mid] >= key
|
||||
if(flag >= 0)
|
||||
{
|
||||
R = mid - 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
L = mid + 1;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
if(L < arrLen && (flag = compare_k_mer(&(arr[L]), key) == 0))
|
||||
{
|
||||
return L;
|
||||
}
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
inline long long lastEqual(k_v* arr, long long arrLen, k_v* key)
|
||||
{
|
||||
long long L = 0, R = arrLen - 1; //[L, R]
|
||||
long long mid;
|
||||
int flag;
|
||||
while( L <= R)
|
||||
{
|
||||
mid = L + (R - L)/2;
|
||||
flag = compare_k_mer(&(arr[mid]), key);
|
||||
///arr[mid] <= key
|
||||
if(flag <= 0)
|
||||
{
|
||||
L = mid + 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
R = mid - 1;
|
||||
}
|
||||
}
|
||||
|
||||
if(R >= 0 && ((flag = compare_k_mer(&(arr[R]), key)) == 0))
|
||||
{
|
||||
return R;
|
||||
}
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
int query_small_hash_table(small_hash_table* target, k_v* query, long long* l_end, long long* r_end)
|
||||
{
|
||||
(*l_end) = -1;
|
||||
(*r_end) = -1;
|
||||
long long left_end;
|
||||
long long right_end;
|
||||
|
||||
left_end = firstEqual(target->buffer, target->length, query);
|
||||
|
||||
if(left_end != -1)
|
||||
{
|
||||
right_end = lastEqual(target->buffer + left_end, target->length - left_end, query) + left_end;
|
||||
|
||||
(*l_end) = left_end;
|
||||
(*r_end) = right_end;
|
||||
|
||||
if(right_end == -1)
|
||||
{
|
||||
fprintf(stderr, "error\n");
|
||||
}
|
||||
|
||||
|
||||
return right_end - left_end + 1;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
114
kmer.h
114
kmer.h
@@ -1,114 +0,0 @@
|
||||
#ifndef __KMER__
|
||||
#define __KMER__
|
||||
#include "Process_Read.h"
|
||||
|
||||
///#define ALL (0xffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffffff)
|
||||
#define ALL (0xffffffffffffffff)
|
||||
/****************************may have bugs********************************/
|
||||
#define SAFE_SHIFT(k) k & ((k < 64)?ALL:0)
|
||||
/****************************may have bugs********************************/
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
//can represent at most 64-mer
|
||||
uint64_t x[2];
|
||||
} Hash_code;
|
||||
|
||||
typedef struct {
|
||||
Hash_code key; ///k-mer itself
|
||||
uint64_t value; ///offset
|
||||
} k_v;
|
||||
|
||||
typedef struct {
|
||||
k_v* buffer;
|
||||
uint32_t size;
|
||||
uint32_t length;
|
||||
} small_hash_table;
|
||||
|
||||
void init_small_hash_table(small_hash_table* x);
|
||||
void clear_small_hash_table(small_hash_table* x);
|
||||
void resize_small_hash_table(small_hash_table* x, uint64_t size);
|
||||
void destory_small_hash_table(small_hash_table* x);
|
||||
void add_small_hash_table(small_hash_table* x, k_v* element);
|
||||
void sort_small_hash_table(small_hash_table* x);
|
||||
int compare_k_mer(k_v* x, k_v* y);
|
||||
int query_small_hash_table(small_hash_table* target, k_v* query, long long* l_end, long long* r_end);
|
||||
|
||||
|
||||
typedef struct
|
||||
{
|
||||
char* str;
|
||||
long long l;
|
||||
long long i;
|
||||
long long N_occ;
|
||||
|
||||
} HPC_seq;
|
||||
|
||||
|
||||
inline uint64_t get_HPC_code(HPC_seq* seq, uint64_t* end_pos)
|
||||
{
|
||||
|
||||
if(seq->i < seq ->l)
|
||||
{
|
||||
uint8_t code = seq_nt6_table[(uint8_t)seq->str[seq->i]];
|
||||
|
||||
(*end_pos) = seq->i;
|
||||
|
||||
for (; seq->i < seq->l; seq->i++)
|
||||
{
|
||||
///number of Ns
|
||||
if (seq_nt6_table[(uint8_t)seq->str[seq->i]] >= 4)
|
||||
{
|
||||
seq->N_occ++;
|
||||
}
|
||||
|
||||
if (seq_nt6_table[(uint8_t)seq->str[seq->i]] != code)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return (uint64_t)code;
|
||||
}
|
||||
else
|
||||
{
|
||||
///end
|
||||
return 6;
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
inline void k_mer_append(Hash_code* code, uint64_t c, int k)
|
||||
{
|
||||
|
||||
uint64_t mask = ALL >> (64 -k);
|
||||
|
||||
code->x[0] = ((code->x[0]<<1) | (c&1)) & mask;
|
||||
code->x[1] = ((code->x[1]<<1) | (c>>1)) & mask;
|
||||
}
|
||||
|
||||
inline void Hashcode_to_string(Hash_code* code, char* str, int k)
|
||||
{
|
||||
uint8_t c;
|
||||
int i;
|
||||
for (i = 0; i < k; i++)
|
||||
{
|
||||
c = (code->x[1] >> (k - i - 1)) & ((uint64_t)1);
|
||||
c = c << 1;
|
||||
c = c | ((code->x[0] >> (k - i - 1)) & ((uint64_t)1));
|
||||
|
||||
str[i] = s_H[c];
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
void init_HPC_seq(HPC_seq* seq, char* str, long long l);
|
||||
void init_Hash_code(Hash_code* code);
|
||||
|
||||
|
||||
#endif
|
||||
177
ksw2.h
Normal file
177
ksw2.h
Normal file
@@ -0,0 +1,177 @@
|
||||
#ifndef KSW2_H_
|
||||
#define KSW2_H_
|
||||
|
||||
#include <stdint.h>
|
||||
|
||||
#define KSW_NEG_INF -0x40000000
|
||||
|
||||
#define KSW_EZ_SCORE_ONLY 0x01 // don't record alignment path/cigar
|
||||
#define KSW_EZ_RIGHT 0x02 // right-align gaps
|
||||
#define KSW_EZ_GENERIC_SC 0x04 // without this flag: match/mismatch only; last symbol is a wildcard
|
||||
#define KSW_EZ_APPROX_MAX 0x08 // approximate max; this is faster with sse
|
||||
#define KSW_EZ_APPROX_DROP 0x10 // approximate Z-drop; faster with sse
|
||||
#define KSW_EZ_EXTZ_ONLY 0x40 // only perform extension
|
||||
#define KSW_EZ_REV_CIGAR 0x80 // reverse CIGAR in the output
|
||||
#define KSW_EZ_SPLICE_FOR 0x100
|
||||
#define KSW_EZ_SPLICE_REV 0x200
|
||||
#define KSW_EZ_SPLICE_FLANK 0x400
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
typedef struct {
|
||||
uint32_t max:31, zdropped:1;
|
||||
int max_q, max_t; // max extension coordinate
|
||||
int mqe, mqe_t; // max score when reaching the end of query
|
||||
int mte, mte_q; // max score when reaching the end of target
|
||||
int score; // max score reaching both ends; may be KSW_NEG_INF
|
||||
int m_cigar, n_cigar;
|
||||
int reach_end;
|
||||
uint32_t *cigar;
|
||||
} ksw_extz_t;
|
||||
|
||||
/**
|
||||
* NW-like extension
|
||||
*
|
||||
* @param km memory pool, when used with kalloc
|
||||
* @param qlen query length
|
||||
* @param query query sequence with 0 <= query[i] < m
|
||||
* @param tlen target length
|
||||
* @param target target sequence with 0 <= target[i] < m
|
||||
* @param m number of residue types
|
||||
* @param mat m*m scoring mattrix in one-dimension array
|
||||
* @param gapo gap open penalty; a gap of length l cost "-(gapo+l*gape)"
|
||||
* @param gape gap extension penalty
|
||||
* @param w band width (<0 to disable)
|
||||
* @param zdrop off-diagonal drop-off to stop extension (positive; <0 to disable)
|
||||
* @param flag flag (see KSW_EZ_* macros)
|
||||
* @param ez (out) scores and cigar
|
||||
*/
|
||||
void ksw_extz(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int w, int zdrop, int flag, ksw_extz_t *ez);
|
||||
|
||||
void ksw_extz2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||
|
||||
void ksw_extd(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t gape2, int w, int zdrop, int flag, ksw_extz_t *ez);
|
||||
|
||||
void ksw_extd2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t gape2, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez);
|
||||
|
||||
void ksw_exts2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat,
|
||||
int8_t gapo, int8_t gape, int8_t gapo2, int8_t noncan, int zdrop, int flag, ksw_extz_t *ez);
|
||||
|
||||
void ksw_extf2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t mch, int8_t mis, int8_t e, int w, int xdrop, ksw_extz_t *ez);
|
||||
|
||||
/**
|
||||
* Global alignment
|
||||
*
|
||||
* (first 10 parameters identical to ksw_extz_sse())
|
||||
* @param m_cigar (modified) max CIGAR length; feed 0 if cigar==0
|
||||
* @param n_cigar (out) number of CIGAR elements
|
||||
* @param cigar (out) BAM-encoded CIGAR; caller need to deallocate with kfree(km, )
|
||||
*
|
||||
* @return score of the alignment
|
||||
*/
|
||||
int ksw_gg(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t gapo, int8_t gape, int w, int *m_cigar_, int *n_cigar_, uint32_t **cigar_);
|
||||
int ksw_gg2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t gapo, int8_t gape, int w, int *m_cigar_, int *n_cigar_, uint32_t **cigar_);
|
||||
int ksw_gg2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t gapo, int8_t gape, int w, int *m_cigar_, int *n_cigar_, uint32_t **cigar_);
|
||||
|
||||
void *ksw_ll_qinit(void *km, int size, int qlen, const uint8_t *query, int m, const int8_t *mat);
|
||||
int ksw_ll_i16(void *q, int tlen, const uint8_t *target, int gapo, int gape, int *qe, int *te);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
/************************************
|
||||
*** Private macros and functions ***
|
||||
************************************/
|
||||
|
||||
#ifdef HAVE_KALLOC
|
||||
#include "kalloc.h"
|
||||
#else
|
||||
#include <stdlib.h>
|
||||
#define kmalloc(km, size) malloc((size))
|
||||
#define kcalloc(km, count, size) calloc((count), (size))
|
||||
#define krealloc(km, ptr, size) realloc((ptr), (size))
|
||||
#define kfree(km, ptr) free((ptr))
|
||||
#endif
|
||||
|
||||
static inline uint32_t *ksw_push_cigar(void *km, int *n_cigar, int *m_cigar, uint32_t *cigar, uint32_t op, int len)
|
||||
{
|
||||
if (*n_cigar == 0 || op != (cigar[(*n_cigar) - 1]&0xf)) {
|
||||
if (*n_cigar == *m_cigar) {
|
||||
*m_cigar = *m_cigar? (*m_cigar)<<1 : 4;
|
||||
cigar = (uint32_t*)krealloc(km, cigar, (*m_cigar) << 2);
|
||||
}
|
||||
cigar[(*n_cigar)++] = len<<4 | op;
|
||||
} else cigar[(*n_cigar)-1] += len<<4;
|
||||
return cigar;
|
||||
}
|
||||
|
||||
// In the backtrack matrix, value p[] has the following structure:
|
||||
// bit 0-2: which type gets the max - 0 for H, 1 for E, 2 for F, 3 for \tilde{E} and 4 for \tilde{F}
|
||||
// bit 3/0x08: 1 if a continuation on the E state (bit 5/0x20 for a continuation on \tilde{E})
|
||||
// bit 4/0x10: 1 if a continuation on the F state (bit 6/0x40 for a continuation on \tilde{F})
|
||||
static inline void ksw_backtrack(void *km, int is_rot, int is_rev, int min_intron_len, const uint8_t *p, const int *off, const int *off_end, int n_col, int i0, int j0,
|
||||
int *m_cigar_, int *n_cigar_, uint32_t **cigar_)
|
||||
{ // p[] - lower 3 bits: which type gets the max; bit
|
||||
int n_cigar = 0, m_cigar = *m_cigar_, i = i0, j = j0, r, state = 0;
|
||||
uint32_t *cigar = *cigar_, tmp;
|
||||
while (i >= 0 && j >= 0) { // at the beginning of the loop, _state_ tells us which state to check
|
||||
int force_state = -1;
|
||||
if (is_rot) {
|
||||
r = i + j;
|
||||
if (i < off[r]) force_state = 2;
|
||||
if (off_end && i > off_end[r]) force_state = 1;
|
||||
tmp = force_state < 0? p[(size_t)r * n_col + i - off[r]] : 0;
|
||||
} else {
|
||||
if (j < off[i]) force_state = 2;
|
||||
if (off_end && j > off_end[i]) force_state = 1;
|
||||
tmp = force_state < 0? p[(size_t)i * n_col + j - off[i]] : 0;
|
||||
}
|
||||
if (state == 0) state = tmp & 7; // if requesting the H state, find state one maximizes it.
|
||||
else if (!(tmp >> (state + 2) & 1)) state = 0; // if requesting other states, _state_ stays the same if it is a continuation; otherwise, set to H
|
||||
if (state == 0) state = tmp & 7; // TODO: probably this line can be merged into the "else if" line right above; not 100% sure
|
||||
if (force_state >= 0) state = force_state;
|
||||
if (state == 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 0, 1), --i, --j; // match
|
||||
else if (state == 1 || (state == 3 && min_intron_len <= 0)) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 2, 1), --i; // deletion
|
||||
else if (state == 3 && min_intron_len > 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 3, 1), --i; // intron
|
||||
else cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, 1), --j; // insertion
|
||||
}
|
||||
if (i >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, min_intron_len > 0 && i >= min_intron_len? 3 : 2, i + 1); // first deletion
|
||||
if (j >= 0) cigar = ksw_push_cigar(km, &n_cigar, &m_cigar, cigar, 1, j + 1); // first insertion
|
||||
if (!is_rev)
|
||||
for (i = 0; i < n_cigar>>1; ++i) // reverse CIGAR
|
||||
tmp = cigar[i], cigar[i] = cigar[n_cigar-1-i], cigar[n_cigar-1-i] = tmp;
|
||||
*m_cigar_ = m_cigar, *n_cigar_ = n_cigar, *cigar_ = cigar;
|
||||
}
|
||||
|
||||
static inline void ksw_reset_extz(ksw_extz_t *ez)
|
||||
{
|
||||
ez->max_q = ez->max_t = ez->mqe_t = ez->mte_q = -1;
|
||||
ez->max = 0, ez->score = ez->mqe = ez->mte = KSW_NEG_INF;
|
||||
ez->n_cigar = 0, ez->zdropped = 0, ez->reach_end = 0;
|
||||
}
|
||||
|
||||
static inline int ksw_apply_zdrop(ksw_extz_t *ez, int is_rot, int32_t H, int a, int b, int zdrop, int8_t e)
|
||||
{
|
||||
int r, t;
|
||||
if (is_rot) r = a, t = b;
|
||||
else r = a + b, t = a;
|
||||
if (H > (int32_t)ez->max) {
|
||||
ez->max = H, ez->max_t = t, ez->max_q = r - t;
|
||||
} else if (t >= ez->max_t && r - t >= ez->max_q) {
|
||||
int tl = t - ez->max_t, ql = (r - t) - ez->max_q, l;
|
||||
l = tl > ql? tl - ql : ql - tl;
|
||||
if (zdrop >= 0 && ez->max - H > zdrop + l * e) {
|
||||
ez->zdropped = 1;
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
#endif
|
||||
305
ksw2_extz2_sse.c
Normal file
305
ksw2_extz2_sse.c
Normal file
@@ -0,0 +1,305 @@
|
||||
#include <string.h>
|
||||
#include <assert.h>
|
||||
#include "ksw2.h"
|
||||
|
||||
#ifdef __SSE2__
|
||||
#include <emmintrin.h>
|
||||
|
||||
#ifdef KSW_SSE2_ONLY
|
||||
#undef __SSE4_1__
|
||||
#endif
|
||||
|
||||
#ifdef __SSE4_1__
|
||||
#include <smmintrin.h>
|
||||
#endif
|
||||
|
||||
#ifdef KSW_CPU_DISPATCH
|
||||
#ifdef __SSE4_1__
|
||||
void ksw_extz2_sse41(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez)
|
||||
#else
|
||||
void ksw_extz2_sse2(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez)
|
||||
#endif
|
||||
#else
|
||||
void ksw_extz2_sse(void *km, int qlen, const uint8_t *query, int tlen, const uint8_t *target, int8_t m, const int8_t *mat, int8_t q, int8_t e, int w, int zdrop, int end_bonus, int flag, ksw_extz_t *ez)
|
||||
#endif // ~KSW_CPU_DISPATCH
|
||||
{
|
||||
#define __dp_code_block1 \
|
||||
z = _mm_add_epi8(_mm_load_si128(&s[t]), qe2_); \
|
||||
xt1 = _mm_load_si128(&x[t]); /* xt1 <- x[r-1][t..t+15] */ \
|
||||
tmp = _mm_srli_si128(xt1, 15); /* tmp <- x[r-1][t+15] */ \
|
||||
xt1 = _mm_or_si128(_mm_slli_si128(xt1, 1), x1_); /* xt1 <- x[r-1][t-1..t+14] */ \
|
||||
x1_ = tmp; \
|
||||
vt1 = _mm_load_si128(&v[t]); /* vt1 <- v[r-1][t..t+15] */ \
|
||||
tmp = _mm_srli_si128(vt1, 15); /* tmp <- v[r-1][t+15] */ \
|
||||
vt1 = _mm_or_si128(_mm_slli_si128(vt1, 1), v1_); /* vt1 <- v[r-1][t-1..t+14] */ \
|
||||
v1_ = tmp; \
|
||||
a = _mm_add_epi8(xt1, vt1); /* a <- x[r-1][t-1..t+14] + v[r-1][t-1..t+14] */ \
|
||||
ut = _mm_load_si128(&u[t]); /* ut <- u[t..t+15] */ \
|
||||
b = _mm_add_epi8(_mm_load_si128(&y[t]), ut); /* b <- y[r-1][t..t+15] + u[r-1][t..t+15] */
|
||||
|
||||
#define __dp_code_block2 \
|
||||
z = _mm_max_epu8(z, b); /* z = max(z, b); this works because both are non-negative */ \
|
||||
z = _mm_min_epu8(z, max_sc_); \
|
||||
_mm_store_si128(&u[t], _mm_sub_epi8(z, vt1)); /* u[r][t..t+15] <- z - v[r-1][t-1..t+14] */ \
|
||||
_mm_store_si128(&v[t], _mm_sub_epi8(z, ut)); /* v[r][t..t+15] <- z - u[r-1][t..t+15] */ \
|
||||
z = _mm_sub_epi8(z, q_); \
|
||||
a = _mm_sub_epi8(a, z); \
|
||||
b = _mm_sub_epi8(b, z);
|
||||
|
||||
int r, t, qe = q + e, n_col_, *off = 0, *off_end = 0, tlen_, qlen_, last_st, last_en, wl, wr, max_sc, min_sc;
|
||||
int with_cigar = !(flag&KSW_EZ_SCORE_ONLY), approx_max = !!(flag&KSW_EZ_APPROX_MAX);
|
||||
int32_t *H = 0, H0 = 0, last_H0_t = 0;
|
||||
uint8_t *qr, *sf, *mem, *mem2 = 0;
|
||||
__m128i q_, qe2_, zero_, flag1_, flag2_, flag8_, flag16_, sc_mch_, sc_mis_, sc_N_, m1_, max_sc_;
|
||||
__m128i *u, *v, *x, *y, *s, *p = 0;
|
||||
|
||||
ksw_reset_extz(ez);
|
||||
if (m <= 0 || qlen <= 0 || tlen <= 0) return;
|
||||
|
||||
zero_ = _mm_set1_epi8(0);
|
||||
q_ = _mm_set1_epi8(q);
|
||||
qe2_ = _mm_set1_epi8((q + e) * 2);
|
||||
flag1_ = _mm_set1_epi8(1);
|
||||
flag2_ = _mm_set1_epi8(2);
|
||||
flag8_ = _mm_set1_epi8(0x08);
|
||||
flag16_ = _mm_set1_epi8(0x10);
|
||||
sc_mch_ = _mm_set1_epi8(mat[0]);
|
||||
sc_mis_ = _mm_set1_epi8(mat[1]);
|
||||
sc_N_ = mat[m*m-1] == 0? _mm_set1_epi8(-e) : _mm_set1_epi8(mat[m*m-1]);
|
||||
m1_ = _mm_set1_epi8(m - 1); // wildcard
|
||||
max_sc_ = _mm_set1_epi8(mat[0] + (q + e) * 2);
|
||||
|
||||
if (w < 0) w = tlen > qlen? tlen : qlen;
|
||||
wl = wr = w;
|
||||
tlen_ = (tlen + 15) / 16;
|
||||
n_col_ = qlen < tlen? qlen : tlen;
|
||||
n_col_ = ((n_col_ < w + 1? n_col_ : w + 1) + 15) / 16 + 1;
|
||||
qlen_ = (qlen + 15) / 16;
|
||||
for (t = 1, max_sc = mat[0], min_sc = mat[1]; t < m * m; ++t) {
|
||||
max_sc = max_sc > mat[t]? max_sc : mat[t];
|
||||
min_sc = min_sc < mat[t]? min_sc : mat[t];
|
||||
}
|
||||
if (-min_sc > 2 * (q + e)) return; // otherwise, we won't see any mismatches
|
||||
|
||||
mem = (uint8_t*)kcalloc(km, tlen_ * 6 + qlen_ + 1, 16);
|
||||
u = (__m128i*)(((size_t)mem + 15) >> 4 << 4); // 16-byte aligned
|
||||
v = u + tlen_, x = v + tlen_, y = x + tlen_, s = y + tlen_, sf = (uint8_t*)(s + tlen_), qr = sf + tlen_ * 16;
|
||||
if (!approx_max) {
|
||||
H = (int32_t*)kmalloc(km, tlen_ * 16 * 4);
|
||||
for (t = 0; t < tlen_ * 16; ++t) H[t] = KSW_NEG_INF;
|
||||
}
|
||||
if (with_cigar) {
|
||||
mem2 = (uint8_t*)kmalloc(km, ((size_t)(qlen + tlen - 1) * n_col_ + 1) * 16);
|
||||
p = (__m128i*)(((size_t)mem2 + 15) >> 4 << 4);
|
||||
off = (int*)kmalloc(km, (qlen + tlen - 1) * sizeof(int) * 2);
|
||||
off_end = off + qlen + tlen - 1;
|
||||
}
|
||||
|
||||
for (t = 0; t < qlen; ++t) qr[t] = query[qlen - 1 - t];
|
||||
memcpy(sf, target, tlen);
|
||||
|
||||
for (r = 0, last_st = last_en = -1; r < qlen + tlen - 1; ++r) {
|
||||
int st = 0, en = tlen - 1, st0, en0, st_, en_;
|
||||
int8_t x1, v1;
|
||||
uint8_t *qrr = qr + (qlen - 1 - r), *u8 = (uint8_t*)u, *v8 = (uint8_t*)v;
|
||||
__m128i x1_, v1_;
|
||||
// find the boundaries
|
||||
if (st < r - qlen + 1) st = r - qlen + 1;
|
||||
if (en > r) en = r;
|
||||
if (st < (r-wr+1)>>1) st = (r-wr+1)>>1; // take the ceil
|
||||
if (en > (r+wl)>>1) en = (r+wl)>>1; // take the floor
|
||||
if (st > en) {
|
||||
ez->zdropped = 1;
|
||||
break;
|
||||
}
|
||||
st0 = st, en0 = en;
|
||||
st = st / 16 * 16, en = (en + 16) / 16 * 16 - 1;
|
||||
// set boundary conditions
|
||||
if (st > 0) {
|
||||
if (st - 1 >= last_st && st - 1 <= last_en)
|
||||
x1 = ((uint8_t*)x)[st - 1], v1 = v8[st - 1]; // (r-1,s-1) calculated in the last round
|
||||
else x1 = v1 = 0; // not calculated; set to zeros
|
||||
} else x1 = 0, v1 = r? q : 0;
|
||||
if (en >= r) ((uint8_t*)y)[r] = 0, u8[r] = r? q : 0;
|
||||
// loop fission: set scores first
|
||||
if (!(flag & KSW_EZ_GENERIC_SC)) {
|
||||
for (t = st0; t <= en0; t += 16) {
|
||||
__m128i sq, st, tmp, mask;
|
||||
sq = _mm_loadu_si128((__m128i*)&sf[t]);
|
||||
st = _mm_loadu_si128((__m128i*)&qrr[t]);
|
||||
mask = _mm_or_si128(_mm_cmpeq_epi8(sq, m1_), _mm_cmpeq_epi8(st, m1_));
|
||||
tmp = _mm_cmpeq_epi8(sq, st);
|
||||
#ifdef __SSE4_1__
|
||||
tmp = _mm_blendv_epi8(sc_mis_, sc_mch_, tmp);
|
||||
tmp = _mm_blendv_epi8(tmp, sc_N_, mask);
|
||||
#else
|
||||
tmp = _mm_or_si128(_mm_andnot_si128(tmp, sc_mis_), _mm_and_si128(tmp, sc_mch_));
|
||||
tmp = _mm_or_si128(_mm_andnot_si128(mask, tmp), _mm_and_si128(mask, sc_N_));
|
||||
#endif
|
||||
_mm_storeu_si128((__m128i*)((uint8_t*)s + t), tmp);
|
||||
}
|
||||
} else {
|
||||
for (t = st0; t <= en0; ++t)
|
||||
((uint8_t*)s)[t] = mat[sf[t] * m + qrr[t]];
|
||||
}
|
||||
// core loop
|
||||
x1_ = _mm_cvtsi32_si128(x1);
|
||||
v1_ = _mm_cvtsi32_si128(v1);
|
||||
st_ = st / 16, en_ = en / 16;
|
||||
assert(en_ - st_ + 1 <= n_col_);
|
||||
if (!with_cigar) { // score only
|
||||
for (t = st_; t <= en_; ++t) {
|
||||
__m128i z, a, b, xt1, vt1, ut, tmp;
|
||||
__dp_code_block1;
|
||||
#ifdef __SSE4_1__
|
||||
z = _mm_max_epi8(z, a); // z = z > a? z : a (signed)
|
||||
#else // we need to emulate SSE4.1 intrinsics _mm_max_epi8()
|
||||
z = _mm_and_si128(z, _mm_cmpgt_epi8(z, zero_)); // z = z > 0? z : 0;
|
||||
z = _mm_max_epu8(z, a); // z = max(z, a); this works because both are non-negative
|
||||
#endif
|
||||
__dp_code_block2;
|
||||
#ifdef __SSE4_1__
|
||||
_mm_store_si128(&x[t], _mm_max_epi8(a, zero_));
|
||||
_mm_store_si128(&y[t], _mm_max_epi8(b, zero_));
|
||||
#else
|
||||
tmp = _mm_cmpgt_epi8(a, zero_);
|
||||
_mm_store_si128(&x[t], _mm_and_si128(a, tmp));
|
||||
tmp = _mm_cmpgt_epi8(b, zero_);
|
||||
_mm_store_si128(&y[t], _mm_and_si128(b, tmp));
|
||||
#endif
|
||||
}
|
||||
} else if (!(flag&KSW_EZ_RIGHT)) { // gap left-alignment
|
||||
__m128i *pr = p + (size_t)r * n_col_ - st_;
|
||||
off[r] = st, off_end[r] = en;
|
||||
for (t = st_; t <= en_; ++t) {
|
||||
__m128i d, z, a, b, xt1, vt1, ut, tmp;
|
||||
__dp_code_block1;
|
||||
d = _mm_and_si128(_mm_cmpgt_epi8(a, z), flag1_); // d = a > z? 1 : 0
|
||||
#ifdef __SSE4_1__
|
||||
z = _mm_max_epi8(z, a); // z = z > a? z : a (signed)
|
||||
tmp = _mm_cmpgt_epi8(b, z);
|
||||
d = _mm_blendv_epi8(d, flag2_, tmp); // d = b > z? 2 : d
|
||||
#else // we need to emulate SSE4.1 intrinsics _mm_max_epi8() and _mm_blendv_epi8()
|
||||
z = _mm_and_si128(z, _mm_cmpgt_epi8(z, zero_)); // z = z > 0? z : 0;
|
||||
z = _mm_max_epu8(z, a); // z = max(z, a); this works because both are non-negative
|
||||
tmp = _mm_cmpgt_epi8(b, z);
|
||||
d = _mm_or_si128(_mm_andnot_si128(tmp, d), _mm_and_si128(tmp, flag2_)); // d = b > z? 2 : d; emulating blendv
|
||||
#endif
|
||||
__dp_code_block2;
|
||||
tmp = _mm_cmpgt_epi8(a, zero_);
|
||||
_mm_store_si128(&x[t], _mm_and_si128(tmp, a));
|
||||
d = _mm_or_si128(d, _mm_and_si128(tmp, flag8_)); // d = a > 0? 0x08 : 0
|
||||
tmp = _mm_cmpgt_epi8(b, zero_);
|
||||
_mm_store_si128(&y[t], _mm_and_si128(tmp, b));
|
||||
d = _mm_or_si128(d, _mm_and_si128(tmp, flag16_)); // d = b > 0? 0x10 : 0
|
||||
_mm_store_si128(&pr[t], d);
|
||||
}
|
||||
} else { // gap right-alignment
|
||||
__m128i *pr = p + (size_t)r * n_col_ - st_;
|
||||
off[r] = st, off_end[r] = en;
|
||||
for (t = st_; t <= en_; ++t) {
|
||||
__m128i d, z, a, b, xt1, vt1, ut, tmp;
|
||||
__dp_code_block1;
|
||||
d = _mm_andnot_si128(_mm_cmpgt_epi8(z, a), flag1_); // d = z > a? 0 : 1
|
||||
#ifdef __SSE4_1__
|
||||
z = _mm_max_epi8(z, a); // z = z > a? z : a (signed)
|
||||
tmp = _mm_cmpgt_epi8(z, b);
|
||||
d = _mm_blendv_epi8(flag2_, d, tmp); // d = z > b? d : 2
|
||||
#else // we need to emulate SSE4.1 intrinsics _mm_max_epi8() and _mm_blendv_epi8()
|
||||
z = _mm_and_si128(z, _mm_cmpgt_epi8(z, zero_)); // z = z > 0? z : 0;
|
||||
z = _mm_max_epu8(z, a); // z = max(z, a); this works because both are non-negative
|
||||
tmp = _mm_cmpgt_epi8(z, b);
|
||||
d = _mm_or_si128(_mm_andnot_si128(tmp, flag2_), _mm_and_si128(tmp, d)); // d = z > b? d : 2; emulating blendv
|
||||
#endif
|
||||
__dp_code_block2;
|
||||
tmp = _mm_cmpgt_epi8(zero_, a);
|
||||
_mm_store_si128(&x[t], _mm_andnot_si128(tmp, a));
|
||||
d = _mm_or_si128(d, _mm_andnot_si128(tmp, flag8_)); // d = 0 > a? 0 : 0x08
|
||||
tmp = _mm_cmpgt_epi8(zero_, b);
|
||||
_mm_store_si128(&y[t], _mm_andnot_si128(tmp, b));
|
||||
d = _mm_or_si128(d, _mm_andnot_si128(tmp, flag16_)); // d = 0 > b? 0 : 0x10
|
||||
_mm_store_si128(&pr[t], d);
|
||||
}
|
||||
}
|
||||
if (!approx_max) { // find the exact max with a 32-bit score array
|
||||
int32_t max_H, max_t;
|
||||
// compute H[], max_H and max_t
|
||||
if (r > 0) {
|
||||
int32_t HH[4], tt[4], en1 = st0 + (en0 - st0) / 4 * 4, i;
|
||||
__m128i max_H_, max_t_, qe_;
|
||||
max_H = H[en0] = en0 > 0? H[en0-1] + u8[en0] - qe : H[en0] + v8[en0] - qe; // special casing the last element
|
||||
max_t = en0;
|
||||
max_H_ = _mm_set1_epi32(max_H);
|
||||
max_t_ = _mm_set1_epi32(max_t);
|
||||
qe_ = _mm_set1_epi32(q + e);
|
||||
for (t = st0; t < en1; t += 4) { // this implements: H[t]+=v8[t]-qe; if(H[t]>max_H) max_H=H[t],max_t=t;
|
||||
__m128i H1, tmp, t_;
|
||||
H1 = _mm_loadu_si128((__m128i*)&H[t]);
|
||||
t_ = _mm_setr_epi32(v8[t], v8[t+1], v8[t+2], v8[t+3]);
|
||||
H1 = _mm_add_epi32(H1, t_);
|
||||
H1 = _mm_sub_epi32(H1, qe_);
|
||||
_mm_storeu_si128((__m128i*)&H[t], H1);
|
||||
t_ = _mm_set1_epi32(t);
|
||||
tmp = _mm_cmpgt_epi32(H1, max_H_);
|
||||
#ifdef __SSE4_1__
|
||||
max_H_ = _mm_blendv_epi8(max_H_, H1, tmp);
|
||||
max_t_ = _mm_blendv_epi8(max_t_, t_, tmp);
|
||||
#else
|
||||
max_H_ = _mm_or_si128(_mm_and_si128(tmp, H1), _mm_andnot_si128(tmp, max_H_));
|
||||
max_t_ = _mm_or_si128(_mm_and_si128(tmp, t_), _mm_andnot_si128(tmp, max_t_));
|
||||
#endif
|
||||
}
|
||||
_mm_storeu_si128((__m128i*)HH, max_H_);
|
||||
_mm_storeu_si128((__m128i*)tt, max_t_);
|
||||
for (i = 0; i < 4; ++i)
|
||||
if (max_H < HH[i]) max_H = HH[i], max_t = tt[i] + i;
|
||||
for (; t < en0; ++t) { // for the rest of values that haven't been computed with SSE
|
||||
H[t] += (int32_t)v8[t] - qe;
|
||||
if (H[t] > max_H)
|
||||
max_H = H[t], max_t = t;
|
||||
}
|
||||
} else H[0] = v8[0] - qe - qe, max_H = H[0], max_t = 0; // special casing r==0
|
||||
// update ez
|
||||
if (en0 == tlen - 1 && H[en0] > ez->mte)
|
||||
ez->mte = H[en0], ez->mte_q = r - en;
|
||||
if (r - st0 == qlen - 1 && H[st0] > ez->mqe)
|
||||
ez->mqe = H[st0], ez->mqe_t = st0;
|
||||
if (ksw_apply_zdrop(ez, 1, max_H, r, max_t, zdrop, e)) break;
|
||||
if (r == qlen + tlen - 2 && en0 == tlen - 1)
|
||||
ez->score = H[tlen - 1];
|
||||
} else { // find approximate max; Z-drop might be inaccurate, too.
|
||||
if (r > 0) {
|
||||
if (last_H0_t >= st0 && last_H0_t <= en0 && last_H0_t + 1 >= st0 && last_H0_t + 1 <= en0) {
|
||||
int32_t d0 = v8[last_H0_t] - qe;
|
||||
int32_t d1 = u8[last_H0_t + 1] - qe;
|
||||
if (d0 > d1) H0 += d0;
|
||||
else H0 += d1, ++last_H0_t;
|
||||
} else if (last_H0_t >= st0 && last_H0_t <= en0) {
|
||||
H0 += v8[last_H0_t] - qe;
|
||||
} else {
|
||||
++last_H0_t, H0 += u8[last_H0_t] - qe;
|
||||
}
|
||||
if ((flag & KSW_EZ_APPROX_DROP) && ksw_apply_zdrop(ez, 1, H0, r, last_H0_t, zdrop, e)) break;
|
||||
} else H0 = v8[0] - qe - qe, last_H0_t = 0;
|
||||
if (r == qlen + tlen - 2 && en0 == tlen - 1)
|
||||
ez->score = H0;
|
||||
}
|
||||
last_st = st, last_en = en;
|
||||
//for (t = st0; t <= en0; ++t) printf("(%d,%d)\t(%d,%d,%d,%d)\t%d\n", r, t, ((int8_t*)u)[t], ((int8_t*)v)[t], ((int8_t*)x)[t], ((int8_t*)y)[t], H[t]); // for debugging
|
||||
}
|
||||
kfree(km, mem);
|
||||
if (!approx_max) kfree(km, H);
|
||||
if (with_cigar) { // backtrack
|
||||
int rev_cigar = !!(flag & KSW_EZ_REV_CIGAR);
|
||||
if (!ez->zdropped && !(flag&KSW_EZ_EXTZ_ONLY)) {
|
||||
ksw_backtrack(km, 1, rev_cigar, 0, (uint8_t*)p, off, off_end, n_col_*16, tlen-1, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||
} else if (!ez->zdropped && (flag&KSW_EZ_EXTZ_ONLY) && ez->mqe + end_bonus > (int)ez->max) {
|
||||
ez->reach_end = 1;
|
||||
ksw_backtrack(km, 1, rev_cigar, 0, (uint8_t*)p, off, off_end, n_col_*16, ez->mqe_t, qlen-1, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||
} else if (ez->max_t >= 0 && ez->max_q >= 0) {
|
||||
ksw_backtrack(km, 1, rev_cigar, 0, (uint8_t*)p, off, off_end, n_col_*16, ez->max_t, ez->max_q, &ez->m_cigar, &ez->n_cigar, &ez->cigar);
|
||||
}
|
||||
kfree(km, mem2); kfree(km, off);
|
||||
}
|
||||
}
|
||||
#endif // __SSE2__
|
||||
159
kthread.cpp
Normal file
159
kthread.cpp
Normal file
@@ -0,0 +1,159 @@
|
||||
#include <pthread.h>
|
||||
#include <stdlib.h>
|
||||
#include <limits.h>
|
||||
#include <stdint.h>
|
||||
#include "kthread.h"
|
||||
|
||||
#if (defined(WIN32) || defined(_WIN32)) && defined(_MSC_VER)
|
||||
#define __sync_fetch_and_add(ptr, addend) _InterlockedExchangeAdd((void*)ptr, addend)
|
||||
#endif
|
||||
|
||||
/************
|
||||
* kt_for() *
|
||||
************/
|
||||
|
||||
struct kt_for_t;
|
||||
|
||||
typedef struct {
|
||||
struct kt_for_t *t;
|
||||
long i;
|
||||
} ktf_worker_t;
|
||||
|
||||
typedef struct kt_for_t {
|
||||
int n_threads;
|
||||
long n;
|
||||
ktf_worker_t *w;
|
||||
void (*func)(void*,long,int);
|
||||
void *data;
|
||||
} kt_for_t;
|
||||
|
||||
static inline long steal_work(kt_for_t *t)
|
||||
{
|
||||
int i, min_i = -1;
|
||||
long k, min = LONG_MAX;
|
||||
for (i = 0; i < t->n_threads; ++i)
|
||||
if (min > t->w[i].i) min = t->w[i].i, min_i = i;
|
||||
k = __sync_fetch_and_add(&t->w[min_i].i, t->n_threads);
|
||||
return k >= t->n? -1 : k;
|
||||
}
|
||||
|
||||
static void *ktf_worker(void *data)
|
||||
{
|
||||
ktf_worker_t *w = (ktf_worker_t*)data;
|
||||
long i;
|
||||
for (;;) {
|
||||
i = __sync_fetch_and_add(&w->i, w->t->n_threads);
|
||||
if (i >= w->t->n) break;
|
||||
w->t->func(w->t->data, i, w - w->t->w);
|
||||
}
|
||||
while ((i = steal_work(w->t)) >= 0)
|
||||
w->t->func(w->t->data, i, w - w->t->w);
|
||||
pthread_exit(0);
|
||||
}
|
||||
|
||||
void kt_for(int n_threads, void (*func)(void*,long,int), void *data, long n)
|
||||
{
|
||||
if (n_threads > 1) {
|
||||
int i;
|
||||
kt_for_t t;
|
||||
pthread_t *tid;
|
||||
t.func = func, t.data = data, t.n_threads = n_threads, t.n = n;
|
||||
t.w = (ktf_worker_t*)calloc(n_threads, sizeof(ktf_worker_t));
|
||||
tid = (pthread_t*)calloc(n_threads, sizeof(pthread_t));
|
||||
for (i = 0; i < n_threads; ++i)
|
||||
t.w[i].t = &t, t.w[i].i = i;
|
||||
for (i = 0; i < n_threads; ++i) pthread_create(&tid[i], 0, ktf_worker, &t.w[i]);
|
||||
for (i = 0; i < n_threads; ++i) pthread_join(tid[i], 0);
|
||||
free(tid); free(t.w);
|
||||
} else {
|
||||
long j;
|
||||
for (j = 0; j < n; ++j) func(data, j, 0);
|
||||
}
|
||||
}
|
||||
|
||||
/*****************
|
||||
* kt_pipeline() *
|
||||
*****************/
|
||||
|
||||
struct ktp_t;
|
||||
|
||||
typedef struct {
|
||||
struct ktp_t *pl;
|
||||
int64_t index;
|
||||
int step;
|
||||
void *data;
|
||||
} ktp_worker_t;
|
||||
|
||||
typedef struct ktp_t {
|
||||
void *shared;
|
||||
void *(*func)(void*, int, void*);
|
||||
int64_t index;
|
||||
int n_workers, n_steps;
|
||||
ktp_worker_t *workers;
|
||||
pthread_mutex_t mutex;
|
||||
pthread_cond_t cv;
|
||||
} ktp_t;
|
||||
|
||||
static void *ktp_worker(void *data)
|
||||
{
|
||||
ktp_worker_t *w = (ktp_worker_t*)data;
|
||||
ktp_t *p = w->pl;
|
||||
while (w->step < p->n_steps) {
|
||||
// test whether we can kick off the job with this worker
|
||||
pthread_mutex_lock(&p->mutex);
|
||||
for (;;) {
|
||||
int i;
|
||||
// test whether another worker is doing the same step
|
||||
for (i = 0; i < p->n_workers; ++i) {
|
||||
if (w == &p->workers[i]) continue; // ignore itself
|
||||
if (p->workers[i].step <= w->step && p->workers[i].index < w->index)
|
||||
break;
|
||||
}
|
||||
if (i == p->n_workers) break; // no workers with smaller indices are doing w->step or the previous steps
|
||||
pthread_cond_wait(&p->cv, &p->mutex);
|
||||
}
|
||||
pthread_mutex_unlock(&p->mutex);
|
||||
|
||||
// working on w->step
|
||||
w->data = p->func(p->shared, w->step, w->step? w->data : 0); // for the first step, input is NULL
|
||||
|
||||
// update step and let other workers know
|
||||
pthread_mutex_lock(&p->mutex);
|
||||
w->step = w->step == p->n_steps - 1 || w->data? (w->step + 1) % p->n_steps : p->n_steps;
|
||||
if (w->step == 0) w->index = p->index++;
|
||||
pthread_cond_broadcast(&p->cv);
|
||||
pthread_mutex_unlock(&p->mutex);
|
||||
}
|
||||
pthread_exit(0);
|
||||
}
|
||||
|
||||
void kt_pipeline(int n_threads, void *(*func)(void*, int, void*), void *shared_data, int n_steps)
|
||||
{
|
||||
ktp_t aux;
|
||||
pthread_t *tid;
|
||||
int i;
|
||||
|
||||
if (n_threads < 1) n_threads = 1;
|
||||
aux.n_workers = n_threads;
|
||||
aux.n_steps = n_steps;
|
||||
aux.func = func;
|
||||
aux.shared = shared_data;
|
||||
aux.index = 0;
|
||||
pthread_mutex_init(&aux.mutex, 0);
|
||||
pthread_cond_init(&aux.cv, 0);
|
||||
|
||||
aux.workers = (ktp_worker_t*)calloc(n_threads, sizeof(ktp_worker_t));
|
||||
for (i = 0; i < n_threads; ++i) {
|
||||
ktp_worker_t *w = &aux.workers[i];
|
||||
w->step = 0; w->pl = &aux; w->data = 0;
|
||||
w->index = aux.index++;
|
||||
}
|
||||
|
||||
tid = (pthread_t*)calloc(n_threads, sizeof(pthread_t));
|
||||
for (i = 0; i < n_threads; ++i) pthread_create(&tid[i], 0, ktp_worker, &aux.workers[i]);
|
||||
for (i = 0; i < n_threads; ++i) pthread_join(tid[i], 0);
|
||||
free(tid); free(aux.workers);
|
||||
|
||||
pthread_mutex_destroy(&aux.mutex);
|
||||
pthread_cond_destroy(&aux.cv);
|
||||
}
|
||||
15
kthread.h
Normal file
15
kthread.h
Normal file
@@ -0,0 +1,15 @@
|
||||
#ifndef KTHREAD_H
|
||||
#define KTHREAD_H
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
void kt_for(int n_threads, void (*func)(void*,long,int), void *data, long n);
|
||||
void kt_pipeline(int n_threads, void *(*func)(void*, int, void*), void *shared_data, int n_steps);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
16
main.cpp
16
main.cpp
@@ -4,16 +4,20 @@
|
||||
#include "Process_Read.h"
|
||||
#include "Assembly.h"
|
||||
#include "Levenshtein_distance.h"
|
||||
#include "htab.h"
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
int i, ret;
|
||||
yak_reset_realtime();
|
||||
init_opt(&asm_opt);
|
||||
|
||||
if (!CommandLine_process(argc, argv, &asm_opt)) return 1;
|
||||
|
||||
Correct_Reads(asm_opt.number_of_round);
|
||||
|
||||
ret = ha_assemble();
|
||||
destory_opt(&asm_opt);
|
||||
|
||||
return 0;
|
||||
fprintf(stderr, "[M::%s] Version: %s\n", __func__, HA_VERSION);
|
||||
fprintf(stderr, "[M::%s] CMD:", __func__);
|
||||
for (i = 0; i < argc; ++i)
|
||||
fprintf(stderr, " %s", argv[i]);
|
||||
fprintf(stderr, "\n[M::%s] Real time: %.3f sec; CPU: %.3f sec; Peak RSS: %.3f GB\n", __func__, yak_realtime(), yak_cputime(), yak_peakrss_in_gb());
|
||||
return ret;
|
||||
}
|
||||
|
||||
110
sketch.cpp
Normal file
110
sketch.cpp
Normal file
@@ -0,0 +1,110 @@
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <assert.h>
|
||||
#include <string.h>
|
||||
#include "kvec.h"
|
||||
#include "htab.h"
|
||||
|
||||
typedef struct { // a simplified version of kdq
|
||||
int front, count;
|
||||
int a[64];
|
||||
} tiny_queue_t;
|
||||
|
||||
static inline void tq_push(tiny_queue_t *q, int x)
|
||||
{
|
||||
q->a[((q->count++) + q->front) & 0x3f] = x;
|
||||
}
|
||||
|
||||
static inline int tq_shift(tiny_queue_t *q)
|
||||
{
|
||||
int x;
|
||||
if (q->count == 0) return -1;
|
||||
x = q->a[q->front++];
|
||||
q->front &= 0x3f;
|
||||
--q->count;
|
||||
return x;
|
||||
}
|
||||
|
||||
/**
|
||||
* Find symmetric (w,k)-minimizers on a DNA sequence
|
||||
*
|
||||
* @param str DNA sequence
|
||||
* @param len length of $str
|
||||
* @param w find a minimizer for every $w consecutive k-mers
|
||||
* @param k k-mer size
|
||||
* @param rid reference ID; will be copied to the output $p array
|
||||
* @param is_hpc homopolymer-compressed or not
|
||||
* @param p minimizers
|
||||
*/
|
||||
void ha_sketch(const char *str, int len, int w, int k, uint32_t rid, int is_hpc, ha_mz1_v *p, const void *hf)
|
||||
{
|
||||
static const ha_mz1_t dummy = { UINT64_MAX, 0, 0, 0 };
|
||||
uint64_t shift1 = k - 1, mask = (1ULL<<k) - 1, kmer[4] = {0,0,0,0};
|
||||
int i, j, l, buf_pos, min_pos, kmer_span = 0;
|
||||
ha_mz1_t buf[256], min = dummy;
|
||||
tiny_queue_t tq;
|
||||
|
||||
assert(len > 0 && len < 1<<27 && rid < 1<<28 && (w > 0 && w < 256) && (k > 0 && k <= 63));
|
||||
memset(buf, 0xff, w * 16);
|
||||
memset(&tq, 0, sizeof(tiny_queue_t));
|
||||
kv_resize(ha_mz1_t, *p, p->n + len/w);
|
||||
|
||||
for (i = l = buf_pos = min_pos = 0; i < len; ++i) {
|
||||
int c = seq_nt4_table[(uint8_t)str[i]];
|
||||
ha_mz1_t info = dummy;
|
||||
if (c < 4) { // not an ambiguous base
|
||||
int z;
|
||||
if (is_hpc) {
|
||||
int skip_len = 1;
|
||||
if (i + 1 < len && seq_nt4_table[(uint8_t)str[i + 1]] == c) {
|
||||
for (skip_len = 2; i + skip_len < len; ++skip_len)
|
||||
if (seq_nt4_table[(uint8_t)str[i + skip_len]] != c)
|
||||
break;
|
||||
i += skip_len - 1; // put $i at the end of the current homopolymer run
|
||||
}
|
||||
tq_push(&tq, skip_len);
|
||||
kmer_span += skip_len;
|
||||
if (tq.count > k) kmer_span -= tq_shift(&tq);
|
||||
} else kmer_span = l + 1 < k? l + 1 : k;
|
||||
kmer[0] = (kmer[0] << 1 | (c&1)) & mask; // forward k-mer
|
||||
kmer[1] = (kmer[1] << 1 | (c>>1)) & mask;
|
||||
kmer[2] = kmer[2] >> 1 | (uint64_t)(1 - (c&1)) << shift1; // reverse k-mer
|
||||
kmer[3] = kmer[3] >> 1 | (uint64_t)(1 - (c>>1)) << shift1;
|
||||
if (kmer[1] == kmer[3]) continue; // skip "symmetric k-mers" as we don't know it strand
|
||||
z = kmer[1] < kmer[3]? 0 : 1; // strand
|
||||
++l;
|
||||
if (l >= k && kmer_span < 256) {
|
||||
uint64_t y;
|
||||
y = yak_hash64_64(kmer[z<<1|0]) + yak_hash64_64(kmer[z<<1|1]);
|
||||
if (hf == 0 || ha_ft_isflt(hf, y) == 0)
|
||||
info.x = y, info.rid = rid, info.pos = i, info.rev = z, info.span = kmer_span;
|
||||
}
|
||||
} else l = 0, tq.count = tq.front = 0, kmer_span = 0;
|
||||
buf[buf_pos] = info; // need to do this here as appropriate buf_pos and buf[buf_pos] are needed below
|
||||
if (l == w + k - 1 && min.x != UINT64_MAX) { // special case for the first window - because identical k-mers are not stored yet
|
||||
for (j = buf_pos + 1; j < w; ++j)
|
||||
if (min.x == buf[j].x && buf[j].pos != min.pos) kv_push(ha_mz1_t, *p, buf[j]);
|
||||
for (j = 0; j < buf_pos; ++j)
|
||||
if (min.x == buf[j].x && buf[j].pos != min.pos) kv_push(ha_mz1_t, *p, buf[j]);
|
||||
}
|
||||
if (info.x <= min.x) { // a new minimum; then write the old min
|
||||
if (l >= w + k && min.x != UINT64_MAX) kv_push(ha_mz1_t, *p, min);
|
||||
min = info, min_pos = buf_pos;
|
||||
} else if (buf_pos == min_pos) { // old min has moved outside the window
|
||||
if (l >= w + k - 1 && min.x != UINT64_MAX) kv_push(ha_mz1_t, *p, min);
|
||||
for (j = buf_pos + 1, min.x = UINT64_MAX; j < w; ++j) // the two loops are necessary when there are identical k-mers
|
||||
if (min.x >= buf[j].x) min = buf[j], min_pos = j; // >= is important s.t. min is always the closest k-mer
|
||||
for (j = 0; j <= buf_pos; ++j)
|
||||
if (min.x >= buf[j].x) min = buf[j], min_pos = j;
|
||||
if (l >= w + k - 1 && min.x != UINT64_MAX) { // write identical k-mers
|
||||
for (j = buf_pos + 1; j < w; ++j) // these two loops make sure the output is sorted
|
||||
if (min.x == buf[j].x && min.pos != buf[j].pos) kv_push(ha_mz1_t, *p, buf[j]);
|
||||
for (j = 0; j <= buf_pos; ++j)
|
||||
if (min.x == buf[j].x && min.pos != buf[j].pos) kv_push(ha_mz1_t, *p, buf[j]);
|
||||
}
|
||||
}
|
||||
if (++buf_pos == w) buf_pos = 0;
|
||||
}
|
||||
if (min.x != UINT64_MAX)
|
||||
kv_push(ha_mz1_t, *p, min);
|
||||
}
|
||||
53
sys.cpp
Normal file
53
sys.cpp
Normal file
@@ -0,0 +1,53 @@
|
||||
#include <sys/resource.h>
|
||||
#include <sys/time.h>
|
||||
#include "htab.h"
|
||||
|
||||
int yak_verbose = 3;
|
||||
|
||||
static double yak_realtime0;
|
||||
|
||||
double yak_cputime(void)
|
||||
{
|
||||
struct rusage r;
|
||||
getrusage(RUSAGE_SELF, &r);
|
||||
return r.ru_utime.tv_sec + r.ru_stime.tv_sec + 1e-6 * (r.ru_utime.tv_usec + r.ru_stime.tv_usec);
|
||||
}
|
||||
|
||||
static inline double yak_realtime_core(void)
|
||||
{
|
||||
struct timeval tp;
|
||||
struct timezone tzp;
|
||||
gettimeofday(&tp, &tzp);
|
||||
return tp.tv_sec + tp.tv_usec * 1e-6;
|
||||
}
|
||||
|
||||
void yak_reset_realtime(void)
|
||||
{
|
||||
yak_realtime0 = yak_realtime_core();
|
||||
}
|
||||
|
||||
double yak_realtime(void)
|
||||
{
|
||||
return yak_realtime_core() - yak_realtime0;
|
||||
}
|
||||
|
||||
long yak_peakrss(void)
|
||||
{
|
||||
struct rusage r;
|
||||
getrusage(RUSAGE_SELF, &r);
|
||||
#ifdef __linux__
|
||||
return r.ru_maxrss * 1024;
|
||||
#else
|
||||
return r.ru_maxrss;
|
||||
#endif
|
||||
}
|
||||
|
||||
double yak_peakrss_in_gb(void)
|
||||
{
|
||||
return yak_peakrss() / 1073741824.0;
|
||||
}
|
||||
|
||||
double yak_cpu_usage(void)
|
||||
{
|
||||
return (yak_cputime() + 1e-9) / (yak_realtime() + 1e-9);
|
||||
}
|
||||
Reference in New Issue
Block a user