mirror of
https://github.com/chhylp123/hifiasm.git
synced 2026-09-28 09:08:12 +08:00
edit distance based error correction
This commit is contained in:
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Vendored
+6
-1
@@ -1,3 +1,8 @@
|
||||
{
|
||||
"C_Cpp.errorSquiggles": "Enabled"
|
||||
"C_Cpp.errorSquiggles": "Enabled",
|
||||
"files.associations": {
|
||||
"limits": "cpp",
|
||||
"random": "cpp",
|
||||
"functional": "cpp"
|
||||
}
|
||||
}
|
||||
+66
-3
@@ -7,6 +7,7 @@
|
||||
#include "kmer.h"
|
||||
#include "Hash_Table.h"
|
||||
#include "POA.h"
|
||||
#include "Correct.h"
|
||||
|
||||
Total_Count_Table TCB;
|
||||
Total_Pos_Table PCB;
|
||||
@@ -693,6 +694,9 @@ void* Overlap_calculate_heap_merge(void* arg)
|
||||
k_mer_pos* list;
|
||||
uint64_t list_length;
|
||||
uint64_t sub_ID;
|
||||
long long total_shared_seed = 0;
|
||||
long long candidate_overlap_reads = 0;
|
||||
|
||||
|
||||
Candidates_list l;
|
||||
//Candidates_list debug_l;
|
||||
@@ -712,16 +716,27 @@ void* Overlap_calculate_heap_merge(void* arg)
|
||||
|
||||
Init_Heap(&heap);
|
||||
|
||||
Correct_dumy correct;
|
||||
init_Correct_dumy(&correct);
|
||||
|
||||
for (i = thr_ID; i < R_INF.total_reads; i = i + thread_num)
|
||||
{
|
||||
/**
|
||||
if (thr_ID == 0 && i % 1000 == 0)
|
||||
if (i < 4)
|
||||
{
|
||||
fprintf(stderr, "i: %llu\n", i);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (i > 4)
|
||||
{
|
||||
break;
|
||||
}
|
||||
fprintf(stderr, "i: %u\n", i);
|
||||
fflush(stderr);
|
||||
**/
|
||||
|
||||
|
||||
|
||||
clear_Heap(&heap);
|
||||
clear_Candidates_list(&l);
|
||||
///clear_Candidates_list(&debug_l);
|
||||
@@ -833,8 +848,48 @@ void* Overlap_calculate_heap_merge(void* arg)
|
||||
}
|
||||
**/
|
||||
|
||||
///以x_pos_e,即结束位置为主元排序
|
||||
calculate_overlap_region(&l, &overlap_list, i, g_read.length, &R_INF);
|
||||
|
||||
|
||||
correct_overlap(&overlap_list, &R_INF, &g_read, &correct);
|
||||
|
||||
|
||||
|
||||
/**
|
||||
POA_i = 0;
|
||||
fprintf(stderr, "\n\n**************\ni: %u\n", i);
|
||||
|
||||
for (POA_i = 0; POA_i < overlap_list.length; POA_i++)
|
||||
{
|
||||
fprintf(stderr, "x_id: %u, y_id: %u\n", overlap_list.list[POA_i].x_id, overlap_list.list[POA_i].y_id);
|
||||
fprintf(stderr, "x_strand: %u, y_strand: %u\n", overlap_list.list[POA_i].x_pos_strand, overlap_list.list[POA_i].y_pos_strand);
|
||||
fprintf(stderr, "x_pos_s: %u\n", overlap_list.list[POA_i].x_pos_s);
|
||||
|
||||
}
|
||||
**/
|
||||
|
||||
/**
|
||||
POA_i = 0;
|
||||
|
||||
for (POA_i = 1; POA_i < overlap_list.length; POA_i++)
|
||||
{
|
||||
if(overlap_list.list[POA_i].x_pos_s < overlap_list.list[POA_i - 1].x_pos_s)
|
||||
{
|
||||
fprintf(stderr, "1 sbsbsbsbs\n");
|
||||
}
|
||||
else if(overlap_list.list[POA_i].x_pos_s == overlap_list.list[POA_i - 1].x_pos_s &&
|
||||
overlap_list.list[POA_i].x_pos_e > overlap_list.list[POA_i - 1].x_pos_e)
|
||||
{
|
||||
fprintf(stderr, "2 sbsbsbsbs\n");
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
**/
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
///merge_k_mer_pos_list_alloc_heap_sort_advance(&array_list, &l, &heap);
|
||||
@@ -843,10 +898,12 @@ void* Overlap_calculate_heap_merge(void* arg)
|
||||
|
||||
///debug_merge_result(&l, &debug_l);
|
||||
|
||||
/**
|
||||
clear_Graph(&POA_Graph);
|
||||
|
||||
|
||||
Perform_POA(&POA_Graph, &overlap_list, &R_INF, &g_read);
|
||||
**/
|
||||
|
||||
///fprintf(stderr, "i: %u\n", i);
|
||||
|
||||
@@ -875,7 +932,11 @@ void* Overlap_calculate_heap_merge(void* arg)
|
||||
|
||||
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
fprintf(stderr, "candidate_overlap_reads: %llu\n", candidate_overlap_reads);
|
||||
fprintf(stderr, "total_shared_seed: %llu\n", total_shared_seed);
|
||||
**/
|
||||
destory_Candidates_list(&l);
|
||||
destory_overlap_region_alloc(&overlap_list);
|
||||
//destory_Candidates_list(&debug_l);
|
||||
@@ -885,6 +946,8 @@ void* Overlap_calculate_heap_merge(void* arg)
|
||||
|
||||
destory_Graph(&POA_Graph);
|
||||
destory_UC_Read(&g_read);
|
||||
|
||||
destory_Correct_dumy(&correct);
|
||||
}
|
||||
|
||||
|
||||
|
||||
+685
@@ -0,0 +1,685 @@
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <stdint.h>
|
||||
#include "Correct.h"
|
||||
#include "Levenshtein_distance.h"
|
||||
#include "edlib.h"
|
||||
|
||||
long long T_total_match=0;
|
||||
long long T_total_unmatch=0;
|
||||
long long T_total_mis=0;
|
||||
|
||||
|
||||
#define MAX(x, y) ((x >= y)?x:y)
|
||||
#define MIN(x, y) ((x <= y)?x:y)
|
||||
#define OVERLAP(x_start, x_end, y_start, y_end) (MIN(x_end, y_end) - MAX(x_start, y_start) + 1)
|
||||
///#define OVERLAP(x_start, x_end, y_start, y_end) MIN(x_end, y_end) - MAX(x_start, y_start) + 1
|
||||
|
||||
|
||||
|
||||
///y_length > x_length
|
||||
unsigned int edit_distance_normal_test_banded(char* y, int y_length, char* x, int x_length, int error_cut, int matrix[1000][1000] )
|
||||
{ memset(matrix, 0, sizeof(matrix));
|
||||
|
||||
int i, j;
|
||||
for (i = 0; i <= x_length; i++)
|
||||
{
|
||||
matrix[i][0] = i;
|
||||
}
|
||||
|
||||
int digonal, up, left;
|
||||
unsigned int min;
|
||||
|
||||
///一列列算的
|
||||
for (i = 0; i < x_length; i++)
|
||||
{
|
||||
for (j = 0; j < y_length; j++)
|
||||
{
|
||||
///matrix[i + 1][j + 1]
|
||||
digonal = matrix[i][j] + (x[i] != y[j]);
|
||||
up = matrix[i + 1][j] + 1;
|
||||
left = matrix[i][j + 1] + 1;
|
||||
min = digonal;
|
||||
if (up < min)
|
||||
{
|
||||
min = up;
|
||||
}
|
||||
|
||||
if (left< min)
|
||||
{
|
||||
min = left;
|
||||
}
|
||||
|
||||
matrix[i + 1][j + 1] = min;
|
||||
}
|
||||
}
|
||||
|
||||
min = (unsigned int)-1;
|
||||
for (j = x_length; j <= y_length; j++)
|
||||
{
|
||||
if (matrix[i][j] < min)
|
||||
{
|
||||
min = matrix[i][j];
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
return min <= error_cut?min:(unsigned int)(-1);
|
||||
}
|
||||
|
||||
void verify_get_interval(long long window_start, long long window_end, overlap_region_alloc* overlap_list,Correct_dumy* dumy)
|
||||
{
|
||||
long long i;
|
||||
long long match_length = 0;
|
||||
long long match_lengthNT = 0;
|
||||
long long Len;
|
||||
|
||||
for (i = 0; i < overlap_list->length; i++)
|
||||
{
|
||||
if((Len = OVERLAP(window_start, window_end, overlap_list->list[i].x_pos_s, overlap_list->list[i].x_pos_e)) > 0)
|
||||
{
|
||||
if (Len == WINDOW)
|
||||
{
|
||||
match_length++;
|
||||
long long j;
|
||||
for (j = 0; j < dumy->length; j++)
|
||||
{
|
||||
if (i==dumy->overlapID[j])
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (j >= dumy->length)
|
||||
{
|
||||
fprintf(stderr, "+ERROR interval\n");
|
||||
|
||||
fprintf(stderr, "i: %u, window_start: %u, window_end: %u, x_pos_s: %u, x_pos_e: %u\n",
|
||||
i, window_start, window_end, overlap_list->list[i].x_pos_s, overlap_list->list[i].x_pos_e);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
match_lengthNT++;
|
||||
long long j;
|
||||
for (j = 0; j < dumy->lengthNT; j++)
|
||||
{
|
||||
if (i==dumy->overlapID[dumy->size - j - 1])
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (j >= dumy->lengthNT)
|
||||
{
|
||||
fprintf(stderr, "-ERROR interval\n");
|
||||
|
||||
fprintf(stderr, "i: %u, window_start: %u, window_end: %u, x_pos_s: %u, x_pos_e: %u, dumy->lengthNT: %u\n",
|
||||
i, window_start, window_end, overlap_list->list[i].x_pos_s, overlap_list->list[i].x_pos_e, dumy->lengthNT);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
if (match_length != dumy->length || match_lengthNT != dumy->lengthNT)
|
||||
{
|
||||
fprintf(stderr, "****************ERROR interval length*******************\n");
|
||||
fprintf(stderr, "match_length: %u\n", match_length);
|
||||
fprintf(stderr, "dumy->length: %u\n", dumy->length);
|
||||
fprintf(stderr, "window_start: %u, window_end: %u\n", window_start, window_end);
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
inline int get_interval_back(long long window_start, long long window_end, overlap_region_alloc* overlap_list, Correct_dumy* dumy)
|
||||
{
|
||||
long long i;
|
||||
int flag = 0;
|
||||
|
||||
for (i = dumy->start_i; i < overlap_list->length; i++)
|
||||
{
|
||||
///只会发生在这个interval比list里所有元素都小的情况
|
||||
///这种情况下一个interval需要从0开始
|
||||
if (window_start < overlap_list->list[i].x_pos_s)
|
||||
{
|
||||
dumy->start_i = 0;
|
||||
dumy->length = 0;
|
||||
return -1;
|
||||
}
|
||||
else if(window_start >= overlap_list->list[i].x_pos_s && window_start <= overlap_list->list[i].x_pos_e)
|
||||
{
|
||||
dumy->start_i = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
///只会发生在这个window比list里所有元素都大的情况
|
||||
///这种情况下一个window也无需遍历了
|
||||
if (i >= overlap_list->length)
|
||||
{
|
||||
dumy->start_i = overlap_list->length;
|
||||
dumy->length = 0;
|
||||
return -2;
|
||||
}
|
||||
|
||||
///走到这里的时候,至少window_start的要求是满足了
|
||||
dumy->length = 0;
|
||||
|
||||
for (; i < overlap_list->length; i++)
|
||||
{
|
||||
if(overlap_list->list[i].x_pos_s <= window_start && overlap_list->list[i].x_pos_e >= window_end)
|
||||
{
|
||||
dumy->overlapID[dumy->length] = i;
|
||||
dumy->length++;
|
||||
}
|
||||
else if(overlap_list->list[i].x_pos_s > window_start)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if ( dumy->length == 0)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
inline int get_interval(long long window_start, long long window_end, overlap_region_alloc* overlap_list, Correct_dumy* dumy)
|
||||
{
|
||||
long long i;
|
||||
int flag = 0;
|
||||
long long Begin, End, Len;
|
||||
|
||||
|
||||
for (i = dumy->start_i; i < overlap_list->length; i++)
|
||||
{
|
||||
///只会发生在这个interval比list里所有元素都小的情况
|
||||
///这种情况下一个interval需要从0开始
|
||||
if (window_end < overlap_list->list[i].x_pos_s)
|
||||
{
|
||||
dumy->start_i = 0;
|
||||
dumy->length = 0;
|
||||
return 0;
|
||||
}
|
||||
else ///只要window_end >= overlap_list->list[i].x_pos_s,就有可能重叠
|
||||
{
|
||||
dumy->start_i = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
///只会发生在这个window比list里所有元素都大的情况
|
||||
///这种情况下一个window也无需遍历了
|
||||
if (i >= overlap_list->length)
|
||||
{
|
||||
dumy->start_i = overlap_list->length;
|
||||
dumy->length = 0;
|
||||
return -2;
|
||||
}
|
||||
|
||||
dumy->length = 0;
|
||||
dumy->lengthNT = 0;
|
||||
|
||||
for (; i < overlap_list->length; i++)
|
||||
{
|
||||
|
||||
if((Len = OVERLAP(window_start, window_end, overlap_list->list[i].x_pos_s, overlap_list->list[i].x_pos_e)) > 0)
|
||||
{
|
||||
if (Len == WINDOW)
|
||||
{
|
||||
dumy->overlapID[dumy->length] = i;
|
||||
dumy->length++;
|
||||
}
|
||||
else
|
||||
{
|
||||
dumy->lengthNT++;
|
||||
dumy->overlapID[dumy->size - dumy->lengthNT] = i;
|
||||
}
|
||||
}
|
||||
|
||||
if(overlap_list->list[i].x_pos_s > window_end)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if ( dumy->length + dumy->lengthNT == 0)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
}
|
||||
|
||||
///Len = OVERLAP(window_start, window_end, overlap_list->list[i].x_pos_s, overlap_list->list[i].x_pos_e))
|
||||
|
||||
void print_string(char* s, int l)
|
||||
{
|
||||
for (size_t i = 0; i < l; i++)
|
||||
{
|
||||
fprintf(stderr, "%c", s[i]);
|
||||
}
|
||||
|
||||
fprintf(stderr, "\n");
|
||||
|
||||
}
|
||||
|
||||
void test_edit_distance_by_edlib(char* x_string, char* y_string, long long x_len,
|
||||
long long o_len, int threashold, int error, long long* total_mis)
|
||||
{
|
||||
EdlibAlignResult result = edlibAlign(x_string, x_len, y_string, o_len,
|
||||
edlibNewAlignConfig(threashold, EDLIB_MODE_HW, EDLIB_TASK_PATH, NULL, 0));
|
||||
|
||||
if (result.status == EDLIB_STATUS_OK) {
|
||||
if (result.editDistance != error)
|
||||
///if (result.editDistance != error && result.endLocations[0] > x_len)
|
||||
{
|
||||
|
||||
(*total_mis)++;
|
||||
|
||||
|
||||
if ((int)error != -1 && result.editDistance==-1)
|
||||
{
|
||||
fprintf(stderr, "ERROR1\n");
|
||||
}
|
||||
|
||||
if ((int)error < result.editDistance==-1 &&
|
||||
(int)error != -1 && result.editDistance!=-1)
|
||||
{
|
||||
fprintf(stderr, "ERROR2\n");
|
||||
}
|
||||
|
||||
int up_length = o_len - result.endLocations[0] - 1;
|
||||
int left_length = result.endLocations[0] - x_len;
|
||||
|
||||
|
||||
char* cigar = edlibAlignmentToCigar(result.alignment, result.alignmentLength, EDLIB_CIGAR_STANDARD);
|
||||
int cigar_length = strlen(cigar);
|
||||
|
||||
/**
|
||||
for (int i = cigar_length - 1; i >= 0; i--)
|
||||
{
|
||||
switch (cigar[i])
|
||||
{
|
||||
case 'I':
|
||||
|
||||
break;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
**/
|
||||
|
||||
///fprintf(stderr,"%s\n", cigar);
|
||||
free(cigar);
|
||||
|
||||
/**
|
||||
fprintf(stderr, "****\ni: %u, edlib: %d, alignmentLength: %d, startLocations: %d, endLocations: %d\n",
|
||||
i, result.editDistance, result.alignmentLength, result.startLocations[0], result.endLocations[0]);
|
||||
char* cigar = edlibAlignmentToCigar(result.alignment, result.alignmentLength, EDLIB_CIGAR_STANDARD);
|
||||
fprintf(stderr,"%s\n", cigar);
|
||||
free(cigar);
|
||||
print_string(x_string, x_len);
|
||||
print_string(y_string, o_len);
|
||||
|
||||
fprintf(stderr, "BPM: %d\n", error);
|
||||
**/
|
||||
|
||||
}
|
||||
}
|
||||
edlibFreeAlignResult(result);
|
||||
}
|
||||
|
||||
void verify_window(long long window_start, long long window_end, overlap_region_alloc* overlap_list,Correct_dumy* dumy, All_reads* R_INF,
|
||||
char* r_string)
|
||||
{
|
||||
|
||||
long long i;
|
||||
long long currentID, currentIDLen;
|
||||
long long x_start, y_start, o_len;
|
||||
long long Window_Len = WINDOW + (THRESHOLD << 1);
|
||||
char* x_string = NULL;
|
||||
char* y_string = NULL;
|
||||
long long x_end, x_len;
|
||||
int end_site;
|
||||
unsigned int error;
|
||||
long long total_match=0;
|
||||
long long total_unmatch=0;
|
||||
long long total_mis=0;
|
||||
|
||||
|
||||
///这些是整个window被完全覆盖的
|
||||
for (i = 0; i < dumy->length; i++)
|
||||
{
|
||||
///整个window被覆盖的话,read本身上的区间就是[window_start, window_end]
|
||||
x_len = WINDOW;
|
||||
currentID = dumy->overlapID[i];
|
||||
x_start = window_start;
|
||||
///y上的相对位置
|
||||
y_start = (x_start - overlap_list->list[currentID].x_pos_s) + overlap_list->list[currentID].y_pos_s;
|
||||
///y上的起始
|
||||
y_start = y_start - THRESHOLD;
|
||||
if (y_start < 0)
|
||||
{
|
||||
y_start = 0;
|
||||
}
|
||||
///当前y的长度
|
||||
currentIDLen = Get_READ_LENGTH((*R_INF), overlap_list->list[currentID].y_id);
|
||||
///不能超过y的剩余长度
|
||||
o_len = MIN(Window_Len, currentIDLen - y_start);
|
||||
|
||||
recover_UC_Read_sub_region(dumy->overlap_region, y_start, o_len, overlap_list->list[currentID].y_pos_strand,
|
||||
R_INF, overlap_list->list[currentID].y_id);
|
||||
|
||||
x_string = r_string + x_start;
|
||||
y_string = dumy->overlap_region;
|
||||
|
||||
|
||||
|
||||
|
||||
///这个长度足够,可以用来一起比
|
||||
if (Window_Len == o_len)
|
||||
{
|
||||
|
||||
end_site = Reserve_Banded_BPM(y_string, o_len, x_string, x_len, THRESHOLD, &error);
|
||||
|
||||
|
||||
if (error!=(unsigned int)-1)
|
||||
{
|
||||
total_match++;
|
||||
}
|
||||
else
|
||||
{
|
||||
total_unmatch++;
|
||||
}
|
||||
|
||||
|
||||
test_edit_distance_by_edlib(x_string, y_string, x_len,
|
||||
o_len, THRESHOLD, error, &total_mis);
|
||||
|
||||
}
|
||||
else ///这个不够,只能单个比 ///不够的地方要置N
|
||||
{
|
||||
/**
|
||||
end_site = Reserve_Banded_BPM(y_string, o_len, x_string, x_len, THRESHOLD, &error);
|
||||
|
||||
|
||||
if (error!=-1)
|
||||
{
|
||||
total_match++;
|
||||
}
|
||||
else
|
||||
{
|
||||
total_unmatch++;
|
||||
}
|
||||
**/
|
||||
}
|
||||
}
|
||||
|
||||
long long reverse_i = dumy->size - 1;
|
||||
|
||||
///这些是整个window被部分覆盖的
|
||||
for (i = 0; i < dumy->lengthNT; i++)
|
||||
{
|
||||
currentID = dumy->overlapID[reverse_i--];
|
||||
x_start = MAX(window_start, overlap_list->list[currentID].x_pos_s);
|
||||
x_end = MIN(window_end, overlap_list->list[currentID].x_pos_e);
|
||||
///这个是和当前窗口重叠的长度
|
||||
x_len = x_end - x_start + 1;
|
||||
|
||||
if (x_len <= 0)
|
||||
{
|
||||
fprintf(stderr, "ERROR\n");
|
||||
}
|
||||
|
||||
///y上的相对位置
|
||||
y_start = (x_start - overlap_list->list[currentID].x_pos_s) + overlap_list->list[currentID].y_pos_s;
|
||||
///y上的起始
|
||||
y_start = y_start - THRESHOLD;
|
||||
if (y_start < 0)
|
||||
{
|
||||
y_start = 0;
|
||||
}
|
||||
|
||||
///当前y的长度
|
||||
currentIDLen = Get_READ_LENGTH((*R_INF), overlap_list->list[currentID].y_id);
|
||||
|
||||
|
||||
///不能超过y的剩余长度
|
||||
Window_Len = x_len + (THRESHOLD << 1);
|
||||
o_len = MIN(Window_Len, currentIDLen - y_start);
|
||||
|
||||
recover_UC_Read_sub_region(dumy->overlap_region, y_start, o_len, overlap_list->list[currentID].y_pos_strand,
|
||||
R_INF, overlap_list->list[currentID].y_id);
|
||||
|
||||
x_string = r_string + x_start;
|
||||
y_string = dumy->overlap_region;
|
||||
|
||||
///不够的地方要置N
|
||||
|
||||
end_site = Reserve_Banded_BPM(y_string, o_len, x_string, x_len, THRESHOLD, &error);
|
||||
|
||||
/**
|
||||
if (error!=-1)
|
||||
{
|
||||
total_match++;
|
||||
}
|
||||
else
|
||||
{
|
||||
total_unmatch++;
|
||||
}
|
||||
**/
|
||||
}
|
||||
|
||||
/**
|
||||
if (total_match!=0)
|
||||
{
|
||||
|
||||
fprintf(stderr, "total_match: %u\n", total_match);
|
||||
fprintf(stderr, "total_unmatch: %u\n", total_unmatch);
|
||||
fprintf(stderr, "total_mis: %u\n", total_mis);
|
||||
|
||||
}
|
||||
**/
|
||||
|
||||
T_total_match = T_total_match + total_match;
|
||||
T_total_unmatch = T_total_unmatch + total_unmatch;
|
||||
T_total_mis = T_total_mis + total_mis;
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
void correct_overlap(overlap_region_alloc* overlap_list, All_reads* R_INF, UC_Read* g_read, Correct_dumy* dumy)
|
||||
{
|
||||
reverse_complement(g_read->seq, g_read->length);
|
||||
|
||||
clear_Correct_dumy(dumy, overlap_list);
|
||||
|
||||
long long window_num = (g_read->length + WINDOW - 1) / WINDOW;
|
||||
|
||||
long long i;
|
||||
|
||||
long long window_start, window_end;
|
||||
|
||||
window_start = 0;
|
||||
window_end = WINDOW - 1;
|
||||
|
||||
|
||||
|
||||
/**
|
||||
UC_Read debug_read;
|
||||
init_UC_Read(&debug_read);
|
||||
|
||||
for (i = 0; i < overlap_list->length; i++)
|
||||
{
|
||||
|
||||
if (overlap_list->list[i].y_pos_strand)
|
||||
{
|
||||
recover_UC_Read_RC(&debug_read, R_INF, overlap_list->list[i].y_id);
|
||||
}
|
||||
else
|
||||
{
|
||||
recover_UC_Read(&debug_read, R_INF, overlap_list->list[i].y_id);
|
||||
}
|
||||
|
||||
|
||||
long long x_length = overlap_list->list[i].x_pos_e - overlap_list->list[i].x_pos_s + 1;
|
||||
long long y_length = overlap_list->list[i].y_pos_e - overlap_list->list[i].y_pos_s + 1;
|
||||
|
||||
EdlibAlignResult result = edlibAlign(g_read->seq+overlap_list->list[i].x_pos_s,
|
||||
x_length,
|
||||
debug_read.seq + overlap_list->list[i].y_pos_s,
|
||||
y_length,
|
||||
edlibNewAlignConfig(-1, EDLIB_MODE_NW, EDLIB_TASK_DISTANCE, NULL, 0));
|
||||
|
||||
|
||||
if (result.status == EDLIB_STATUS_OK) {
|
||||
if (result.editDistance>0 && result.editDistance < x_length*0.02)
|
||||
{
|
||||
fprintf(stderr, "inner_i: %u, x_length: %u, y_length: %u, editDistance: %u\n",
|
||||
i, x_length, y_length, result.editDistance);
|
||||
fprintf(stderr, "x_pos_s: %u, x_pos_e: %u\n",
|
||||
overlap_list->list[i].x_pos_s, overlap_list->list[i].x_pos_e);
|
||||
fprintf(stderr, "y_pos_s: %u, y_pos_e: %u, y_pos_strand: %u\n",
|
||||
overlap_list->list[i].y_pos_s, overlap_list->list[i].y_pos_e, overlap_list->list[i].y_pos_strand);
|
||||
|
||||
|
||||
EdlibAlignResult n_result = edlibAlign(g_read->seq+overlap_list->list[i].x_pos_s,
|
||||
350,
|
||||
debug_read.seq + overlap_list->list[i].y_pos_s - 15,
|
||||
380,
|
||||
edlibNewAlignConfig(-1, EDLIB_MODE_HW, EDLIB_TASK_DISTANCE, NULL, 0));
|
||||
|
||||
fprintf(stderr, "n_editDistance: %u\n",
|
||||
n_result.editDistance);
|
||||
|
||||
|
||||
|
||||
|
||||
edlibFreeAlignResult(n_result);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
edlibFreeAlignResult(result);
|
||||
|
||||
|
||||
}
|
||||
|
||||
destory_UC_Read(&debug_read);
|
||||
**/
|
||||
///return;
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
int flag;
|
||||
|
||||
for (i = 0; i < window_num; i++)
|
||||
{
|
||||
|
||||
dumy->length = 0;
|
||||
dumy->lengthNT = 0;
|
||||
flag = get_interval(window_start, window_end, overlap_list, dumy);
|
||||
|
||||
switch (flag)
|
||||
{
|
||||
case 1: ///找到匹配
|
||||
break;
|
||||
case 0: ///没找到匹配
|
||||
break;
|
||||
case -2: ///下一个window也不会存在匹配, 直接跳出
|
||||
i = window_num;
|
||||
break;
|
||||
}
|
||||
|
||||
if(dumy->length + dumy->lengthNT>overlap_list->length)
|
||||
{
|
||||
fprintf(stderr, "error length\n");
|
||||
}
|
||||
|
||||
///verify_get_interval(window_start, window_end, overlap_list, dumy);
|
||||
|
||||
verify_window(window_start, window_end, overlap_list, dumy, R_INF, g_read->seq);
|
||||
|
||||
window_start = window_start + WINDOW;
|
||||
window_end = window_end + WINDOW;
|
||||
if (window_end >= g_read->length)
|
||||
{
|
||||
window_end = g_read->length - 1;
|
||||
}
|
||||
|
||||
///break;
|
||||
|
||||
}
|
||||
|
||||
|
||||
fprintf(stderr, "total_match: %u, total_unmatch: %u, total_mis: %u\n",
|
||||
T_total_match, T_total_unmatch, T_total_mis);
|
||||
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
|
||||
void init_Correct_dumy(Correct_dumy* list)
|
||||
{
|
||||
list->size = 0;
|
||||
list->length = 0;
|
||||
list->lengthNT = 0;
|
||||
list->start_i = 0;
|
||||
list->overlapID = NULL;
|
||||
}
|
||||
|
||||
void destory_Correct_dumy(Correct_dumy* list)
|
||||
{
|
||||
free(list->overlapID);
|
||||
}
|
||||
|
||||
void clear_Correct_dumy(Correct_dumy* list, overlap_region_alloc* overlap_list)
|
||||
{
|
||||
list->length = 0;
|
||||
list->lengthNT = 0;
|
||||
list->start_i = 0;
|
||||
|
||||
if (list->size < overlap_list->length)
|
||||
{
|
||||
list->size = overlap_list->length;
|
||||
list->overlapID = (uint64_t*)realloc(list->overlapID, list->size*sizeof(uint64_t));
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
#ifndef __CORRECT__
|
||||
#define __CORRECT__
|
||||
#include <stdint.h>
|
||||
#include "Hash_Table.h"
|
||||
|
||||
#define WINDOW 350
|
||||
#define THRESHOLD 15
|
||||
|
||||
typedef struct
|
||||
{
|
||||
uint64_t* overlapID;
|
||||
uint64_t length;
|
||||
uint64_t lengthNT;
|
||||
uint64_t size;
|
||||
uint64_t start_i;
|
||||
char overlap_region[WINDOW + THRESHOLD*2 + 10];
|
||||
} Correct_dumy;
|
||||
|
||||
|
||||
void correct_overlap(overlap_region_alloc* overlap_list, All_reads* R_INF, UC_Read* g_read, Correct_dumy* dumy);
|
||||
|
||||
void init_Correct_dumy(Correct_dumy* list);
|
||||
void destory_Correct_dumy(Correct_dumy* list);
|
||||
void clear_Correct_dumy(Correct_dumy* list, overlap_region_alloc* overlap_list);
|
||||
|
||||
|
||||
#endif
|
||||
@@ -483,6 +483,64 @@ void print_overlap_region(Candidates_list* candidates, overlap_region_alloc* ove
|
||||
}
|
||||
|
||||
|
||||
int cmp_by_x_pos_s(const void * a, const void * b)
|
||||
{
|
||||
if ((*(overlap_region*)a).x_pos_s > (*(overlap_region*)b).x_pos_s)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
else if ((*(overlap_region*)a).x_pos_s < (*(overlap_region*)b).x_pos_s)
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
if ((*(overlap_region*)a).x_pos_e > (*(overlap_region*)b).x_pos_e)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
else if ((*(overlap_region*)a).x_pos_e < (*(overlap_region*)b).x_pos_e)
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
else
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
int cmp_by_x_pos_e(const void * a, const void * b)
|
||||
{
|
||||
if ((*(overlap_region*)a).x_pos_e > (*(overlap_region*)b).x_pos_e)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
else if ((*(overlap_region*)a).x_pos_e < (*(overlap_region*)b).x_pos_e)
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
if ((*(overlap_region*)a).x_pos_s > (*(overlap_region*)b).x_pos_s)
|
||||
{
|
||||
return 1;
|
||||
}
|
||||
else if ((*(overlap_region*)a).x_pos_s < (*(overlap_region*)b).x_pos_s)
|
||||
{
|
||||
return -1;
|
||||
}
|
||||
else
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
///r->length = Get_READ_LENGTH((*R_INF), ID);
|
||||
|
||||
@@ -500,6 +558,7 @@ uint64_t readID, uint64_t readLength, All_reads* R_INF)
|
||||
long long constant_distance = 5;
|
||||
double error_rate = 0.05;
|
||||
|
||||
|
||||
if (candidates->length == 0)
|
||||
{
|
||||
return;
|
||||
@@ -556,14 +615,19 @@ uint64_t readID, uint64_t readLength, All_reads* R_INF)
|
||||
}
|
||||
|
||||
///自己和自己重叠的要排除
|
||||
///if (tmp_region.x_id != tmp_region.y_id && tmp_region.shared_seed > 1)
|
||||
if (tmp_region.x_id != tmp_region.y_id)
|
||||
{
|
||||
|
||||
append_overlap_region_alloc(overlap_list, &tmp_region, R_INF);
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
|
||||
///以x_pos_e,即结束位置为主元排序
|
||||
///qsort(overlap_list->list, overlap_list->length, sizeof(overlap_region), cmp_by_x_pos_e);
|
||||
qsort(overlap_list->list, overlap_list->length, sizeof(overlap_region), cmp_by_x_pos_s);
|
||||
|
||||
///debug_overlap_region(candidates, overlap_list, readID);
|
||||
///print_overlap_region(candidates, overlap_list, R_INF);
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
#include "Levenshtein_distance.h"
|
||||
File diff suppressed because it is too large
Load Diff
@@ -3,7 +3,7 @@ CC=g++
|
||||
CFLAGS = -w -c -msse4.2 -mpopcnt -fomit-frame-pointer -Winline -O3 -lz
|
||||
LDFLAGS = -lm -lz -lpthread -O3 -mpopcnt -msse4.2 -lz -w
|
||||
|
||||
SOURCES = main.cpp CommandLines.cpp Process_Read.cpp Assembly.cpp kmer.cpp Hash_Table.cpp POA.cpp
|
||||
SOURCES = main.cpp CommandLines.cpp Process_Read.cpp Assembly.cpp kmer.cpp Hash_Table.cpp POA.cpp Correct.cpp Levenshtein_distance.cpp edlib.cpp
|
||||
OBJECTS = $(SOURCES:.c=.o)
|
||||
EXECUTABLE = ccs_assembly
|
||||
|
||||
|
||||
@@ -237,6 +237,111 @@ void init_UC_Read(UC_Read* r)
|
||||
}
|
||||
|
||||
|
||||
void recover_UC_Read_sub_region(char* r, long long start_pos, long long length, uint8_t strand, All_reads* R_INF, long long ID)
|
||||
{
|
||||
|
||||
|
||||
|
||||
long long readLen = Get_READ_LENGTH((*R_INF), ID);
|
||||
uint8_t* src = Get_READ((*R_INF), ID);
|
||||
|
||||
long long i;
|
||||
long long copyLen;
|
||||
long long end_pos = start_pos + length - 1;
|
||||
|
||||
if (strand == 0)
|
||||
{
|
||||
|
||||
i = start_pos;
|
||||
copyLen = 0;
|
||||
|
||||
long long initLen = start_pos % 4;
|
||||
|
||||
if (initLen != 0)
|
||||
{
|
||||
memcpy(r, bit_t_seq_table[src[i>>2]] + initLen, 4 - initLen);
|
||||
copyLen = copyLen + 4 - initLen;
|
||||
i = i + copyLen;
|
||||
}
|
||||
|
||||
|
||||
while (copyLen < length)
|
||||
{
|
||||
memcpy(r+copyLen, bit_t_seq_table[src[i>>2]], 4);
|
||||
copyLen = copyLen + 4;
|
||||
i = i + 4;
|
||||
}
|
||||
|
||||
|
||||
if (R_INF->N_site[ID])
|
||||
{
|
||||
for (i = 1; i <= R_INF->N_site[ID][0]; i++)
|
||||
{
|
||||
if (R_INF->N_site[ID][i] >= start_pos && R_INF->N_site[ID][i] <= end_pos)
|
||||
{
|
||||
r[R_INF->N_site[ID][i] - start_pos] = 'N';
|
||||
}
|
||||
else if(R_INF->N_site[ID][i] > end_pos)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
}
|
||||
else
|
||||
{
|
||||
|
||||
start_pos = readLen - start_pos - 1;
|
||||
end_pos = readLen - end_pos - 1;
|
||||
|
||||
|
||||
|
||||
///start_pos > end_pos
|
||||
i = start_pos;
|
||||
copyLen = 0;
|
||||
long long initLen = (start_pos + 1) % 4;
|
||||
|
||||
if (initLen != 0)
|
||||
{
|
||||
memcpy(r, bit_t_seq_table_rc[src[i>>2]] + 4 - initLen, initLen);
|
||||
copyLen = copyLen + initLen;
|
||||
i = i - initLen;
|
||||
}
|
||||
|
||||
while (copyLen < length)
|
||||
{
|
||||
memcpy(r+copyLen, bit_t_seq_table_rc[src[i>>2]], 4);
|
||||
copyLen = copyLen + 4;
|
||||
i = i - 4;
|
||||
}
|
||||
|
||||
if (R_INF->N_site[ID])
|
||||
{
|
||||
long long offset = readLen - start_pos - 1;
|
||||
|
||||
for (i = 1; i <= R_INF->N_site[ID][0]; i++)
|
||||
{
|
||||
|
||||
if (R_INF->N_site[ID][i] >= end_pos && R_INF->N_site[ID][i] <= start_pos)
|
||||
{
|
||||
r[readLen - R_INF->N_site[ID][i] - 1 - offset] = 'N';
|
||||
}
|
||||
else if(R_INF->N_site[ID][i] > start_pos)
|
||||
{
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
|
||||
void recover_UC_Read(UC_Read* r, All_reads* R_INF, uint64_t ID)
|
||||
{
|
||||
r->length = Get_READ_LENGTH((*R_INF), ID);
|
||||
|
||||
@@ -111,6 +111,7 @@ void compress_base(uint8_t* dest, char* src, uint64_t src_l, uint64_t** N_site_l
|
||||
void init_UC_Read(UC_Read* r);
|
||||
void recover_UC_Read(UC_Read* r, All_reads* R_INF, uint64_t ID);
|
||||
void recover_UC_Read_RC(UC_Read* r, All_reads* R_INF, uint64_t ID);
|
||||
void recover_UC_Read_sub_region(char* r, long long start_pos, long long length, uint8_t strand, All_reads* R_INF, long long ID);
|
||||
void destory_UC_Read(UC_Read* r);
|
||||
void reverse_complement(char* pattern, uint64_t length);
|
||||
void write_All_reads(All_reads* r, char* read_file_name);
|
||||
|
||||
@@ -0,0 +1,258 @@
|
||||
#ifndef EDLIB_H
|
||||
#define EDLIB_H
|
||||
|
||||
/**
|
||||
* @file
|
||||
* @author Martin Sosic
|
||||
* @brief Main header file, containing all public functions and structures.
|
||||
*/
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
// Status codes
|
||||
#define EDLIB_STATUS_OK 0
|
||||
#define EDLIB_STATUS_ERROR 1
|
||||
|
||||
/**
|
||||
* Alignment methods - how should Edlib treat gaps before and after query?
|
||||
*/
|
||||
typedef enum {
|
||||
/**
|
||||
* Global method. This is the standard method.
|
||||
* Useful when you want to find out how similar is first sequence to second sequence.
|
||||
*/
|
||||
EDLIB_MODE_NW,
|
||||
/**
|
||||
* Prefix method. Similar to global method, but with a small twist - gap at query end is not penalized.
|
||||
* What that means is that deleting elements from the end of second sequence is "free"!
|
||||
* For example, if we had "AACT" and "AACTGGC", edit distance would be 0, because removing "GGC" from the end
|
||||
* of second sequence is "free" and does not count into total edit distance. This method is appropriate
|
||||
* when you want to find out how well first sequence fits at the beginning of second sequence.
|
||||
*/
|
||||
EDLIB_MODE_SHW,
|
||||
/**
|
||||
* Infix method. Similar as prefix method, but with one more twist - gaps at query end and start are
|
||||
* not penalized. What that means is that deleting elements from the start and end of second sequence is "free"!
|
||||
* For example, if we had ACT and CGACTGAC, edit distance would be 0, because removing CG from the start
|
||||
* and GAC from the end of second sequence is "free" and does not count into total edit distance.
|
||||
* This method is appropriate when you want to find out how well first sequence fits at any part of
|
||||
* second sequence.
|
||||
* For example, if your second sequence was a long text and your first sequence was a sentence from that text,
|
||||
* but slightly scrambled, you could use this method to discover how scrambled it is and where it fits in
|
||||
* that text. In bioinformatics, this method is appropriate for aligning read to a sequence.
|
||||
*/
|
||||
EDLIB_MODE_HW
|
||||
} EdlibAlignMode;
|
||||
|
||||
/**
|
||||
* Alignment tasks - what do you want Edlib to do?
|
||||
*/
|
||||
typedef enum {
|
||||
EDLIB_TASK_DISTANCE, //!< Find edit distance and end locations.
|
||||
EDLIB_TASK_LOC, //!< Find edit distance, end locations and start locations.
|
||||
EDLIB_TASK_PATH //!< Find edit distance, end locations and start locations and alignment path.
|
||||
} EdlibAlignTask;
|
||||
|
||||
/**
|
||||
* Describes cigar format.
|
||||
* @see http://samtools.github.io/hts-specs/SAMv1.pdf
|
||||
* @see http://drive5.com/usearch/manual/cigar.html
|
||||
*/
|
||||
typedef enum {
|
||||
EDLIB_CIGAR_STANDARD, //!< Match: 'M', Insertion: 'I', Deletion: 'D', Mismatch: 'M'.
|
||||
EDLIB_CIGAR_EXTENDED //!< Match: '=', Insertion: 'I', Deletion: 'D', Mismatch: 'X'.
|
||||
} EdlibCigarFormat;
|
||||
|
||||
// Edit operations.
|
||||
#define EDLIB_EDOP_MATCH 0 //!< Match.
|
||||
#define EDLIB_EDOP_INSERT 1 //!< Insertion to target = deletion from query.
|
||||
#define EDLIB_EDOP_DELETE 2 //!< Deletion from target = insertion to query.
|
||||
#define EDLIB_EDOP_MISMATCH 3 //!< Mismatch.
|
||||
|
||||
/**
|
||||
* @brief Defines two given characters as equal.
|
||||
*/
|
||||
typedef struct {
|
||||
char first;
|
||||
char second;
|
||||
} EdlibEqualityPair;
|
||||
|
||||
/**
|
||||
* @brief Configuration object for edlibAlign() function.
|
||||
*/
|
||||
typedef struct {
|
||||
/**
|
||||
* Set k to non-negative value to tell edlib that edit distance is not larger than k.
|
||||
* Smaller k can significantly improve speed of computation.
|
||||
* If edit distance is larger than k, edlib will set edit distance to -1.
|
||||
* Set k to negative value and edlib will internally auto-adjust k until score is found.
|
||||
*/
|
||||
int k;
|
||||
|
||||
/**
|
||||
* Alignment method.
|
||||
* EDLIB_MODE_NW: global (Needleman-Wunsch)
|
||||
* EDLIB_MODE_SHW: prefix. Gap after query is not penalized.
|
||||
* EDLIB_MODE_HW: infix. Gaps before and after query are not penalized.
|
||||
*/
|
||||
EdlibAlignMode mode;
|
||||
|
||||
/**
|
||||
* Alignment task - tells Edlib what to calculate. Less to calculate, faster it is.
|
||||
* EDLIB_TASK_DISTANCE - find edit distance and end locations of optimal alignment paths in target.
|
||||
* EDLIB_TASK_LOC - find edit distance and start and end locations of optimal alignment paths in target.
|
||||
* EDLIB_TASK_PATH - find edit distance, alignment path (and start and end locations of it in target).
|
||||
*/
|
||||
EdlibAlignTask task;
|
||||
|
||||
/**
|
||||
* List of pairs of characters, where each pair defines two characters as equal.
|
||||
* This way you can extend edlib's definition of equality (which is that each character is equal only
|
||||
* to itself).
|
||||
* This can be useful if you have some wildcard characters that should match multiple other characters,
|
||||
* or e.g. if you want edlib to be case insensitive.
|
||||
* Can be set to NULL if there are none.
|
||||
*/
|
||||
EdlibEqualityPair* additionalEqualities;
|
||||
|
||||
/**
|
||||
* Number of additional equalities, which is non-negative number.
|
||||
* 0 if there are none.
|
||||
*/
|
||||
int additionalEqualitiesLength;
|
||||
} EdlibAlignConfig;
|
||||
|
||||
/**
|
||||
* Helper method for easy construction of configuration object.
|
||||
* @return Configuration object filled with given parameters.
|
||||
*/
|
||||
EdlibAlignConfig edlibNewAlignConfig(int k, EdlibAlignMode mode, EdlibAlignTask task,
|
||||
EdlibEqualityPair* additionalEqualities,
|
||||
int additionalEqualitiesLength);
|
||||
|
||||
/**
|
||||
* @return Default configuration object, with following defaults:
|
||||
* k = -1, mode = EDLIB_MODE_NW, task = EDLIB_TASK_DISTANCE, no additional equalities.
|
||||
*/
|
||||
EdlibAlignConfig edlibDefaultAlignConfig(void);
|
||||
|
||||
|
||||
/**
|
||||
* Container for results of alignment done by edlibAlign() function.
|
||||
*/
|
||||
typedef struct {
|
||||
/**
|
||||
* EDLIB_STATUS_OK or EDLIB_STATUS_ERROR. If error, all other fields will have undefined values.
|
||||
*/
|
||||
int status;
|
||||
|
||||
/**
|
||||
* -1 if k is non-negative and edit distance is larger than k.
|
||||
*/
|
||||
int editDistance;
|
||||
|
||||
/**
|
||||
* Array of zero-based positions in target where optimal alignment paths end.
|
||||
* If gap after query is penalized, gap counts as part of query (NW), otherwise not.
|
||||
* Set to NULL if edit distance is larger than k.
|
||||
* If you do not free whole result object using edlibFreeAlignResult(), do not forget to use free().
|
||||
*/
|
||||
int* endLocations;
|
||||
|
||||
/**
|
||||
* Array of zero-based positions in target where optimal alignment paths start,
|
||||
* they correspond to endLocations.
|
||||
* If gap before query is penalized, gap counts as part of query (NW), otherwise not.
|
||||
* Set to NULL if not calculated or if edit distance is larger than k.
|
||||
* If you do not free whole result object using edlibFreeAlignResult(), do not forget to use free().
|
||||
*/
|
||||
int* startLocations;
|
||||
|
||||
/**
|
||||
* Number of end (and start) locations.
|
||||
*/
|
||||
int numLocations;
|
||||
|
||||
/**
|
||||
* Alignment is found for first pair of start and end locations.
|
||||
* Set to NULL if not calculated.
|
||||
* Alignment is sequence of numbers: 0, 1, 2, 3.
|
||||
* 0 stands for match.
|
||||
* 1 stands for insertion to target.
|
||||
* 2 stands for insertion to query.
|
||||
* 3 stands for mismatch.
|
||||
* Alignment aligns query to target from begining of query till end of query.
|
||||
* If gaps are not penalized, they are not in alignment.
|
||||
* If you do not free whole result object using edlibFreeAlignResult(), do not forget to use free().
|
||||
*/
|
||||
unsigned char* alignment;
|
||||
|
||||
/**
|
||||
* Length of alignment.
|
||||
*/
|
||||
int alignmentLength;
|
||||
|
||||
/**
|
||||
* Number of different characters in query and target together.
|
||||
*/
|
||||
int alphabetLength;
|
||||
} EdlibAlignResult;
|
||||
|
||||
/**
|
||||
* Frees memory in EdlibAlignResult that was allocated by edlib.
|
||||
* If you do not use it, make sure to free needed members manually using free().
|
||||
*/
|
||||
void edlibFreeAlignResult(EdlibAlignResult result);
|
||||
|
||||
|
||||
/**
|
||||
* Aligns two sequences (query and target) using edit distance (levenshtein distance).
|
||||
* Through config parameter, this function supports different alignment methods (global, prefix, infix),
|
||||
* as well as different modes of search (tasks).
|
||||
* It always returns edit distance and end locations of optimal alignment in target.
|
||||
* It optionally returns start locations of optimal alignment in target and alignment path,
|
||||
* if you choose appropriate tasks.
|
||||
* @param [in] query First sequence.
|
||||
* @param [in] queryLength Number of characters in first sequence.
|
||||
* @param [in] target Second sequence.
|
||||
* @param [in] targetLength Number of characters in second sequence.
|
||||
* @param [in] config Additional alignment parameters, like alignment method and wanted results.
|
||||
* @return Result of alignment, which can contain edit distance, start and end locations and alignment path.
|
||||
* Make sure to clean up the object using edlibFreeAlignResult() or by manually freeing needed members.
|
||||
*/
|
||||
EdlibAlignResult edlibAlign(const char* query, int queryLength,
|
||||
const char* target, int targetLength,
|
||||
const EdlibAlignConfig config);
|
||||
|
||||
|
||||
/**
|
||||
* Builds cigar string from given alignment sequence.
|
||||
* @param [in] alignment Alignment sequence.
|
||||
* 0 stands for match.
|
||||
* 1 stands for insertion to target.
|
||||
* 2 stands for insertion to query.
|
||||
* 3 stands for mismatch.
|
||||
* @param [in] alignmentLength
|
||||
* @param [in] cigarFormat Cigar will be returned in specified format.
|
||||
* @return Cigar string.
|
||||
* I stands for insertion.
|
||||
* D stands for deletion.
|
||||
* X stands for mismatch. (used only in extended format)
|
||||
* = stands for match. (used only in extended format)
|
||||
* M stands for (mis)match. (used only in standard format)
|
||||
* String is null terminated.
|
||||
* Needed memory is allocated and given pointer is set to it.
|
||||
* Do not forget to free it later using free()!
|
||||
*/
|
||||
char* edlibAlignmentToCigar(const unsigned char* alignment, int alignmentLength,
|
||||
EdlibCigarFormat cigarFormat);
|
||||
|
||||
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif // EDLIB_H
|
||||
@@ -3,7 +3,8 @@
|
||||
#include "CommandLines.h"
|
||||
#include "Process_Read.h"
|
||||
#include "Assembly.h"
|
||||
|
||||
#include "Levenshtein_distance.h"
|
||||
#include "edlib.h"
|
||||
/********************************for debug***************************************/
|
||||
///使用这个函数的时候,必须把Counting_multiple_thr()里的destory_Total_Count_Table(&TCB)注释掉
|
||||
void debug_Counting()
|
||||
@@ -15,6 +16,182 @@ void debug_Counting()
|
||||
}
|
||||
|
||||
|
||||
int matrix[1000][1000] = {0};
|
||||
///y_length > x_length
|
||||
int edit_distance_normal(char* y, int y_length, char* x, int x_length)
|
||||
{ memset(matrix, 0, sizeof(matrix));
|
||||
|
||||
int i, j;
|
||||
for (i = 0; i <= x_length; i++)
|
||||
{
|
||||
matrix[i][0] = i;
|
||||
}
|
||||
|
||||
int digonal, up, left, min;
|
||||
|
||||
///一列列算的
|
||||
for (i = 0; i < x_length; i++)
|
||||
{
|
||||
for (j = 0; j < y_length; j++)
|
||||
{
|
||||
///matrix[i + 1][j + 1]
|
||||
digonal = matrix[i][j] + (x[i] != y[j]);
|
||||
up = matrix[i + 1][j] + 1;
|
||||
left = matrix[i][j + 1] + 1;
|
||||
min = digonal;
|
||||
if (up < min)
|
||||
{
|
||||
min = up;
|
||||
}
|
||||
|
||||
if (left< min)
|
||||
{
|
||||
min = left;
|
||||
}
|
||||
|
||||
matrix[i + 1][j + 1] = min;
|
||||
}
|
||||
}
|
||||
|
||||
min = 999999;
|
||||
for (j = 0; j <= y_length; j++)
|
||||
{
|
||||
if (matrix[i][j] < min)
|
||||
{
|
||||
min = matrix[i][j];
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
return min;
|
||||
}
|
||||
|
||||
|
||||
///y_length > x_length
|
||||
int edit_distance_normal_banded(char* y, int y_length, char* x, int x_length, int error)
|
||||
{ memset(matrix, 0, sizeof(matrix));
|
||||
|
||||
int i, j;
|
||||
for (i = 0; i <= x_length; i++)
|
||||
{
|
||||
for (j = 0; j <= y_length; j++)
|
||||
{
|
||||
matrix[i][j] = 1000000;
|
||||
}
|
||||
}
|
||||
|
||||
for (i = 0; i <= x_length; i++)
|
||||
{
|
||||
matrix[i][0] = i;
|
||||
}
|
||||
|
||||
for (i = 0; i <= y_length; i++)
|
||||
{
|
||||
matrix[0][i] = 0;
|
||||
}
|
||||
|
||||
int banded_length = error*2 + 1;
|
||||
|
||||
int digonal, up, left, min;
|
||||
|
||||
|
||||
for (i = 0; i < x_length; i++)
|
||||
{
|
||||
///for (j = 0; j < y_length; j++)
|
||||
for (j = i; j < banded_length + i; j++)
|
||||
{
|
||||
///matrix[i + 1][j + 1]
|
||||
digonal = matrix[i][j] + (x[i] != y[j]);
|
||||
up = matrix[i + 1][j] + 1;
|
||||
left = matrix[i][j + 1] + 1;
|
||||
min = digonal;
|
||||
if (up < min)
|
||||
{
|
||||
min = up;
|
||||
}
|
||||
|
||||
if (left< min)
|
||||
{
|
||||
min = left;
|
||||
}
|
||||
|
||||
matrix[i + 1][j + 1] = min;
|
||||
}
|
||||
}
|
||||
min = 999999;
|
||||
for (j = 0; j <= y_length; j++)
|
||||
///for (j = i; j < banded_length + i; j++)
|
||||
{
|
||||
if (matrix[i][j] < min)
|
||||
{
|
||||
min = matrix[i][j];
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
return min;
|
||||
}
|
||||
|
||||
void debug_edit_distance()
|
||||
{
|
||||
/**
|
||||
char* x = "TTCCATACGATTCCATTCAATTCGAGACCATTCTATTCCTGTCCATTCCTTGTGGTTCGATTCCATTTCACTCTAGTCCATTCCATTCCATTCAATTCCATTCGACTCTATTCCGTTCCACTCAATTCCATTCCATTCGATTCCATTTTTTTCGAGAACCTTCCATTACACTCCCTTCCATTCCAGTGCATTCCATTCCAGTCTCTTCAGTTCGATTCCATTCCATTCGTTTCGATTCCTTTCCATTCCAGCCCATTCCATTCCATTCCATTCCTTTCCTTTCCGTTTCATTAGATTCCATTGCATTCGATTCCATTCAAATCAATTCCGTTCTATTCAATTTGATTCAT";
|
||||
char* y = "CCATACGATTCCATTCAATTCGAGACCATTCTATTCCTGTCCATTCCTTGTGGTTCGATTCCATTTCACTCTAGTCCATTCCATTCCATTCAATTCCATTCGACTCTATTCCGTTCCATTCAATTCCATTCCATTCGATTCCATTTTTTTCGAGAACCTTCCATTACACTCCCTTCCATTCCAGTGCATTCCATTCCAGTCTCTTCACTTCGATTCCATTCCATTCGTTTCGATTCCTTTCCATTCCAGCCCATTCCATTCCATTCCATTCCTTTCCTTTCCGTTTCATTAGATTCCATTGCATTCCATTCCATTCAATTCAATTCCGTGCTATTCAATTTGATTCATTTCCATTTAATTCCATTCCATTAGATTCCATT";
|
||||
**/
|
||||
unsigned short toold = 7;
|
||||
char* x = "TTCCATACGATTCCATTCAATTCGAGACCATTCTATTCCT";
|
||||
char* y = "CCATACGATTCCATTCAATTCGAGACCATTCTATTCCTGTCCATTCCTTGTGGT";
|
||||
fprintf(stderr, "x_length: %u\n", strlen(x));
|
||||
fprintf(stderr, "y_length: %u\n", strlen(y));
|
||||
|
||||
|
||||
EdlibAlignResult result = edlibAlign(x, strlen(x), y, strlen(y),
|
||||
edlibNewAlignConfig(toold, EDLIB_MODE_HW, EDLIB_TASK_PATH, NULL, 0));
|
||||
|
||||
if (result.status == EDLIB_STATUS_OK) {
|
||||
|
||||
fprintf(stderr, "****\nedlib: %d, alignmentLength: %d, startLocations: %d, endLocations: %d\n",
|
||||
result.editDistance, result.alignmentLength, result.startLocations[0], result.endLocations[0]);
|
||||
char* cigar = edlibAlignmentToCigar(result.alignment, result.alignmentLength, EDLIB_CIGAR_STANDARD);
|
||||
fprintf(stderr,"%s\n", cigar);
|
||||
free(cigar);
|
||||
}
|
||||
edlibFreeAlignResult(result);
|
||||
|
||||
|
||||
|
||||
unsigned int error;
|
||||
int end_site = Reserve_Banded_BPM(y, strlen(y), x, strlen(x), toold, &error);
|
||||
|
||||
fprintf(stderr, "BPM: error: %u, end_site: %u\n", error, end_site);
|
||||
|
||||
|
||||
unsigned short band_length=(toold+1)*3-1-1-toold;
|
||||
unsigned short band_down=toold-1;
|
||||
unsigned short band_blew=2*(toold+1)-1-1;
|
||||
|
||||
|
||||
|
||||
int return_err = 99999;
|
||||
///注意pattern/text和band_down/band_blew是反的
|
||||
Reserve_Banded_BPM_new(y, strlen(y), x, strlen(x),
|
||||
toold,band_blew,band_down,band_length, &return_err, 0);
|
||||
|
||||
fprintf(stderr, "new BPM: error: %u\n", return_err);
|
||||
|
||||
|
||||
return_err = edit_distance_normal(y, strlen(y), x, strlen(x));
|
||||
fprintf(stderr, "edit_distance_normal: error: %u\n", return_err);
|
||||
|
||||
return_err = edit_distance_normal_banded(y, strlen(y), x, strlen(x), toold);
|
||||
fprintf(stderr, "edit_distance_normal_banded: error: %u\n", return_err);
|
||||
|
||||
end_site = Reserve_Banded_BPM_debug(y, strlen(y), x, strlen(x), toold, &error, matrix);
|
||||
|
||||
fprintf(stderr, "BPM debug: error: %u, end_site: %u\n", error, end_site);
|
||||
}
|
||||
|
||||
|
||||
int main(int argc, char *argv[])
|
||||
{
|
||||
|
||||
@@ -22,6 +199,13 @@ int main(int argc, char *argv[])
|
||||
return 1;
|
||||
|
||||
|
||||
/**
|
||||
debug_edit_distance();
|
||||
|
||||
return 1;
|
||||
**/
|
||||
|
||||
|
||||
fprintf(stdout, "k-mer length: %d\n",k_mer_length);
|
||||
|
||||
if (load_index_from_disk && load_pre_cauculated_index())
|
||||
|
||||
Reference in New Issue
Block a user