mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-06 12:38:12 +08:00
Improve diffexp for tiledb (#1388)
* Improve diffexp for tiledb - The rows from the A and B sets are gathered and processed at the same time. In this way the matrix is only accessed once instead of twice for each tile. - There is now a single thread queue that gets shared between all callers of the diffexp. This will slow down work if diffexp gets too busy. - There is a target_workunit amount of work given to each thread. Previously the workunit was (rows selected * width of tile), which could be small. Now multiple column tiles can be combined into one workunit. If the target is too small then thread and other overheads may reduce performance. If target_workunit is too large then the size of the gathered sub matrix may take up too much memory. - add configuration parameters (max_workers, cpu_multiplier, and target_workunit)
This commit is contained in:
@@ -8,7 +8,7 @@ from server.common.constants import Axis
|
||||
from server.data_common.data_adaptor import DataAdaptor
|
||||
from server.data_common.fbs.matrix import encode_matrix_fbs
|
||||
from server.data_cxg.cxg_util import pack_selector_from_mask
|
||||
import server.compute.diffexp_tiledb as diffexp_tiledb
|
||||
import server.compute.diffexp_cxg as diffexp_cxg
|
||||
from server.common.immutable_kvcache import ImmutableKVCache
|
||||
import tiledb
|
||||
import numpy as np
|
||||
@@ -192,7 +192,7 @@ class CxgAdaptor(DataAdaptor):
|
||||
top_n = self.config.diffexp__top_n
|
||||
if lfc_cutoff is None:
|
||||
lfc_cutoff = self.config.diffexp__lfc_cutoff
|
||||
return diffexp_tiledb.diffexp_ttest(self, maskA, maskB, top_n, lfc_cutoff)
|
||||
return diffexp_cxg.diffexp_ttest(self, maskA, maskB, top_n, lfc_cutoff)
|
||||
|
||||
def get_X_array(self, obs_mask=None, var_mask=None):
|
||||
obs_items = pack_selector_from_mask(obs_mask)
|
||||
|
||||
Reference in New Issue
Block a user