Improve diffexp for tiledb (#1388)

* Improve diffexp for tiledb

- The rows from the A and B sets are gathered and processed at the same time.  In this
  way the matrix is only accessed once instead of twice for each tile.
- There is now a single thread queue that gets shared between all callers of the diffexp.
  This will slow down work if diffexp gets too busy.
- There is a target_workunit amount of work given to each thread.  Previously the
  workunit was (rows selected * width of tile), which could be small.  Now multiple
  column tiles can be combined into one workunit.  If the target is too small then
  thread and other overheads may reduce performance.  If target_workunit is too large
  then the size of the gathered sub matrix may take up too much memory.
- add configuration parameters (max_workers, cpu_multiplier, and  target_workunit)
This commit is contained in:
bmccandless
2020-04-13 18:53:13 -07:00
committed by GitHub
parent 136093d583
commit 5c0b8c6296
8 changed files with 233 additions and 130 deletions
+19 -4
View File
@@ -1,6 +1,6 @@
from server import __version__ as cellxgene_version
from flatten_dict import flatten
from os import mkdir, environ
import os
from os.path import splitext, basename, isdir
import sys
from urllib.parse import urlparse
@@ -16,8 +16,9 @@ from server.common.utils import find_available_port, is_port_available
import warnings
from server.common.annotations import AnnotationsLocalFile
from server.common.utils import custom_format_warning
import server.compute.diffexp_cxg as diffexp_tiledb
DEFAULT_SERVER_PORT = int(environ.get("CXG_SERVER_PORT", "5005"))
DEFAULT_SERVER_PORT = int(os.environ.get("CXG_SERVER_PORT", "5005"))
# anything bigger than this will generate a special message
BIG_FILE_SIZE_THRESHOLD = 100 * 2 ** 20 # 100MB
@@ -85,6 +86,9 @@ class AppConfig(object):
self.diffexp__enable = dc["diffexp"]["enable"]
self.diffexp__lfc_cutoff = dc["diffexp"]["lfc_cutoff"]
self.diffexp__top_n = dc["diffexp"]["top_n"]
self.diffexp__alg_cxg__max_workers = dc["diffexp"]["alg_cxg"]["max_workers"]
self.diffexp__alg_cxg__cpu_multiplier = dc["diffexp"]["alg_cxg"]["cpu_multiplier"]
self.diffexp__alg_cxg__target_workunit = dc["diffexp"]["alg_cxg"]["target_workunit"]
self.data_locator__s3__region_name = dc["data_locator"]["s3"]["region_name"]
@@ -256,7 +260,7 @@ class AppConfig(object):
# secret key:
# first, from CXG_SECRET_KEY environment variable
# second, from config file
self.server__flask_secret_key = environ.get("CXG_SECRET_KEY", self.server__flask_secret_key)
self.server__flask_secret_key = os.environ.get("CXG_SECRET_KEY", self.server__flask_secret_key)
def handle_data_locator(self, context):
self.__check_attr("data_locator__s3__region_name", (type(None), bool, str))
@@ -381,7 +385,7 @@ class AppConfig(object):
if dirname is not None and not isdir(dirname):
try:
mkdir(dirname)
os.mkdir(dirname)
except OSError:
raise ConfigurationError("Unable to create directory specified by --annotations-dir")
@@ -433,6 +437,9 @@ class AppConfig(object):
self.__check_attr("diffexp__enable", bool)
self.__check_attr("diffexp__lfc_cutoff", float)
self.__check_attr("diffexp__top_n", int)
self.__check_attr("diffexp__alg_cxg__max_workers", (str, int))
self.__check_attr("diffexp__alg_cxg__cpu_multiplier", int)
self.__check_attr("diffexp__alg_cxg__target_workunit", int)
if self.single_dataset__datapath:
with self.matrix_data_cache_manager.data_adaptor(self.single_dataset__datapath, self) as data_adaptor:
@@ -442,6 +449,14 @@ class AppConfig(object):
f"running differential expression may take longer or fail."
)
max_workers = self.diffexp__alg_cxg__max_workers
cpu_multiplier = self.diffexp__alg_cxg__cpu_multiplier
cpu_count = os.cpu_count()
max_workers = min(max_workers, cpu_multiplier * cpu_count)
diffexp_tiledb.set_config(
max_workers,
self.diffexp__alg_cxg__target_workunit)
def handle_adaptor(self, context):
# cxg
self.__check_attr("adaptor__cxg_adaptor__tiledb_ctx", dict)
+9
View File
@@ -66,6 +66,15 @@ diffexp:
enable: true
lfc_cutoff: 0.01
top_n: 10
alg_cxg:
# The number of threads to use is computed from: min(max_workers, cpu_multipler * cpu_count).
# Where cpu_count is determined at runtime.
max_workers: 64
cpu_multiplier: 4
# The target number of matrix elements that are evaluated
# together in one thread.
target_workunit: 16_000_000
data_locator:
s3:
+8
View File
@@ -76,3 +76,11 @@ class ExceedsLimitError(Exception):
"""
pass
class ComputeError(Exception):
"""
Raised when an error occurs during a compute algorithm (such as diffexp)
"""
pass