fix for incorrect stats computation in diff exp t-test (#2318)

* 2211 fixes

* lint

* lint

* add missing test and bug found by test

* change terminology for count distribution

* update scanpy requirement

* update scanpy requirement
This commit is contained in:
Bruce Martin
2021-07-23 11:36:26 -07:00
committed by GitHub
parent 1ebde2213d
commit 1ea2b7fe80
28 changed files with 336 additions and 90 deletions
@@ -44,6 +44,8 @@ class DatasetConfig(BaseConfig):
self.diffexp__lfc_cutoff = default_config["diffexp"]["lfc_cutoff"]
self.diffexp__top_n = default_config["diffexp"]["top_n"]
self.X_approx_distribution = default_config["X_approx_distribution"]
except KeyError as e:
raise ConfigurationError(f"Unexpected config: {str(e)}")
@@ -58,6 +60,7 @@ class DatasetConfig(BaseConfig):
self.handle_user_annotations(context)
self.handle_embeddings()
self.handle_diffexp(context)
self.handle_X_approx_distribution()
def handle_app(self):
self.validate_correct_type_of_configuration_attribute("app__scripts", list)
@@ -199,3 +202,10 @@ class DatasetConfig(BaseConfig):
"CAUTION: due to the size of your dataset, "
"running differential expression may take longer or fail."
)
def handle_X_approx_distribution(self):
self.validate_correct_type_of_configuration_attribute("X_approx_distribution", str)
if self.X_approx_distribution not in ["normal", "count"]:
raise ConfigurationError(
"X_approx_distribution has unknown value -- must be 'normal' or 'count'."
)
@@ -8,9 +8,9 @@ from scipy import sparse
import backend.common.compute.diffexp_generic as diffexp_generic
from backend.common.colors import convert_anndata_category_colors_to_cxg_category_colors
from backend.common.constants import Axis, MAX_LAYOUTS
from backend.common.constants import Axis, MAX_LAYOUTS, XApproxDistribution
from backend.czi_hosted.common.corpora import corpora_get_props_from_anndata
from backend.common.errors import PrepareError, DatasetAccessError
from backend.common.errors import PrepareError, DatasetAccessError, ConfigurationError
from backend.common.utils.type_conversion_utils import get_schema_type_hint_of_array
from backend.czi_hosted.data_common.data_adaptor import DataAdaptor
from backend.common.fbs.matrix import encode_matrix_fbs
@@ -28,6 +28,7 @@ class AnndataAdaptor(DataAdaptor):
def __init__(self, data_locator, app_config=None, dataset_config=None):
super().__init__(data_locator, app_config, dataset_config)
self.data = None
self.X_approx_distribution = None
self._load_data(data_locator)
self._validate_and_initialize()
@@ -190,6 +191,10 @@ class AnndataAdaptor(DataAdaptor):
self.gene_count = self.data.shape[1]
self._create_schema()
if self.dataset_config.X_approx_distribution == "auto":
raise ConfigurationError("X-approx-distribution 'auto' mode unsupported.")
self.X_approx_distribution = self.dataset_config.X_approx_distribution
# heuristic
n_values = self.data.shape[0] * self.data.shape[1]
if (n_values > 1e8 and self.server_config.adaptor__anndata_adaptor__backed is True) or (n_values > 5e8):
@@ -309,13 +314,22 @@ class AnndataAdaptor(DataAdaptor):
return convert_anndata_category_colors_to_cxg_category_colors(self.data)
def get_X_array(self, obs_mask=None, var_mask=None):
# H5Py does not support boolean indexing (masks), so convert to integer indexing
# when backed (ie, when AnnData is using H5Py indexing)
if obs_mask is None:
obs_mask = slice(None)
elif self.data.isbacked and obs_mask.dtype == bool:
obs_mask = obs_mask.nonzero()[0]
if var_mask is None:
var_mask = slice(None)
elif self.data.isbacked and var_mask.dtype == bool:
var_mask = var_mask.nonzero()[0]
X = self.data.X[obs_mask, var_mask]
return X
def get_X_approx_distribution(self) -> XApproxDistribution:
return self.X_approx_distribution
def get_shape(self):
return self.data.shape
+17 -5
View File
@@ -7,8 +7,14 @@ from scipy import sparse
from server_timing import Timing as ServerTiming
from backend.czi_hosted.common.config.app_config import AppConfig
from backend.common.constants import Axis
from backend.common.errors import FilterError, JSONEncodingValueError, ExceedsLimitError, UnsupportedSummaryMethod, DatasetAccessError
from backend.common.constants import Axis, XApproxDistribution
from backend.common.errors import (
FilterError,
JSONEncodingValueError,
ExceedsLimitError,
UnsupportedSummaryMethod,
DatasetAccessError,
)
from backend.common.utils.utils import jsonify_numpy
from backend.common.fbs.matrix import encode_matrix_fbs
@@ -77,6 +83,11 @@ class DataAdaptor(metaclass=ABCMeta):
the return type is either ndarray or scipy.sparse.spmatrix."""
pass
@abstractmethod
def get_X_approx_distribution(self) -> XApproxDistribution:
"""return the approximate distribution of the X matrix."""
pass
@abstractmethod
def get_shape(self):
pass
@@ -158,7 +169,7 @@ class DataAdaptor(metaclass=ABCMeta):
mask = np.zeros((count,), dtype=np.bool)
for i in filter:
if type(i) == list:
mask[i[0]: i[1]] = True
mask[i[0] : i[1]] = True
else:
mask[i] = True
return mask
@@ -316,12 +327,13 @@ class DataAdaptor(metaclass=ABCMeta):
top_n = self.dataset_config.diffexp__top_n
if self.server_config.exceeds_limit(
"diffexp_cellcount_max", np.count_nonzero(obs_mask_A) + np.count_nonzero(obs_mask_B)
"diffexp_cellcount_max", np.count_nonzero(obs_mask_A) + np.count_nonzero(obs_mask_B)
):
raise ExceedsLimitError("Diffexp request exceeds max cell count limit")
result = self.compute_diffexp_ttest(
maskA=obs_mask_A, maskB=obs_mask_B, top_n=top_n, lfc_cutoff=self.dataset_config.diffexp__lfc_cutoff)
maskA=obs_mask_A, maskB=obs_mask_B, top_n=top_n, lfc_cutoff=self.dataset_config.diffexp__lfc_cutoff
)
try:
return jsonify_numpy(result)
+9 -1
View File
@@ -8,7 +8,7 @@ import pandas as pd
import tiledb
from server_timing import Timing as ServerTiming
from backend.common.constants import Axis
from backend.common.constants import Axis, XApproxDistribution
from backend.common.errors import DatasetAccessError, ConfigurationError
from backend.czi_hosted.common.immutable_kvcache import ImmutableKVCache
from backend.common.utils.type_conversion_utils import get_schema_type_hint_from_dtype
@@ -37,6 +37,7 @@ class CxgAdaptor(DataAdaptor):
self.lsuri_results = ImmutableKVCache(lambda key: self._lsuri(uri=key, tiledb_ctx=self.tiledb_ctx))
self.arrays = ImmutableKVCache(lambda key: self._open_array(uri=key, tiledb_ctx=self.tiledb_ctx))
self.schema = None
self.X_approx_distribution = None
self._validate_and_initialize()
@@ -175,6 +176,10 @@ class CxgAdaptor(DataAdaptor):
if cxg_version not in ["0.0", "0.1", "0.2.0"]:
raise DatasetAccessError(f"cxg matrix is not valid: {self.url}")
if self.dataset_config.X_approx_distribution == "auto":
raise ConfigurationError("X-approx-distribution 'auto' mode unsupported.")
self.X_approx_distribution = self.dataset_config.X_approx_distribution
self.title = title
self.about = about
self.cxg_version = cxg_version
@@ -281,6 +286,9 @@ class CxgAdaptor(DataAdaptor):
data = X.multi_index[obs_items, var_items][""]
return data
def get_X_approx_distribution(self) -> XApproxDistribution:
return self.X_approx_distribution
def get_shape(self):
X = self.open_array("X")
return X.shape
+2
View File
@@ -203,6 +203,8 @@ dataset:
lfc_cutoff: 0.01
top_n: 10
X_approx_distribution: normal # currently fixed config
external:
# You can retrieve configuration parameters from this config file, the environment,
# the AWS secrets manager, or from the "cellxgene launch" command line arguments.
+1 -1
View File
@@ -1,4 +1,4 @@
python-igraph
louvain>=0.6
scanpy==1.4.6 # Until we move to anndata 0.7.4 scanpy needs to be pinned here
scanpy
umap-learn<0.5.0 # The pinned version scanpy is not compatible with latest umap-learn
+2 -2
View File
@@ -1,4 +1,4 @@
anndata>=0.7.0
anndata>=0.7.6 # we use to_memory(), added in 0.7.6
boto3>=1.12.18
click>=7.1.2
Flask>=1.0.2,<2.0.0 # Flask 2.0 is not compatible with the latest version of Flask-RESTful (0.3.8)
@@ -11,7 +11,7 @@ flatbuffers>=1.11.0,<2.0.0 # cellxgene is not compatible with 2.0.0. Requires mi
flatten-dict>=0.2.0
fsspec>=0.4.4,<0.8.0
gunicorn>=20.0.4
h5py<3.0.0 # h5py>=3.0.0 had a breaking change; there is a fix in anndata>=0.7.5
h5py>=3.0.0
numba>=0.49.1,<0.53.0
numpy>=1.15.0
packaging>=20.0