mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-03 17:08:12 +08:00
fix for incorrect stats computation in diff exp t-test (#2318)
* 2211 fixes * lint * lint * add missing test and bug found by test * change terminology for count distribution * update scanpy requirement * update scanpy requirement
This commit is contained in:
@@ -44,6 +44,8 @@ class DatasetConfig(BaseConfig):
|
||||
self.diffexp__lfc_cutoff = default_config["diffexp"]["lfc_cutoff"]
|
||||
self.diffexp__top_n = default_config["diffexp"]["top_n"]
|
||||
|
||||
self.X_approx_distribution = default_config["X_approx_distribution"]
|
||||
|
||||
except KeyError as e:
|
||||
raise ConfigurationError(f"Unexpected config: {str(e)}")
|
||||
|
||||
@@ -58,6 +60,7 @@ class DatasetConfig(BaseConfig):
|
||||
self.handle_user_annotations(context)
|
||||
self.handle_embeddings()
|
||||
self.handle_diffexp(context)
|
||||
self.handle_X_approx_distribution()
|
||||
|
||||
def handle_app(self):
|
||||
self.validate_correct_type_of_configuration_attribute("app__scripts", list)
|
||||
@@ -199,3 +202,10 @@ class DatasetConfig(BaseConfig):
|
||||
"CAUTION: due to the size of your dataset, "
|
||||
"running differential expression may take longer or fail."
|
||||
)
|
||||
|
||||
def handle_X_approx_distribution(self):
|
||||
self.validate_correct_type_of_configuration_attribute("X_approx_distribution", str)
|
||||
if self.X_approx_distribution not in ["normal", "count"]:
|
||||
raise ConfigurationError(
|
||||
"X_approx_distribution has unknown value -- must be 'normal' or 'count'."
|
||||
)
|
||||
|
||||
@@ -8,9 +8,9 @@ from scipy import sparse
|
||||
|
||||
import backend.common.compute.diffexp_generic as diffexp_generic
|
||||
from backend.common.colors import convert_anndata_category_colors_to_cxg_category_colors
|
||||
from backend.common.constants import Axis, MAX_LAYOUTS
|
||||
from backend.common.constants import Axis, MAX_LAYOUTS, XApproxDistribution
|
||||
from backend.czi_hosted.common.corpora import corpora_get_props_from_anndata
|
||||
from backend.common.errors import PrepareError, DatasetAccessError
|
||||
from backend.common.errors import PrepareError, DatasetAccessError, ConfigurationError
|
||||
from backend.common.utils.type_conversion_utils import get_schema_type_hint_of_array
|
||||
from backend.czi_hosted.data_common.data_adaptor import DataAdaptor
|
||||
from backend.common.fbs.matrix import encode_matrix_fbs
|
||||
@@ -28,6 +28,7 @@ class AnndataAdaptor(DataAdaptor):
|
||||
def __init__(self, data_locator, app_config=None, dataset_config=None):
|
||||
super().__init__(data_locator, app_config, dataset_config)
|
||||
self.data = None
|
||||
self.X_approx_distribution = None
|
||||
self._load_data(data_locator)
|
||||
self._validate_and_initialize()
|
||||
|
||||
@@ -190,6 +191,10 @@ class AnndataAdaptor(DataAdaptor):
|
||||
self.gene_count = self.data.shape[1]
|
||||
self._create_schema()
|
||||
|
||||
if self.dataset_config.X_approx_distribution == "auto":
|
||||
raise ConfigurationError("X-approx-distribution 'auto' mode unsupported.")
|
||||
self.X_approx_distribution = self.dataset_config.X_approx_distribution
|
||||
|
||||
# heuristic
|
||||
n_values = self.data.shape[0] * self.data.shape[1]
|
||||
if (n_values > 1e8 and self.server_config.adaptor__anndata_adaptor__backed is True) or (n_values > 5e8):
|
||||
@@ -309,13 +314,22 @@ class AnndataAdaptor(DataAdaptor):
|
||||
return convert_anndata_category_colors_to_cxg_category_colors(self.data)
|
||||
|
||||
def get_X_array(self, obs_mask=None, var_mask=None):
|
||||
# H5Py does not support boolean indexing (masks), so convert to integer indexing
|
||||
# when backed (ie, when AnnData is using H5Py indexing)
|
||||
if obs_mask is None:
|
||||
obs_mask = slice(None)
|
||||
elif self.data.isbacked and obs_mask.dtype == bool:
|
||||
obs_mask = obs_mask.nonzero()[0]
|
||||
if var_mask is None:
|
||||
var_mask = slice(None)
|
||||
elif self.data.isbacked and var_mask.dtype == bool:
|
||||
var_mask = var_mask.nonzero()[0]
|
||||
X = self.data.X[obs_mask, var_mask]
|
||||
return X
|
||||
|
||||
def get_X_approx_distribution(self) -> XApproxDistribution:
|
||||
return self.X_approx_distribution
|
||||
|
||||
def get_shape(self):
|
||||
return self.data.shape
|
||||
|
||||
|
||||
@@ -7,8 +7,14 @@ from scipy import sparse
|
||||
from server_timing import Timing as ServerTiming
|
||||
|
||||
from backend.czi_hosted.common.config.app_config import AppConfig
|
||||
from backend.common.constants import Axis
|
||||
from backend.common.errors import FilterError, JSONEncodingValueError, ExceedsLimitError, UnsupportedSummaryMethod, DatasetAccessError
|
||||
from backend.common.constants import Axis, XApproxDistribution
|
||||
from backend.common.errors import (
|
||||
FilterError,
|
||||
JSONEncodingValueError,
|
||||
ExceedsLimitError,
|
||||
UnsupportedSummaryMethod,
|
||||
DatasetAccessError,
|
||||
)
|
||||
from backend.common.utils.utils import jsonify_numpy
|
||||
from backend.common.fbs.matrix import encode_matrix_fbs
|
||||
|
||||
@@ -77,6 +83,11 @@ class DataAdaptor(metaclass=ABCMeta):
|
||||
the return type is either ndarray or scipy.sparse.spmatrix."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_X_approx_distribution(self) -> XApproxDistribution:
|
||||
"""return the approximate distribution of the X matrix."""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def get_shape(self):
|
||||
pass
|
||||
@@ -158,7 +169,7 @@ class DataAdaptor(metaclass=ABCMeta):
|
||||
mask = np.zeros((count,), dtype=np.bool)
|
||||
for i in filter:
|
||||
if type(i) == list:
|
||||
mask[i[0]: i[1]] = True
|
||||
mask[i[0] : i[1]] = True
|
||||
else:
|
||||
mask[i] = True
|
||||
return mask
|
||||
@@ -316,12 +327,13 @@ class DataAdaptor(metaclass=ABCMeta):
|
||||
top_n = self.dataset_config.diffexp__top_n
|
||||
|
||||
if self.server_config.exceeds_limit(
|
||||
"diffexp_cellcount_max", np.count_nonzero(obs_mask_A) + np.count_nonzero(obs_mask_B)
|
||||
"diffexp_cellcount_max", np.count_nonzero(obs_mask_A) + np.count_nonzero(obs_mask_B)
|
||||
):
|
||||
raise ExceedsLimitError("Diffexp request exceeds max cell count limit")
|
||||
|
||||
result = self.compute_diffexp_ttest(
|
||||
maskA=obs_mask_A, maskB=obs_mask_B, top_n=top_n, lfc_cutoff=self.dataset_config.diffexp__lfc_cutoff)
|
||||
maskA=obs_mask_A, maskB=obs_mask_B, top_n=top_n, lfc_cutoff=self.dataset_config.diffexp__lfc_cutoff
|
||||
)
|
||||
|
||||
try:
|
||||
return jsonify_numpy(result)
|
||||
|
||||
@@ -8,7 +8,7 @@ import pandas as pd
|
||||
import tiledb
|
||||
from server_timing import Timing as ServerTiming
|
||||
|
||||
from backend.common.constants import Axis
|
||||
from backend.common.constants import Axis, XApproxDistribution
|
||||
from backend.common.errors import DatasetAccessError, ConfigurationError
|
||||
from backend.czi_hosted.common.immutable_kvcache import ImmutableKVCache
|
||||
from backend.common.utils.type_conversion_utils import get_schema_type_hint_from_dtype
|
||||
@@ -37,6 +37,7 @@ class CxgAdaptor(DataAdaptor):
|
||||
self.lsuri_results = ImmutableKVCache(lambda key: self._lsuri(uri=key, tiledb_ctx=self.tiledb_ctx))
|
||||
self.arrays = ImmutableKVCache(lambda key: self._open_array(uri=key, tiledb_ctx=self.tiledb_ctx))
|
||||
self.schema = None
|
||||
self.X_approx_distribution = None
|
||||
|
||||
self._validate_and_initialize()
|
||||
|
||||
@@ -175,6 +176,10 @@ class CxgAdaptor(DataAdaptor):
|
||||
if cxg_version not in ["0.0", "0.1", "0.2.0"]:
|
||||
raise DatasetAccessError(f"cxg matrix is not valid: {self.url}")
|
||||
|
||||
if self.dataset_config.X_approx_distribution == "auto":
|
||||
raise ConfigurationError("X-approx-distribution 'auto' mode unsupported.")
|
||||
self.X_approx_distribution = self.dataset_config.X_approx_distribution
|
||||
|
||||
self.title = title
|
||||
self.about = about
|
||||
self.cxg_version = cxg_version
|
||||
@@ -281,6 +286,9 @@ class CxgAdaptor(DataAdaptor):
|
||||
data = X.multi_index[obs_items, var_items][""]
|
||||
return data
|
||||
|
||||
def get_X_approx_distribution(self) -> XApproxDistribution:
|
||||
return self.X_approx_distribution
|
||||
|
||||
def get_shape(self):
|
||||
X = self.open_array("X")
|
||||
return X.shape
|
||||
|
||||
@@ -203,6 +203,8 @@ dataset:
|
||||
lfc_cutoff: 0.01
|
||||
top_n: 10
|
||||
|
||||
X_approx_distribution: normal # currently fixed config
|
||||
|
||||
external:
|
||||
# You can retrieve configuration parameters from this config file, the environment,
|
||||
# the AWS secrets manager, or from the "cellxgene launch" command line arguments.
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
python-igraph
|
||||
louvain>=0.6
|
||||
scanpy==1.4.6 # Until we move to anndata 0.7.4 scanpy needs to be pinned here
|
||||
scanpy
|
||||
umap-learn<0.5.0 # The pinned version scanpy is not compatible with latest umap-learn
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
anndata>=0.7.0
|
||||
anndata>=0.7.6 # we use to_memory(), added in 0.7.6
|
||||
boto3>=1.12.18
|
||||
click>=7.1.2
|
||||
Flask>=1.0.2,<2.0.0 # Flask 2.0 is not compatible with the latest version of Flask-RESTful (0.3.8)
|
||||
@@ -11,7 +11,7 @@ flatbuffers>=1.11.0,<2.0.0 # cellxgene is not compatible with 2.0.0. Requires mi
|
||||
flatten-dict>=0.2.0
|
||||
fsspec>=0.4.4,<0.8.0
|
||||
gunicorn>=20.0.4
|
||||
h5py<3.0.0 # h5py>=3.0.0 had a breaking change; there is a fix in anndata>=0.7.5
|
||||
h5py>=3.0.0
|
||||
numba>=0.49.1,<0.53.0
|
||||
numpy>=1.15.0
|
||||
packaging>=20.0
|
||||
|
||||
Reference in New Issue
Block a user