fix for incorrect stats computation in diff exp t-test (#2318)

* 2211 fixes

* lint

* lint

* add missing test and bug found by test

* change terminology for count distribution

* update scanpy requirement

* update scanpy requirement
This commit is contained in:
Bruce Martin
2021-07-23 11:36:26 -07:00
committed by GitHub
parent 1ebde2213d
commit 1ea2b7fe80
28 changed files with 336 additions and 90 deletions
+10
View File
@@ -150,6 +150,14 @@ def dataset_args(func):
metavar="<URL>",
help="URL providing more information about the dataset (hint: must be a fully specified absolute URL).",
)
@click.option(
"--X-approx-distribution",
default=DEFAULT_CONFIG.dataset_config.X_approx_distribution,
show_default=True,
type=click.Choice(["auto", "normal", "count"], case_sensitive=False),
help="Specify the approximate distribution of X matrix values. 'auto' will use a heuristic "
"to determine the approximate distribution. Mode 'auto' is incompatible with --backed.",
)
@functools.wraps(func)
def wrapper(*args, **kwargs):
return func(*args, **kwargs)
@@ -318,6 +326,7 @@ def launch(
disable_diffexp,
config_file,
dump_default_config,
x_approx_distribution,
):
"""Launch the cellxgene data viewer.
This web app lets you explore single-cell expression data.
@@ -376,6 +385,7 @@ def launch(
embeddings__names=embedding,
diffexp__enable=not disable_diffexp,
diffexp__lfc_cutoff=diffexp_lfc_cutoff,
X_approx_distribution=x_approx_distribution,
)
diff = cli_config.server_config.changes_from_default()
@@ -38,6 +38,8 @@ class DatasetConfig(BaseConfig):
self.diffexp__lfc_cutoff = default_config["diffexp"]["lfc_cutoff"]
self.diffexp__top_n = default_config["diffexp"]["top_n"]
self.X_approx_distribution = default_config["X_approx_distribution"]
except KeyError as e:
raise ConfigurationError(f"Unexpected config: {str(e)}")
@@ -50,6 +52,7 @@ class DatasetConfig(BaseConfig):
self.handle_user_annotations(context)
self.handle_embeddings()
self.handle_diffexp(context)
self.handle_X_approx_distribution()
def get_data_adaptor(self):
server_config = self.app_config.server_config
@@ -182,3 +185,10 @@ class DatasetConfig(BaseConfig):
context["messagefn"](
"CAUTION: due to the size of your dataset, " "running differential expression may take longer or fail."
)
def handle_X_approx_distribution(self):
self.validate_correct_type_of_configuration_attribute("X_approx_distribution", str)
if self.X_approx_distribution not in ["auto", "normal", "count"]:
raise ConfigurationError(
"X_approx_distribution has unknown value -- must be 'auto', 'normal' or 'count'."
)
+26 -1
View File
@@ -7,8 +7,9 @@ from pandas.core.dtypes.dtypes import CategoricalDtype
from scipy import sparse
import backend.common.compute.diffexp_generic as diffexp_generic
import backend.common.compute.estimate_distribution as estimate_distribution
from backend.common.colors import convert_anndata_category_colors_to_cxg_category_colors
from backend.common.constants import Axis, MAX_LAYOUTS
from backend.common.constants import Axis, MAX_LAYOUTS, XApproxDistribution
from backend.server.common.corpora import corpora_get_props_from_anndata
from backend.common.errors import PrepareError, DatasetAccessError
from backend.common.utils.type_conversion_utils import get_schema_type_hint_of_array
@@ -28,6 +29,7 @@ class AnndataAdaptor(DataAdaptor):
def __init__(self, data_locator, app_config=None, dataset_config=None):
super().__init__(data_locator, app_config, dataset_config)
self.data = None
self.X_approx_distribution = None
self._load_data(data_locator)
self._validate_and_initialize()
@@ -190,6 +192,13 @@ class AnndataAdaptor(DataAdaptor):
self.gene_count = self.data.shape[1]
self._create_schema()
if self.dataset_config.X_approx_distribution == "auto":
"""Lazy evaluate the heuristic if we are backed."""
if not self.data.isbacked:
self.X_approx_distribution = estimate_distribution.estimate_approximate_distribution(self.data.X)
else:
self.X_approx_distribution = self.dataset_config.X_approx_distribution
# heuristic
n_values = self.data.shape[0] * self.data.shape[1]
if (n_values > 1e8 and self.server_config.adaptor__anndata_adaptor__backed is True) or (n_values > 5e8):
@@ -309,13 +318,29 @@ class AnndataAdaptor(DataAdaptor):
return convert_anndata_category_colors_to_cxg_category_colors(self.data)
def get_X_array(self, obs_mask=None, var_mask=None):
# H5Py does not support boolean indexing (masks), so convert to integer indexing
# when backed (ie, when AnnData is using H5Py indexing)
if obs_mask is None:
obs_mask = slice(None)
elif self.data.isbacked and obs_mask.dtype == bool:
obs_mask = obs_mask.nonzero()[0]
if var_mask is None:
var_mask = slice(None)
elif self.data.isbacked and var_mask.dtype == bool:
var_mask = var_mask.nonzero()[0]
X = self.data.X[obs_mask, var_mask]
return X
def get_X_approx_distribution(self) -> XApproxDistribution:
"""return the approximate distribution of the X matrix."""
if self.X_approx_distribution is None:
"""Not yet evaluated."""
assert(self.dataset_config.X_approx_distribution == "auto")
self.data = self.data.to_memory() # loads data
self.X_approx_distribution = estimate_distribution.estimate_approximate_distribution(self.data.X)
return self.X_approx_distribution
def get_shape(self):
return self.data.shape
+5 -1
View File
@@ -6,7 +6,7 @@ from scipy import sparse
from server_timing import Timing as ServerTiming
from backend.server.common.config.app_config import AppConfig
from backend.common.constants import Axis
from backend.common.constants import Axis, XApproxDistribution
from backend.common.errors import FilterError, JSONEncodingValueError, ExceedsLimitError, UnsupportedSummaryMethod
from backend.common.utils.utils import jsonify_numpy
from backend.common.fbs.matrix import encode_matrix_fbs
@@ -72,6 +72,10 @@ class DataAdaptor(metaclass=ABCMeta):
the return type is either ndarray or scipy.sparse.spmatrix."""
pass
def get_X_approx_distribution(self) -> XApproxDistribution:
"""return the approximate distribution of the X matrix."""
return XApproxDistribution.NORMAL
@abstractmethod
def get_shape(self):
pass
+2
View File
@@ -77,6 +77,8 @@ dataset:
lfc_cutoff: 0.01
top_n: 10
X_approx_distribution: auto
external:
# You can retrieve configuration parameters from this config file, the environment,
# the AWS secrets manager, or from the "cellxgene launch" command line arguments.
+1 -1
View File
@@ -1,4 +1,4 @@
python-igraph>=0.8
louvain>=0.6
scanpy==1.4.6 # Until we move to anndata 0.7.4 scanpy needs to be pinned here
scanpy
umap-learn<0.5.0 # The pinned version scanpy is not compatible with latest umap-learn
+3 -3
View File
@@ -1,4 +1,4 @@
anndata>=0.7.0
anndata>=0.7.6 # we need to_memory(), added in 0.7.6
boto3>=1.12.18
click>=7.1.2
Flask>=1.0.2,<2.0.0 # Flask 2.0 is not compatible with the latest version of Flask-RESTful (0.3.8)
@@ -11,9 +11,9 @@ flatbuffers>=1.11.0,<2.0.0 # cellxgene is not compatible with 2.0.0. Requires mi
flatten-dict>=0.2.0
fsspec>=0.4.4,<0.8.0
gunicorn>=20.0.4
h5py<3.0.0 # h5py>=3.0.0 had a breaking change; there is a fix in anndata>=0.7.5
h5py>=3.0.0
jinja2>=2.11.3 # Flask sub-dependency. Added due to CVE-2020-28493
numba>=0.51.2,<0.53.0
numba>=0.51.2
numpy>=1.17.5
packaging>=20.0
pandas>=1.0,!=1.1 # pandas 1.1 breaks tests, https://github.com/pandas-dev/pandas/issues/35446