mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-29 09:58:11 +08:00
fix for incorrect stats computation in diff exp t-test (#2318)
* 2211 fixes * lint * lint * add missing test and bug found by test * change terminology for count distribution * update scanpy requirement * update scanpy requirement
This commit is contained in:
@@ -150,6 +150,14 @@ def dataset_args(func):
|
||||
metavar="<URL>",
|
||||
help="URL providing more information about the dataset (hint: must be a fully specified absolute URL).",
|
||||
)
|
||||
@click.option(
|
||||
"--X-approx-distribution",
|
||||
default=DEFAULT_CONFIG.dataset_config.X_approx_distribution,
|
||||
show_default=True,
|
||||
type=click.Choice(["auto", "normal", "count"], case_sensitive=False),
|
||||
help="Specify the approximate distribution of X matrix values. 'auto' will use a heuristic "
|
||||
"to determine the approximate distribution. Mode 'auto' is incompatible with --backed.",
|
||||
)
|
||||
@functools.wraps(func)
|
||||
def wrapper(*args, **kwargs):
|
||||
return func(*args, **kwargs)
|
||||
@@ -318,6 +326,7 @@ def launch(
|
||||
disable_diffexp,
|
||||
config_file,
|
||||
dump_default_config,
|
||||
x_approx_distribution,
|
||||
):
|
||||
"""Launch the cellxgene data viewer.
|
||||
This web app lets you explore single-cell expression data.
|
||||
@@ -376,6 +385,7 @@ def launch(
|
||||
embeddings__names=embedding,
|
||||
diffexp__enable=not disable_diffexp,
|
||||
diffexp__lfc_cutoff=diffexp_lfc_cutoff,
|
||||
X_approx_distribution=x_approx_distribution,
|
||||
)
|
||||
|
||||
diff = cli_config.server_config.changes_from_default()
|
||||
|
||||
@@ -38,6 +38,8 @@ class DatasetConfig(BaseConfig):
|
||||
self.diffexp__lfc_cutoff = default_config["diffexp"]["lfc_cutoff"]
|
||||
self.diffexp__top_n = default_config["diffexp"]["top_n"]
|
||||
|
||||
self.X_approx_distribution = default_config["X_approx_distribution"]
|
||||
|
||||
except KeyError as e:
|
||||
raise ConfigurationError(f"Unexpected config: {str(e)}")
|
||||
|
||||
@@ -50,6 +52,7 @@ class DatasetConfig(BaseConfig):
|
||||
self.handle_user_annotations(context)
|
||||
self.handle_embeddings()
|
||||
self.handle_diffexp(context)
|
||||
self.handle_X_approx_distribution()
|
||||
|
||||
def get_data_adaptor(self):
|
||||
server_config = self.app_config.server_config
|
||||
@@ -182,3 +185,10 @@ class DatasetConfig(BaseConfig):
|
||||
context["messagefn"](
|
||||
"CAUTION: due to the size of your dataset, " "running differential expression may take longer or fail."
|
||||
)
|
||||
|
||||
def handle_X_approx_distribution(self):
|
||||
self.validate_correct_type_of_configuration_attribute("X_approx_distribution", str)
|
||||
if self.X_approx_distribution not in ["auto", "normal", "count"]:
|
||||
raise ConfigurationError(
|
||||
"X_approx_distribution has unknown value -- must be 'auto', 'normal' or 'count'."
|
||||
)
|
||||
|
||||
@@ -7,8 +7,9 @@ from pandas.core.dtypes.dtypes import CategoricalDtype
|
||||
from scipy import sparse
|
||||
|
||||
import backend.common.compute.diffexp_generic as diffexp_generic
|
||||
import backend.common.compute.estimate_distribution as estimate_distribution
|
||||
from backend.common.colors import convert_anndata_category_colors_to_cxg_category_colors
|
||||
from backend.common.constants import Axis, MAX_LAYOUTS
|
||||
from backend.common.constants import Axis, MAX_LAYOUTS, XApproxDistribution
|
||||
from backend.server.common.corpora import corpora_get_props_from_anndata
|
||||
from backend.common.errors import PrepareError, DatasetAccessError
|
||||
from backend.common.utils.type_conversion_utils import get_schema_type_hint_of_array
|
||||
@@ -28,6 +29,7 @@ class AnndataAdaptor(DataAdaptor):
|
||||
def __init__(self, data_locator, app_config=None, dataset_config=None):
|
||||
super().__init__(data_locator, app_config, dataset_config)
|
||||
self.data = None
|
||||
self.X_approx_distribution = None
|
||||
self._load_data(data_locator)
|
||||
self._validate_and_initialize()
|
||||
|
||||
@@ -190,6 +192,13 @@ class AnndataAdaptor(DataAdaptor):
|
||||
self.gene_count = self.data.shape[1]
|
||||
self._create_schema()
|
||||
|
||||
if self.dataset_config.X_approx_distribution == "auto":
|
||||
"""Lazy evaluate the heuristic if we are backed."""
|
||||
if not self.data.isbacked:
|
||||
self.X_approx_distribution = estimate_distribution.estimate_approximate_distribution(self.data.X)
|
||||
else:
|
||||
self.X_approx_distribution = self.dataset_config.X_approx_distribution
|
||||
|
||||
# heuristic
|
||||
n_values = self.data.shape[0] * self.data.shape[1]
|
||||
if (n_values > 1e8 and self.server_config.adaptor__anndata_adaptor__backed is True) or (n_values > 5e8):
|
||||
@@ -309,13 +318,29 @@ class AnndataAdaptor(DataAdaptor):
|
||||
return convert_anndata_category_colors_to_cxg_category_colors(self.data)
|
||||
|
||||
def get_X_array(self, obs_mask=None, var_mask=None):
|
||||
# H5Py does not support boolean indexing (masks), so convert to integer indexing
|
||||
# when backed (ie, when AnnData is using H5Py indexing)
|
||||
if obs_mask is None:
|
||||
obs_mask = slice(None)
|
||||
elif self.data.isbacked and obs_mask.dtype == bool:
|
||||
obs_mask = obs_mask.nonzero()[0]
|
||||
if var_mask is None:
|
||||
var_mask = slice(None)
|
||||
elif self.data.isbacked and var_mask.dtype == bool:
|
||||
var_mask = var_mask.nonzero()[0]
|
||||
X = self.data.X[obs_mask, var_mask]
|
||||
return X
|
||||
|
||||
def get_X_approx_distribution(self) -> XApproxDistribution:
|
||||
"""return the approximate distribution of the X matrix."""
|
||||
if self.X_approx_distribution is None:
|
||||
"""Not yet evaluated."""
|
||||
assert(self.dataset_config.X_approx_distribution == "auto")
|
||||
self.data = self.data.to_memory() # loads data
|
||||
self.X_approx_distribution = estimate_distribution.estimate_approximate_distribution(self.data.X)
|
||||
|
||||
return self.X_approx_distribution
|
||||
|
||||
def get_shape(self):
|
||||
return self.data.shape
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@ from scipy import sparse
|
||||
from server_timing import Timing as ServerTiming
|
||||
|
||||
from backend.server.common.config.app_config import AppConfig
|
||||
from backend.common.constants import Axis
|
||||
from backend.common.constants import Axis, XApproxDistribution
|
||||
from backend.common.errors import FilterError, JSONEncodingValueError, ExceedsLimitError, UnsupportedSummaryMethod
|
||||
from backend.common.utils.utils import jsonify_numpy
|
||||
from backend.common.fbs.matrix import encode_matrix_fbs
|
||||
@@ -72,6 +72,10 @@ class DataAdaptor(metaclass=ABCMeta):
|
||||
the return type is either ndarray or scipy.sparse.spmatrix."""
|
||||
pass
|
||||
|
||||
def get_X_approx_distribution(self) -> XApproxDistribution:
|
||||
"""return the approximate distribution of the X matrix."""
|
||||
return XApproxDistribution.NORMAL
|
||||
|
||||
@abstractmethod
|
||||
def get_shape(self):
|
||||
pass
|
||||
|
||||
@@ -77,6 +77,8 @@ dataset:
|
||||
lfc_cutoff: 0.01
|
||||
top_n: 10
|
||||
|
||||
X_approx_distribution: auto
|
||||
|
||||
external:
|
||||
# You can retrieve configuration parameters from this config file, the environment,
|
||||
# the AWS secrets manager, or from the "cellxgene launch" command line arguments.
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
python-igraph>=0.8
|
||||
louvain>=0.6
|
||||
scanpy==1.4.6 # Until we move to anndata 0.7.4 scanpy needs to be pinned here
|
||||
scanpy
|
||||
umap-learn<0.5.0 # The pinned version scanpy is not compatible with latest umap-learn
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
anndata>=0.7.0
|
||||
anndata>=0.7.6 # we need to_memory(), added in 0.7.6
|
||||
boto3>=1.12.18
|
||||
click>=7.1.2
|
||||
Flask>=1.0.2,<2.0.0 # Flask 2.0 is not compatible with the latest version of Flask-RESTful (0.3.8)
|
||||
@@ -11,9 +11,9 @@ flatbuffers>=1.11.0,<2.0.0 # cellxgene is not compatible with 2.0.0. Requires mi
|
||||
flatten-dict>=0.2.0
|
||||
fsspec>=0.4.4,<0.8.0
|
||||
gunicorn>=20.0.4
|
||||
h5py<3.0.0 # h5py>=3.0.0 had a breaking change; there is a fix in anndata>=0.7.5
|
||||
h5py>=3.0.0
|
||||
jinja2>=2.11.3 # Flask sub-dependency. Added due to CVE-2020-28493
|
||||
numba>=0.51.2,<0.53.0
|
||||
numba>=0.51.2
|
||||
numpy>=1.17.5
|
||||
packaging>=20.0
|
||||
pandas>=1.0,!=1.1 # pandas 1.1 breaks tests, https://github.com/pandas-dev/pandas/issues/35446
|
||||
|
||||
Reference in New Issue
Block a user