mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-03 22:48:13 +08:00
Add support for anndata backed mode (#943)
* initial cut at backed mode * make flask multithreading conditional on debug flag * update X access to support backed mode * lint * improve help message for backed mode * fix tests * add MatrixProxy to normalize supported matrix types * add FAQ entry for --backed * remove use of matrix.T * clean up * add ability to disable diffexp from CLI; add hueristic to detect likely slow diffexp calculation, and warn user * fix tests * do not print diffexp speed warning if diffexp is disabled * tweak wording of diffexp speed messages * add FAQ entry on --disable-diffexp * revise heuristic for warning about slow diffexp * use quick tooltip delay on diffexp button
This commit is contained in:
@@ -62,13 +62,12 @@ def diffexp_ttest(adata, maskA, maskB, top_n=8, diffexp_lfc_cutoff=0.01):
|
||||
:param diffexp_lfc_cutoff: minimum
|
||||
:return: for top N genes, [ varindex, logfoldchange, pval, pval_adj ]
|
||||
"""
|
||||
|
||||
if top_n > adata.n_obs:
|
||||
top_n = adata.n_obs
|
||||
|
||||
# mean, variance, N - calculate for both selections
|
||||
meanA, vA, nA = _mean_var_n(adata._X[maskA])
|
||||
meanB, vB, nB = _mean_var_n(adata._X[maskB])
|
||||
meanA, vA, nA = _mean_var_n(adata.X[maskA, :])
|
||||
meanB, vB, nB = _mean_var_n(adata.X[maskB, :])
|
||||
|
||||
# variance / N
|
||||
vnA = vA / min(nA, nB) # overestimate variance, would normally be nA
|
||||
@@ -87,7 +86,7 @@ def diffexp_ttest(adata, maskA, maskB, top_n=8, diffexp_lfc_cutoff=0.01):
|
||||
|
||||
# p-value
|
||||
pvals = stats.t.sf(np.abs(tscores), dof) * 2
|
||||
pvals_adj = pvals * adata._X.shape[1]
|
||||
pvals_adj = pvals * adata.X.shape[1]
|
||||
pvals_adj[pvals_adj > 1] = 1 # cap adjusted p-value at 1
|
||||
|
||||
# logfoldchanges: log2(meanA / meanB)
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
|
||||
from server.app.util.matrix_proxy import MatrixProxyView, ArrayProxyView
|
||||
|
||||
"""
|
||||
AnnData/h5py are inconsistent in the API supported by various types of
|
||||
X matrices. Sometimes you get a fully ndarray, sometims a Scipy sparse
|
||||
matrix, sometimes h5py proxies with a subset of our needed interfaces.
|
||||
|
||||
This glue code paves over all of that, providing the core set of methods
|
||||
that the cellxgene ScanPy driver assumes are in existance. Put another
|
||||
way, all of the non-portable assumptions are here.
|
||||
"""
|
||||
|
||||
|
||||
class ArrayProxyView_anndata_h5py(ArrayProxyView):
|
||||
"""
|
||||
override to handle sparse getitem semantics, which differ
|
||||
from numpy.
|
||||
"""
|
||||
def toarray(self):
|
||||
""" sadly, sparse indexing doesn't drop dimensions like numpy! """
|
||||
arr = self.m[self._index[0], self._index[1]]
|
||||
if self._vdim == 0:
|
||||
arr = arr.transpose()
|
||||
return arr.toarray()[0]
|
||||
|
||||
|
||||
class MatrixProxy_anndata_h5py(MatrixProxyView):
|
||||
"""
|
||||
AnnData sparse array stored in H5AD, or proxies for backed data.
|
||||
None of these handle indexing very well, so we plop a proxy on top.
|
||||
"""
|
||||
@classmethod
|
||||
def __supports__(cls):
|
||||
return ("anndata.h5py.h5sparse.SparseDataset",
|
||||
"anndata.h5py.h5sparse.backed_csc_matrix",
|
||||
"anndata.h5py.h5sparse.backed_csr_matrix",
|
||||
"h5py._hl.dataset.Dataset")
|
||||
|
||||
@classmethod
|
||||
def create_array(cls, *args, **kwargs):
|
||||
return ArrayProxyView_anndata_h5py(*args, **kwargs)
|
||||
@@ -21,6 +21,14 @@ from server.app.util.utils import jsonify_scanpy, requires_data
|
||||
from server.app.scanpy_engine.diffexp import diffexp_ttest
|
||||
from server.app.util.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
|
||||
from server.app.scanpy_engine.labels import read_labels, write_labels
|
||||
import server.app.scanpy_engine.matrix_proxy # noqa: F401
|
||||
from server.app.util.matrix_proxy import MatrixProxy
|
||||
|
||||
|
||||
def has_method(o, name):
|
||||
""" return True if `o` has callable method `name` """
|
||||
op = getattr(o, name, None)
|
||||
return op is not None and callable(op)
|
||||
|
||||
|
||||
class ScanpyEngine(CXGDriver):
|
||||
@@ -45,6 +53,9 @@ class ScanpyEngine(CXGDriver):
|
||||
"var_names": None,
|
||||
"diffexp_lfc_cutoff": 0.01,
|
||||
"label_file": None,
|
||||
"backed": False,
|
||||
"disable_diffexp": False,
|
||||
"diffexp_may_be_slow": False
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
@@ -208,7 +219,8 @@ class ScanpyEngine(CXGDriver):
|
||||
with data_locator.local_handle() as lh:
|
||||
# as of AnnData 0.6.19, backed mode performs initial load fast, but at the
|
||||
# cost of significantly slower access to X data.
|
||||
self.data = anndata.read_h5ad(lh)
|
||||
backed = 'r' if self.config['backed'] else None
|
||||
self.data = anndata.read_h5ad(lh, backed=backed)
|
||||
|
||||
except ValueError:
|
||||
raise ScanpyFileError(
|
||||
@@ -251,6 +263,11 @@ class ScanpyEngine(CXGDriver):
|
||||
self._validate_label_data()
|
||||
self._create_schema()
|
||||
|
||||
# heuristic
|
||||
n_values = self.data.shape[0] * self.data.shape[1]
|
||||
if (n_values > 1e8 and self.config['backed'] is True) or (n_values > 5e8):
|
||||
self.config.update({"diffexp_may_be_slow": True})
|
||||
|
||||
@requires_data
|
||||
def _default_and_validate_layouts(self):
|
||||
""" function:
|
||||
@@ -467,22 +484,6 @@ class ScanpyEngine(CXGDriver):
|
||||
|
||||
return jsonify_scanpy({"status": "OK"})
|
||||
|
||||
@staticmethod
|
||||
def slice_columns(X, var_mask):
|
||||
"""
|
||||
Slice columns from the matrix X, as specified by the mask
|
||||
Semantically equivalent to X[:, var_mask], but handles sparse
|
||||
matrices in a more performant manner.
|
||||
"""
|
||||
if var_mask is None: # noop
|
||||
return X
|
||||
if sparse.issparse(X): # use tuned getcol/hstack for performance
|
||||
indices = np.nonzero(var_mask)[0]
|
||||
cols = [X.getcol(i) for i in indices]
|
||||
return sparse.hstack(cols, format="csc")
|
||||
else: # else, just use standard slicing, which is fine for dense arrays
|
||||
return X[:, var_mask]
|
||||
|
||||
@requires_data
|
||||
def data_frame_to_fbs_matrix(self, filter, axis):
|
||||
"""
|
||||
@@ -505,7 +506,8 @@ class ScanpyEngine(CXGDriver):
|
||||
raise FilterError("filtering on obs unsupported")
|
||||
|
||||
# Currently only handles VAR dimension
|
||||
X = self.slice_columns(self.data._X, var_selector)
|
||||
X = MatrixProxy.create(self.data.X if var_selector is None
|
||||
else self.data.X[:, var_selector])
|
||||
return encode_matrix_fbs(X, col_idx=np.nonzero(var_selector)[0], row_idx=None)
|
||||
|
||||
@requires_data
|
||||
|
||||
Reference in New Issue
Block a user