Add support for anndata backed mode (#943)

* initial cut at backed mode

* make flask multithreading conditional on debug flag

* update X access to support backed mode

* lint

* improve help message for backed mode

* fix tests

* add MatrixProxy to normalize supported matrix types

* add FAQ entry for --backed

* remove use of matrix.T

* clean up

* add ability to disable diffexp from CLI; add hueristic to detect likely slow diffexp calculation, and warn user

* fix tests

* do not print diffexp speed warning if diffexp is disabled

* tweak wording of diffexp speed messages

* add FAQ entry on --disable-diffexp

* revise heuristic for warning about slow diffexp

* use quick tooltip delay on diffexp button
This commit is contained in:
Bruce Martin
2019-10-08 11:16:07 -07:00
committed by GitHub
parent 1467357db5
commit 711f3b7048
15 changed files with 881 additions and 99 deletions
+3 -4
View File
@@ -62,13 +62,12 @@ def diffexp_ttest(adata, maskA, maskB, top_n=8, diffexp_lfc_cutoff=0.01):
:param diffexp_lfc_cutoff: minimum
:return: for top N genes, [ varindex, logfoldchange, pval, pval_adj ]
"""
if top_n > adata.n_obs:
top_n = adata.n_obs
# mean, variance, N - calculate for both selections
meanA, vA, nA = _mean_var_n(adata._X[maskA])
meanB, vB, nB = _mean_var_n(adata._X[maskB])
meanA, vA, nA = _mean_var_n(adata.X[maskA, :])
meanB, vB, nB = _mean_var_n(adata.X[maskB, :])
# variance / N
vnA = vA / min(nA, nB) # overestimate variance, would normally be nA
@@ -87,7 +86,7 @@ def diffexp_ttest(adata, maskA, maskB, top_n=8, diffexp_lfc_cutoff=0.01):
# p-value
pvals = stats.t.sf(np.abs(tscores), dof) * 2
pvals_adj = pvals * adata._X.shape[1]
pvals_adj = pvals * adata.X.shape[1]
pvals_adj[pvals_adj > 1] = 1 # cap adjusted p-value at 1
# logfoldchanges: log2(meanA / meanB)
+42
View File
@@ -0,0 +1,42 @@
from server.app.util.matrix_proxy import MatrixProxyView, ArrayProxyView
"""
AnnData/h5py are inconsistent in the API supported by various types of
X matrices. Sometimes you get a fully ndarray, sometims a Scipy sparse
matrix, sometimes h5py proxies with a subset of our needed interfaces.
This glue code paves over all of that, providing the core set of methods
that the cellxgene ScanPy driver assumes are in existance. Put another
way, all of the non-portable assumptions are here.
"""
class ArrayProxyView_anndata_h5py(ArrayProxyView):
"""
override to handle sparse getitem semantics, which differ
from numpy.
"""
def toarray(self):
""" sadly, sparse indexing doesn't drop dimensions like numpy! """
arr = self.m[self._index[0], self._index[1]]
if self._vdim == 0:
arr = arr.transpose()
return arr.toarray()[0]
class MatrixProxy_anndata_h5py(MatrixProxyView):
"""
AnnData sparse array stored in H5AD, or proxies for backed data.
None of these handle indexing very well, so we plop a proxy on top.
"""
@classmethod
def __supports__(cls):
return ("anndata.h5py.h5sparse.SparseDataset",
"anndata.h5py.h5sparse.backed_csc_matrix",
"anndata.h5py.h5sparse.backed_csr_matrix",
"h5py._hl.dataset.Dataset")
@classmethod
def create_array(cls, *args, **kwargs):
return ArrayProxyView_anndata_h5py(*args, **kwargs)
+20 -18
View File
@@ -21,6 +21,14 @@ from server.app.util.utils import jsonify_scanpy, requires_data
from server.app.scanpy_engine.diffexp import diffexp_ttest
from server.app.util.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
from server.app.scanpy_engine.labels import read_labels, write_labels
import server.app.scanpy_engine.matrix_proxy # noqa: F401
from server.app.util.matrix_proxy import MatrixProxy
def has_method(o, name):
""" return True if `o` has callable method `name` """
op = getattr(o, name, None)
return op is not None and callable(op)
class ScanpyEngine(CXGDriver):
@@ -45,6 +53,9 @@ class ScanpyEngine(CXGDriver):
"var_names": None,
"diffexp_lfc_cutoff": 0.01,
"label_file": None,
"backed": False,
"disable_diffexp": False,
"diffexp_may_be_slow": False
}
@staticmethod
@@ -208,7 +219,8 @@ class ScanpyEngine(CXGDriver):
with data_locator.local_handle() as lh:
# as of AnnData 0.6.19, backed mode performs initial load fast, but at the
# cost of significantly slower access to X data.
self.data = anndata.read_h5ad(lh)
backed = 'r' if self.config['backed'] else None
self.data = anndata.read_h5ad(lh, backed=backed)
except ValueError:
raise ScanpyFileError(
@@ -251,6 +263,11 @@ class ScanpyEngine(CXGDriver):
self._validate_label_data()
self._create_schema()
# heuristic
n_values = self.data.shape[0] * self.data.shape[1]
if (n_values > 1e8 and self.config['backed'] is True) or (n_values > 5e8):
self.config.update({"diffexp_may_be_slow": True})
@requires_data
def _default_and_validate_layouts(self):
""" function:
@@ -467,22 +484,6 @@ class ScanpyEngine(CXGDriver):
return jsonify_scanpy({"status": "OK"})
@staticmethod
def slice_columns(X, var_mask):
"""
Slice columns from the matrix X, as specified by the mask
Semantically equivalent to X[:, var_mask], but handles sparse
matrices in a more performant manner.
"""
if var_mask is None: # noop
return X
if sparse.issparse(X): # use tuned getcol/hstack for performance
indices = np.nonzero(var_mask)[0]
cols = [X.getcol(i) for i in indices]
return sparse.hstack(cols, format="csc")
else: # else, just use standard slicing, which is fine for dense arrays
return X[:, var_mask]
@requires_data
def data_frame_to_fbs_matrix(self, filter, axis):
"""
@@ -505,7 +506,8 @@ class ScanpyEngine(CXGDriver):
raise FilterError("filtering on obs unsupported")
# Currently only handles VAR dimension
X = self.slice_columns(self.data._X, var_selector)
X = MatrixProxy.create(self.data.X if var_selector is None
else self.data.X[:, var_selector])
return encode_matrix_fbs(X, col_idx=np.nonzero(var_selector)[0], row_idx=None)
@requires_data