anndata X indexing & version compatibility improvements (#1157)

* revert MatrixProxy; replace with correct use of adata slicing

* work around 0.6 adata slicing bug

* fix incorrect var slice

* simplify slicing of X

* add warning about performance impact of anndata<=0.7

* lint and remove unused code

* improve comment

* lint

* correctly parse versions

* temp files should preserve file suffix if possible - anndata 0.7 compat

* update anndata dependency to 0.6.20

* resolve PR review comments
This commit is contained in:
Bruce Martin
2020-02-19 09:57:51 -07:00
committed by GitHub
parent c630be33df
commit 349c413d8b
7 changed files with 47 additions and 675 deletions
-47
View File
@@ -1,47 +0,0 @@
from server.app.util.matrix_proxy import MatrixProxyView, ArrayProxyView
"""
AnnData/h5py are inconsistent in the API supported by various types of
X matrices. Sometimes you get a fully ndarray, sometims a Scipy sparse
matrix, sometimes h5py proxies with a subset of our needed interfaces.
This glue code paves over all of that, providing the core set of methods
that the cellxgene ScanPy driver assumes are in existance. Put another
way, all of the non-portable assumptions are here.
"""
class ArrayProxyView_anndata_h5py(ArrayProxyView):
"""
override to handle sparse getitem semantics, which differ
from numpy.
"""
def toarray(self):
""" sadly, sparse indexing doesn't drop dimensions like numpy! """
arr = self.m[self._index[0], self._index[1]]
if self._vdim == 0:
arr = arr.transpose()
return arr.toarray()[0]
class MatrixProxy_anndata_h5py(MatrixProxyView):
"""
AnnData sparse array stored in H5AD, or proxies for backed data.
None of these handle indexing very well, so we plop a proxy on top.
"""
@classmethod
def __supports__(cls):
return (
"anndata.h5py.h5sparse.SparseDataset",
"anndata.h5py.h5sparse.backed_csc_matrix",
"anndata.h5py.h5sparse.backed_csr_matrix",
"h5py._hl.dataset.Dataset",
"anndata._core.sparse_dataset.backed_csr_matrix",
"anndata._core.sparse_dataset.backed_csc_matrix",
)
@classmethod
def create_array(cls, *args, **kwargs):
return ArrayProxyView_anndata_h5py(*args, **kwargs)
+25 -5
View File
@@ -5,6 +5,7 @@ from datetime import datetime
import os.path
from hashlib import blake2b
import base64
from packaging import version
import numpy as np
import pandas
@@ -26,8 +27,15 @@ from server.app.util.utils import jsonify_scanpy, requires_data
from server.app.scanpy_engine.diffexp import diffexp_ttest
from server.app.util.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
from server.app.scanpy_engine.labels import read_labels, write_labels
import server.app.scanpy_engine.matrix_proxy # noqa: F401
from server.app.util.matrix_proxy import MatrixProxy
anndata_version = version.parse(str(anndata.__version__)).release
def anndata_version_is_pre_070():
major = anndata_version[0]
minor = anndata_version[1] if len(anndata_version) > 1 else 0
return major == 0 and minor < 7
def has_method(o, name):
@@ -303,6 +311,10 @@ class ScanpyEngine(CXGDriver):
@requires_data
def _validate_and_initialize(self):
if anndata_version_is_pre_070() and self.config['backed']:
warnings.warn(f"Use of --backed mode with anndata versions older than 0.7 will have serious "
"performance issues. Please update to at least anndata 0.7 or later.")
# var and obs column names must be unique
if not self.data.obs.columns.is_unique or not self.data.var.columns.is_unique:
raise KeyError(f"All annotation column names must be unique.")
@@ -370,7 +382,14 @@ class ScanpyEngine(CXGDriver):
@requires_data
def _validate_data_types(self):
if sparse.isspmatrix(self.data.X) and not sparse.isspmatrix_csc(self.data.X):
# The backed API does not support interrogation of the underlying sparsity or sparse matrix type
# Fake it by asking for a small subarray and testing it. NOTE: if the user has ignored our
# anndata <= 0.7 warning, opted for the --backed option, and specified a large, sparse dataset,
# this "small" indexing request will load the entire X array. This is due to a bug in anndata<=0.7
# which will load the entire X matrix to fullfill any slicing request if X is sparse. See
# user warning in _load_data().
X0 = self.data.X[0, 0:1]
if sparse.isspmatrix(X0) and not sparse.isspmatrix_csc(X0):
warnings.warn(
f"Scanpy data matrix is sparse, but not a CSC (columnar) matrix. "
f"Performance may be improved by using CSC."
@@ -577,8 +596,9 @@ class ScanpyEngine(CXGDriver):
raise FilterError("filtering on obs unsupported")
# Currently only handles VAR dimension
X = MatrixProxy.create(self.data.X if var_selector is None else self.data.X[:, var_selector])
return encode_matrix_fbs(X, col_idx=np.nonzero(var_selector)[0], row_idx=None)
X = self.data.X[:, slice(None) if var_selector is None else var_selector]
col_idx = np.nonzero([] if var_selector is None else var_selector)[0]
return encode_matrix_fbs(X, col_idx=col_idx, row_idx=None)
@requires_data
def diffexp_topN(self, obsFilterA, obsFilterB, top_n=None, interactive_limit=None):