mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-03 08:18:13 +08:00
anndata X indexing & version compatibility improvements (#1157)
* revert MatrixProxy; replace with correct use of adata slicing * work around 0.6 adata slicing bug * fix incorrect var slice * simplify slicing of X * add warning about performance impact of anndata<=0.7 * lint and remove unused code * improve comment * lint * correctly parse versions * temp files should preserve file suffix if possible - anndata 0.7 compat * update anndata dependency to 0.6.20 * resolve PR review comments
This commit is contained in:
@@ -1,47 +0,0 @@
|
||||
from server.app.util.matrix_proxy import MatrixProxyView, ArrayProxyView
|
||||
|
||||
"""
|
||||
AnnData/h5py are inconsistent in the API supported by various types of
|
||||
X matrices. Sometimes you get a fully ndarray, sometims a Scipy sparse
|
||||
matrix, sometimes h5py proxies with a subset of our needed interfaces.
|
||||
|
||||
This glue code paves over all of that, providing the core set of methods
|
||||
that the cellxgene ScanPy driver assumes are in existance. Put another
|
||||
way, all of the non-portable assumptions are here.
|
||||
"""
|
||||
|
||||
|
||||
class ArrayProxyView_anndata_h5py(ArrayProxyView):
|
||||
"""
|
||||
override to handle sparse getitem semantics, which differ
|
||||
from numpy.
|
||||
"""
|
||||
|
||||
def toarray(self):
|
||||
""" sadly, sparse indexing doesn't drop dimensions like numpy! """
|
||||
arr = self.m[self._index[0], self._index[1]]
|
||||
if self._vdim == 0:
|
||||
arr = arr.transpose()
|
||||
return arr.toarray()[0]
|
||||
|
||||
|
||||
class MatrixProxy_anndata_h5py(MatrixProxyView):
|
||||
"""
|
||||
AnnData sparse array stored in H5AD, or proxies for backed data.
|
||||
None of these handle indexing very well, so we plop a proxy on top.
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
def __supports__(cls):
|
||||
return (
|
||||
"anndata.h5py.h5sparse.SparseDataset",
|
||||
"anndata.h5py.h5sparse.backed_csc_matrix",
|
||||
"anndata.h5py.h5sparse.backed_csr_matrix",
|
||||
"h5py._hl.dataset.Dataset",
|
||||
"anndata._core.sparse_dataset.backed_csr_matrix",
|
||||
"anndata._core.sparse_dataset.backed_csc_matrix",
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def create_array(cls, *args, **kwargs):
|
||||
return ArrayProxyView_anndata_h5py(*args, **kwargs)
|
||||
@@ -5,6 +5,7 @@ from datetime import datetime
|
||||
import os.path
|
||||
from hashlib import blake2b
|
||||
import base64
|
||||
from packaging import version
|
||||
|
||||
import numpy as np
|
||||
import pandas
|
||||
@@ -26,8 +27,15 @@ from server.app.util.utils import jsonify_scanpy, requires_data
|
||||
from server.app.scanpy_engine.diffexp import diffexp_ttest
|
||||
from server.app.util.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
|
||||
from server.app.scanpy_engine.labels import read_labels, write_labels
|
||||
import server.app.scanpy_engine.matrix_proxy # noqa: F401
|
||||
from server.app.util.matrix_proxy import MatrixProxy
|
||||
|
||||
|
||||
anndata_version = version.parse(str(anndata.__version__)).release
|
||||
|
||||
|
||||
def anndata_version_is_pre_070():
|
||||
major = anndata_version[0]
|
||||
minor = anndata_version[1] if len(anndata_version) > 1 else 0
|
||||
return major == 0 and minor < 7
|
||||
|
||||
|
||||
def has_method(o, name):
|
||||
@@ -303,6 +311,10 @@ class ScanpyEngine(CXGDriver):
|
||||
|
||||
@requires_data
|
||||
def _validate_and_initialize(self):
|
||||
if anndata_version_is_pre_070() and self.config['backed']:
|
||||
warnings.warn(f"Use of --backed mode with anndata versions older than 0.7 will have serious "
|
||||
"performance issues. Please update to at least anndata 0.7 or later.")
|
||||
|
||||
# var and obs column names must be unique
|
||||
if not self.data.obs.columns.is_unique or not self.data.var.columns.is_unique:
|
||||
raise KeyError(f"All annotation column names must be unique.")
|
||||
@@ -370,7 +382,14 @@ class ScanpyEngine(CXGDriver):
|
||||
|
||||
@requires_data
|
||||
def _validate_data_types(self):
|
||||
if sparse.isspmatrix(self.data.X) and not sparse.isspmatrix_csc(self.data.X):
|
||||
# The backed API does not support interrogation of the underlying sparsity or sparse matrix type
|
||||
# Fake it by asking for a small subarray and testing it. NOTE: if the user has ignored our
|
||||
# anndata <= 0.7 warning, opted for the --backed option, and specified a large, sparse dataset,
|
||||
# this "small" indexing request will load the entire X array. This is due to a bug in anndata<=0.7
|
||||
# which will load the entire X matrix to fullfill any slicing request if X is sparse. See
|
||||
# user warning in _load_data().
|
||||
X0 = self.data.X[0, 0:1]
|
||||
if sparse.isspmatrix(X0) and not sparse.isspmatrix_csc(X0):
|
||||
warnings.warn(
|
||||
f"Scanpy data matrix is sparse, but not a CSC (columnar) matrix. "
|
||||
f"Performance may be improved by using CSC."
|
||||
@@ -577,8 +596,9 @@ class ScanpyEngine(CXGDriver):
|
||||
raise FilterError("filtering on obs unsupported")
|
||||
|
||||
# Currently only handles VAR dimension
|
||||
X = MatrixProxy.create(self.data.X if var_selector is None else self.data.X[:, var_selector])
|
||||
return encode_matrix_fbs(X, col_idx=np.nonzero(var_selector)[0], row_idx=None)
|
||||
X = self.data.X[:, slice(None) if var_selector is None else var_selector]
|
||||
col_idx = np.nonzero([] if var_selector is None else var_selector)[0]
|
||||
return encode_matrix_fbs(X, col_idx=col_idx, row_idx=None)
|
||||
|
||||
@requires_data
|
||||
def diffexp_topN(self, obsFilterA, obsFilterB, top_n=None, interactive_limit=None):
|
||||
|
||||
Reference in New Issue
Block a user