Add support for anndata backed mode (#943)

* initial cut at backed mode

* make flask multithreading conditional on debug flag

* update X access to support backed mode

* lint

* improve help message for backed mode

* fix tests

* add MatrixProxy to normalize supported matrix types

* add FAQ entry for --backed

* remove use of matrix.T

* clean up

* add ability to disable diffexp from CLI; add hueristic to detect likely slow diffexp calculation, and warn user

* fix tests

* do not print diffexp speed warning if diffexp is disabled

* tweak wording of diffexp speed messages

* add FAQ entry on --disable-diffexp

* revise heuristic for warning about slow diffexp

* use quick tooltip delay on diffexp button
This commit is contained in:
Bruce Martin
2019-10-08 11:16:07 -07:00
committed by GitHub
parent 1467357db5
commit 711f3b7048
15 changed files with 881 additions and 99 deletions
+11 -12
View File
@@ -12,6 +12,7 @@ import server.app.util.fbs.NetEncoding.Uint32Array as Uint32Array
import server.app.util.fbs.NetEncoding.Float32Array as Float32Array
import server.app.util.fbs.NetEncoding.Float64Array as Float64Array
import server.app.util.fbs.NetEncoding.JSONEncodedArray as JSONEncodedArray
from server.app.util.matrix_proxy import MatrixProxy
# Placeholder until recent enhancements to flatbuffers Python
@@ -24,7 +25,7 @@ def CreateNumpyVector(builder, x):
"""CreateNumpyVector writes a numpy array into the buffer."""
if not isinstance(x, np.ndarray):
raise TypeError("non-numpy-ndarray passed to CreateNumpyVector")
raise TypeError(f"non-numpy-ndarray passed to CreateNumpyVector ({type(x)}")
if x.dtype.kind not in ['b', 'i', 'u', 'f']:
raise TypeError("numpy-ndarray holds elements of unsupported datatype")
@@ -91,7 +92,7 @@ def serialize_typed_array(builder, source_array, encoding_info):
as_json = arr.to_json(orient='records')
arr = np.array(bytearray(as_json, 'utf-8'))
else:
if sparse.issparse(arr):
if MatrixProxy.ismatrixproxy(arr) or sparse.issparse(arr):
arr = arr.toarray()
elif isinstance(arr, pd.Series):
arr = arr.get_values()
@@ -99,8 +100,11 @@ def serialize_typed_array(builder, source_array, encoding_info):
arr = arr.astype(as_type)
# serialize the ndarray into a vector
if arr.ndim == 2 and arr.shape[0] == 1:
arr = arr[0]
if arr.ndim == 2:
if arr.shape[0] == 1:
arr = arr[0]
elif arr.shape[1] == 1:
arr = arr.T[0]
vec = CreateNumpyVector(builder, arr)
# serialize the typed array table
@@ -185,16 +189,11 @@ def encode_matrix_fbs(matrix, row_idx=None, col_idx=None):
# estimate size needed, so we don't unnecessarily realloc.
builder = flatbuffers.Builder(guess_at_mem_needed(matrix))
if isinstance(matrix, pd.DataFrame):
matrix_columns = reversed(tuple(matrix[name] for name in matrix))
else:
matrix_columns = reversed(tuple(c for c in matrix.T))
columns = []
# for idx in reversed(np.arange(n_cols)):
for c in matrix_columns:
for cidx in range(n_cols - 1, -1, -1):
# serialize the typed array
typed_arr = serialize_typed_array(builder, c, column_encoding)
col = matrix.iloc[:, cidx] if isinstance(matrix, pd.DataFrame) else matrix[:, cidx]
typed_arr = serialize_typed_array(builder, col, column_encoding)
# serialize the Column union
columns.append(serialize_column(builder, typed_arr))