mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-29 16:28:12 +08:00
Makefile modularity, test targets, and auto-formatting (#1070)
* Fix Makefile whitespace and .PHONY use
* Fix Makefile filename
* Modularize Makefile into client and server Makefiles
Part of the reason that the Makefile in the root directory is a bit
complicated is that it tries to handle tasks that can be handled
separately in the client and server modules.
This commit pushes some of the make logic specific to each module into
their own makefiles and calls out to those makefiles from that in the
project root.
* Add auto-formatting to client and server modules
One thing that can make linting faster is auto-formatting. This commit
adds the yapf auto-formatting tool to the server module and uses
eslint's "fix" functionality to speed up the linting/formatting process.
* Add yapf for automatic code formatting
* Add a root test target that calls sub-tests
* Apply yapf to python files
* Do not duplicate npm commands, simply pass through
* Update documentation
* Do not shadow reserved word len
* Add general test target
* Fix make call in dev-env
* Use black instead of yapf
* Run flake8 from the root directory
* Revert "Apply yapf to python files"
This reverts commit cdca128a01.
* Apply black to python code
* Resolve lint errors resulting from black format
* Add explanation of server unit tests in dev guidelines
This commit is contained in:
@@ -27,9 +27,7 @@ class DiffExpMode(AugmentedEnum):
|
||||
VAR_FILTER = "varFilter"
|
||||
|
||||
|
||||
JSON_NaN_to_num_warning_msg = (
|
||||
"JSON encoding failure - please verify all data are finite values (no NaN or Infinities)"
|
||||
)
|
||||
JSON_NaN_to_num_warning_msg = "JSON encoding failure - please verify all data are finite values (no NaN or Infinities)"
|
||||
REACTIVE_LIMIT = 1_000_000
|
||||
|
||||
MAX_LAYOUTS = 30
|
||||
|
||||
@@ -4,7 +4,7 @@ import fsspec
|
||||
from datetime import datetime
|
||||
|
||||
|
||||
class DataLocator():
|
||||
class DataLocator:
|
||||
"""
|
||||
DataLocator is a simple wrapper around fsspec functionality, and provides a
|
||||
set of functions to encapsulate a data location (URI or path), interogate
|
||||
@@ -29,7 +29,7 @@ class DataLocator():
|
||||
self.uri_or_path = uri_or_path
|
||||
self.protocol, self.path = DataLocator._get_protocol_and_path(uri_or_path)
|
||||
# work-around for LocalFileSystem not treating file: and None as the same scheme/protocol
|
||||
self.cname = self.path if self.protocol == 'file' else self.uri_or_path
|
||||
self.cname = self.path if self.protocol == "file" else self.uri_or_path
|
||||
# will throw RuntimeError if the protocol is unsupported
|
||||
self.fs = fsspec.filesystem(self.protocol)
|
||||
|
||||
@@ -53,9 +53,9 @@ class DataLocator():
|
||||
""" return datetime object representing last modification time, or None if unavailable """
|
||||
info = self.fs.info(self.cname)
|
||||
if self.islocal() and info is not None:
|
||||
return datetime.fromtimestamp(info['mtime'])
|
||||
return datetime.fromtimestamp(info["mtime"])
|
||||
else:
|
||||
return getattr(info, 'LastModified', None)
|
||||
return getattr(info, "LastModified", None)
|
||||
|
||||
def abspath(self):
|
||||
"""
|
||||
@@ -74,7 +74,7 @@ class DataLocator():
|
||||
return self.fs.open(self.uri_or_path, *args)
|
||||
|
||||
def islocal(self):
|
||||
return self.protocol is None or self.protocol == 'file'
|
||||
return self.protocol is None or self.protocol == "file"
|
||||
|
||||
def local_handle(self):
|
||||
if self.islocal():
|
||||
@@ -90,7 +90,7 @@ class DataLocator():
|
||||
return LocalFilePath(tmp_path, delete=True)
|
||||
|
||||
|
||||
class LocalFilePath():
|
||||
class LocalFilePath:
|
||||
def __init__(self, tmp_path, delete=False):
|
||||
self.tmp_path = tmp_path
|
||||
self.delete = delete
|
||||
|
||||
@@ -27,7 +27,7 @@ def CreateNumpyVector(builder, x):
|
||||
if not isinstance(x, np.ndarray):
|
||||
raise TypeError(f"non-numpy-ndarray passed to CreateNumpyVector ({type(x)}")
|
||||
|
||||
if x.dtype.kind not in ['b', 'i', 'u', 'f']:
|
||||
if x.dtype.kind not in ["b", "i", "u", "f"]:
|
||||
raise TypeError("numpy-ndarray holds elements of unsupported datatype")
|
||||
|
||||
if x.ndim > 1:
|
||||
@@ -42,11 +42,11 @@ def CreateNumpyVector(builder, x):
|
||||
x_little_endian = x.byteswap(inplace=False)
|
||||
|
||||
# Calculate total length
|
||||
len = int(x_little_endian.itemsize * x_little_endian.size)
|
||||
builder.head = int(builder.Head() - len)
|
||||
length = int(x_little_endian.itemsize * x_little_endian.size)
|
||||
builder.head = int(builder.Head() - length)
|
||||
|
||||
# tobytes ensures c_contiguous ordering
|
||||
builder.Bytes[builder.Head():builder.Head() + len] = x_little_endian.tobytes(order='C')
|
||||
builder.Bytes[builder.Head() : builder.Head() + length] = x_little_endian.tobytes(order="C")
|
||||
|
||||
return builder.EndVector(x.size)
|
||||
|
||||
@@ -88,9 +88,9 @@ def serialize_typed_array(builder, source_array, encoding_info):
|
||||
arr = arr.to_series()
|
||||
|
||||
# convert to a simple ndarray
|
||||
if as_type == 'json':
|
||||
as_json = arr.to_json(orient='records')
|
||||
arr = np.array(bytearray(as_json, 'utf-8'))
|
||||
if as_type == "json":
|
||||
as_json = arr.to_json(orient="records")
|
||||
arr = np.array(bytearray(as_json, "utf-8"))
|
||||
else:
|
||||
if MatrixProxy.ismatrixproxy(arr) or sparse.issparse(arr):
|
||||
arr = arr.toarray()
|
||||
@@ -119,18 +119,16 @@ column_encoding_type_map = {
|
||||
np.dtype(np.float64).str: (TypedArray.TypedArray.Float32Array, np.float32),
|
||||
np.dtype(np.float32).str: (TypedArray.TypedArray.Float32Array, np.float32),
|
||||
np.dtype(np.float16).str: (TypedArray.TypedArray.Float32Array, np.float32),
|
||||
|
||||
np.dtype(np.int8).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.int16).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
|
||||
np.dtype(np.uint8).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
np.dtype(np.uint16).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32)
|
||||
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
}
|
||||
column_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, 'json')
|
||||
column_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
|
||||
|
||||
|
||||
def column_encoding(arr):
|
||||
@@ -141,11 +139,10 @@ index_encoding_type_map = {
|
||||
# array protocol string: ( array_type, as_type )
|
||||
np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
|
||||
np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32)
|
||||
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
}
|
||||
index_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, 'json')
|
||||
index_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
|
||||
|
||||
|
||||
def index_encoding(arr):
|
||||
@@ -163,7 +160,7 @@ def guess_at_mem_needed(matrix):
|
||||
guess = 1
|
||||
|
||||
# round up to nearest 1024 bytes
|
||||
guess = (guess + 0x400) & (~0x3ff)
|
||||
guess = (guess + 0x400) & (~0x3FF)
|
||||
return guess
|
||||
|
||||
|
||||
@@ -223,7 +220,7 @@ def deserialize_typed_array(tarr):
|
||||
TypedArray.TypedArray.Int32Array: Int32Array.Int32Array,
|
||||
TypedArray.TypedArray.Float32Array: Float32Array.Float32Array,
|
||||
TypedArray.TypedArray.Float64Array: Float64Array.Float64Array,
|
||||
TypedArray.TypedArray.JSONEncodedArray: JSONEncodedArray.JSONEncodedArray
|
||||
TypedArray.TypedArray.JSONEncodedArray: JSONEncodedArray.JSONEncodedArray,
|
||||
}
|
||||
(u_type, u) = tarr
|
||||
if u_type is TypedArray.TypedArray.NONE:
|
||||
@@ -237,7 +234,7 @@ def deserialize_typed_array(tarr):
|
||||
arr.Init(u.Bytes, u.Pos)
|
||||
narr = arr.DataAsNumpy()
|
||||
if u_type == TypedArray.TypedArray.JSONEncodedArray:
|
||||
narr = json.loads(narr.tostring().decode('utf-8'))
|
||||
narr = json.loads(narr.tostring().decode("utf-8"))
|
||||
return narr
|
||||
|
||||
|
||||
|
||||
@@ -18,6 +18,7 @@ class _ArrayProxyBase(abc.ABC):
|
||||
Private base class for array or matrix proxy. This summarizes
|
||||
the interface used by the rest of cellxgene.
|
||||
"""
|
||||
|
||||
@property
|
||||
@abc.abstractmethod
|
||||
def dtype(self):
|
||||
@@ -68,10 +69,10 @@ class MatrixProxy(_ArrayProxyBase):
|
||||
Sub-classes automatically register.
|
||||
"""
|
||||
base_proxy_registry = {
|
||||
'pandas.core.frame.DataFrame': True,
|
||||
'numpy.ndarray': True,
|
||||
'scipy.sparse.csc.csc_matrix': True,
|
||||
'scipy.sparse.csr.csr_matrix': True,
|
||||
"pandas.core.frame.DataFrame": True,
|
||||
"numpy.ndarray": True,
|
||||
"scipy.sparse.csc.csc_matrix": True,
|
||||
"scipy.sparse.csr.csr_matrix": True,
|
||||
}
|
||||
proxy_registry = None
|
||||
last_cache_token = None
|
||||
@@ -103,7 +104,7 @@ class MatrixProxy(_ArrayProxyBase):
|
||||
"""
|
||||
cls.build_proxy_registry()
|
||||
t = type(matrix)
|
||||
fqtn = t.__module__ + '.' + t.__name__
|
||||
fqtn = t.__module__ + "." + t.__name__
|
||||
proxy_cls = cls.proxy_registry.get(fqtn, None)
|
||||
if proxy_cls is None:
|
||||
raise Exception(f"Matrix format `{fqtn}` is unsupported by proxy.")
|
||||
@@ -128,21 +129,17 @@ class MatrixProxyView(MatrixProxy):
|
||||
"""
|
||||
2D matrix view to a 2D matrix
|
||||
"""
|
||||
def __init__(self, arg1, shape=None, index=(),
|
||||
transposed=False, copy=False):
|
||||
|
||||
def __init__(self, arg1, shape=None, index=(), transposed=False, copy=False):
|
||||
if not copy:
|
||||
m = arg1
|
||||
super().__init__(m)
|
||||
|
||||
if shape is None:
|
||||
shape = m.shape
|
||||
assert(len(shape) == 2)
|
||||
assert len(shape) == 2
|
||||
|
||||
index = tuple(
|
||||
map(lambda s_i:
|
||||
slice(0, s_i[0], 1) if s_i[1] is None else s_i[1],
|
||||
zip_longest(shape, index))
|
||||
)
|
||||
index = tuple(map(lambda s_i: slice(0, s_i[0], 1) if s_i[1] is None else s_i[1], zip_longest(shape, index)))
|
||||
|
||||
self._shape = shape
|
||||
self._index = index
|
||||
@@ -234,20 +231,20 @@ class MatrixProxyView(MatrixProxy):
|
||||
NOTE: these follow the numpy rules for dimensionality reduction
|
||||
when an integer index is specified.
|
||||
"""
|
||||
|
||||
def _getitem_intXint(self, row, col):
|
||||
return self.m[row, col]
|
||||
|
||||
def _getitem_intXslice(self, row, col):
|
||||
shape = (_slice_length(col, self.m.shape[1]), )
|
||||
shape = (_slice_length(col, self.m.shape[1]),)
|
||||
return self.__class__.create_array(self.m, shape=shape, index=(row, col))
|
||||
|
||||
def _getitem_sliceXint(self, row, col):
|
||||
shape = (_slice_length(row, self.m.shape[0]), )
|
||||
shape = (_slice_length(row, self.m.shape[0]),)
|
||||
return self.__class__.create_array(self.m, shape=shape, index=(row, col))
|
||||
|
||||
def _getitem_sliceXslice(self, row, col):
|
||||
shape = (_slice_length(row, self.m.shape[0]),
|
||||
_slice_length(col, self.m.shape[1]))
|
||||
shape = (_slice_length(row, self.m.shape[0]), _slice_length(col, self.m.shape[1]))
|
||||
return self.__class__(self.m, shape=shape, index=(row, col), transposed=self.transposed)
|
||||
|
||||
def toarray(self):
|
||||
@@ -261,22 +258,23 @@ class ArrayProxyView(_ArrayProxyBase):
|
||||
"""
|
||||
1D array view to a 2D matrix
|
||||
"""
|
||||
|
||||
def __init__(self, arg1, shape=None, index=None, copy=False):
|
||||
super().__init__()
|
||||
if not copy:
|
||||
m = arg1
|
||||
|
||||
# one index MUST be an integer and the other MUST be a slice
|
||||
assert(len(index) == 2)
|
||||
assert(all(isinstance(idx, INT_TYPES + (slice, )) for idx in index))
|
||||
assert(isinstance(index[0], INT_TYPES) != isinstance(index[1], INT_TYPES))
|
||||
assert len(index) == 2
|
||||
assert all(isinstance(idx, INT_TYPES + (slice,)) for idx in index)
|
||||
assert isinstance(index[0], INT_TYPES) != isinstance(index[1], INT_TYPES)
|
||||
|
||||
if shape is None:
|
||||
if isinstance(index[0], INT_TYPES):
|
||||
shape = (m.shape[0], )
|
||||
shape = (m.shape[0],)
|
||||
else:
|
||||
shape = (m.shape[1], )
|
||||
assert(len(shape) == 1)
|
||||
shape = (m.shape[1],)
|
||||
assert len(shape) == 1
|
||||
|
||||
self._shape = shape
|
||||
self.m = m
|
||||
@@ -336,7 +334,7 @@ class ArrayProxyView(_ArrayProxyBase):
|
||||
elif isinstance(col, slice):
|
||||
return self._getitem_intXslice(row, col)
|
||||
elif isinstance(row, slice):
|
||||
assert(isinstance(col, INT_TYPES))
|
||||
assert isinstance(col, INT_TYPES)
|
||||
return self._getitem_sliceXint(row, col)
|
||||
|
||||
raise IndexError("unsupported column index types")
|
||||
@@ -345,11 +343,11 @@ class ArrayProxyView(_ArrayProxyBase):
|
||||
return self.m[row, col]
|
||||
|
||||
def _getitem_intXslice(self, row, col):
|
||||
shape = (_slice_length(col, self.m.shape[1]), )
|
||||
shape = (_slice_length(col, self.m.shape[1]),)
|
||||
return self.__class__(self.m, shape=shape, index=(row, col))
|
||||
|
||||
def _getitem_sliceXint(self, row, col):
|
||||
shape = (_slice_length(row, self.m.shape[0]), )
|
||||
shape = (_slice_length(row, self.m.shape[0]),)
|
||||
return self.__class__(self.m, shape=shape, index=(row, col))
|
||||
|
||||
def toarray(self):
|
||||
@@ -358,7 +356,7 @@ class ArrayProxyView(_ArrayProxyBase):
|
||||
|
||||
def _unpack_index(index, shape):
|
||||
if not isinstance(index, tuple):
|
||||
index = (index, )
|
||||
index = (index,)
|
||||
if len(shape) < len(index):
|
||||
raise IndexError("invalid index dimensionality - must be 2")
|
||||
|
||||
@@ -366,7 +364,7 @@ def _unpack_index(index, shape):
|
||||
for shp, idx in zip_longest(shape, index):
|
||||
idx = slice(None) if idx is None else idx
|
||||
idx = _slice_defaults(idx, shp) if isinstance(idx, slice) else idx
|
||||
unpacked += (idx, )
|
||||
unpacked += (idx,)
|
||||
|
||||
return unpacked
|
||||
|
||||
@@ -376,7 +374,7 @@ def _slice_slice(outer, outer_len, inner, inner_len):
|
||||
slice a slice - we take advantage of Python 3 range's support
|
||||
for indexing.
|
||||
"""
|
||||
assert(outer_len >= inner_len)
|
||||
assert outer_len >= inner_len
|
||||
outer_rng = range(*outer.indices(outer_len))
|
||||
rng = outer_rng[inner]
|
||||
start, stop, step = rng.start, rng.stop, rng.step
|
||||
@@ -387,8 +385,8 @@ def _slice_slice(outer, outer_len, inner, inner_len):
|
||||
|
||||
def _range_length(start, stop, step):
|
||||
""" return length of range """
|
||||
assert(step != 0)
|
||||
assert(start is not None and stop is not None and step is not None)
|
||||
assert step != 0
|
||||
assert start is not None and stop is not None and step is not None
|
||||
if step > 0 and start < stop:
|
||||
return 1 + (stop - 1 - start) // step
|
||||
elif step < 0 and start > stop:
|
||||
@@ -404,7 +402,7 @@ def _slice_length(s, length):
|
||||
|
||||
def _slice_defaults(s, length):
|
||||
""" apply slice defaulting conventions """
|
||||
assert(length >= 0)
|
||||
assert length >= 0
|
||||
|
||||
step = 1 if s.step is None else s.step
|
||||
|
||||
|
||||
@@ -40,4 +40,5 @@ def requires_data(func):
|
||||
if self.data is None:
|
||||
raise DriverError(f"error data must be loaded before you call {func.__name__}")
|
||||
return func(self, *args, **kwargs)
|
||||
|
||||
return wrapped_function
|
||||
|
||||
Reference in New Issue
Block a user