Makefile modularity, test targets, and auto-formatting (#1070)

* Fix Makefile whitespace and .PHONY use

* Fix Makefile filename

* Modularize Makefile into client and server Makefiles

Part of the reason that the Makefile in the root directory is a bit
complicated is that it tries to handle tasks that can be handled
separately in the client and server modules.

This commit pushes some of the make logic specific to each module into
their own makefiles and calls out to those makefiles from that in the
project root.

* Add auto-formatting to client and server modules

One thing that can make linting faster is auto-formatting. This commit
adds the yapf auto-formatting tool to the server module and uses
eslint's "fix" functionality to speed up the linting/formatting process.

* Add yapf for automatic code formatting

* Add a root test target that calls sub-tests

* Apply yapf to python files

* Do not duplicate npm commands, simply pass through

* Update documentation

* Do not shadow reserved word len

* Add general test target

* Fix make call in dev-env

* Use black instead of yapf

* Run flake8 from the root directory

* Revert "Apply yapf to python files"

This reverts commit cdca128a01.

* Apply black to python code

* Resolve lint errors resulting from black format

* Add explanation of server unit tests in dev guidelines
This commit is contained in:
Matt Weiden
2019-12-27 14:43:37 -08:00
committed by GitHub
parent ec79995be8
commit f3015cb9df
37 changed files with 738 additions and 806 deletions
+2 -2
View File
@@ -33,7 +33,7 @@ class CXGDriver(metaclass=ABCMeta):
"max_category_items": None,
"diffexp_lfc_cutoff": None,
"disable_diffexp": False,
"diffexp_may_be_slow": False
"diffexp_may_be_slow": False,
}
@abstractmethod
@@ -51,7 +51,7 @@ class CXGDriver(metaclass=ABCMeta):
features = {
"cluster": {"available": False},
"layout": {"obs": {"available": False}, "var": {"available": False}},
"diffexp": {"available": True, "interactiveLimit": 50000}
"diffexp": {"available": True, "interactiveLimit": 50000},
}
# TODO - Interactive limit should be generated from the actual available methods see GH issue #94
if self.config["layout"]:
+34 -91
View File
@@ -8,13 +8,7 @@ from flask_restful import Api, Resource
from server import __version__ as cellxgene_version
from anndata import __version__ as anndata_version
from server.app.util.constants import (
Axis,
DiffExpMode,
JSON_NaN_to_num_warning_msg,
CXGUID,
CXG_ANNO_COLLECTION
)
from server.app.util.constants import Axis, DiffExpMode, JSON_NaN_to_num_warning_msg, CXGUID, CXG_ANNO_COLLECTION
from server.app.util.errors import (
FilterError,
InteractiveError,
@@ -40,41 +34,18 @@ class ConfigAPI(Resource):
config = {
"config": {
"features": [
{
"method": "POST",
"path": "/cluster/",
**current_app.data.features["cluster"],
},
{
"method": "POST",
"path": "/layout/obs",
**current_app.data.features["layout"]["obs"],
},
{
"method": "POST",
"path": "/layout/var",
**current_app.data.features["layout"]["var"],
},
{
"method": "POST",
"path": "/diffexp/",
**current_app.data.features["diffexp"],
},
{"method": "POST", "path": "/cluster/", **current_app.data.features["cluster"]},
{"method": "POST", "path": "/layout/obs", **current_app.data.features["layout"]["obs"]},
{"method": "POST", "path": "/layout/var", **current_app.data.features["layout"]["var"]},
{"method": "POST", "path": "/diffexp/", **current_app.data.features["diffexp"]},
],
"displayNames": {
"engine": f"cellxgene Scanpy engine version ",
"dataset": current_app.config["DATASET_TITLE"],
},
"links": {
"about-dataset": current_app.config["ABOUT_DATASET"]
},
"parameters": {
**current_app.data.get_config_parameters(uid=cxguid, collection=anno_collection)
},
"library_versions": {
"cellxgene": cellxgene_version,
"anndata": anndata_version
}
"links": {"about-dataset": current_app.config["ABOUT_DATASET"]},
"parameters": {**current_app.data.get_config_parameters(uid=cxguid, collection=anno_collection)},
"library_versions": {"cellxgene": cellxgene_version, "anndata": anndata_version},
}
}
@@ -84,17 +55,13 @@ class ConfigAPI(Resource):
class AnnotationsObsAPI(Resource):
def get(self):
fields = request.args.getlist("annotation-name", None)
preferred_mimetype = request.accept_mimetypes.best_match(
["application/octet-stream"]
)
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
cxguid = get_userid(session)
anno_collection = get_anno_collection(session)
try:
if preferred_mimetype == "application/octet-stream":
fbs = current_app.data.annotation_to_fbs_matrix("obs", fields, uid=cxguid, collection=anno_collection)
return make_response(fbs,
HTTPStatus.OK,
{"Content-Type": "application/octet-stream"})
return make_response(fbs, HTTPStatus.OK, {"Content-Type": "application/octet-stream"})
else:
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
except KeyError:
@@ -115,9 +82,7 @@ class AnnotationsObsAPI(Resource):
try:
fbs = request.get_data()
res = current_app.data.annotation_put_fbs("obs", fbs, uid=cxguid, collection=anno_collection)
return make_response(
res, HTTPStatus.OK, {"Content-Type": "application/json"}
)
return make_response(res, HTTPStatus.OK, {"Content-Type": "application/json"})
except (ValueError, DisabledFeatureError, KeyError) as e:
return make_response(str(e), HTTPStatus.BAD_REQUEST)
except Exception as e:
@@ -127,14 +92,14 @@ class AnnotationsObsAPI(Resource):
class AnnotationsVarAPI(Resource):
def get(self):
fields = request.args.getlist("annotation-name", None)
preferred_mimetype = request.accept_mimetypes.best_match(
["application/octet-stream"]
)
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
try:
if preferred_mimetype == "application/octet-stream":
return make_response(current_app.data.annotation_to_fbs_matrix("var", fields),
HTTPStatus.OK,
{"Content-Type": "application/octet-stream"})
return make_response(
current_app.data.annotation_to_fbs_matrix("var", fields),
HTTPStatus.OK,
{"Content-Type": "application/octet-stream"},
)
else:
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
except KeyError:
@@ -145,19 +110,16 @@ class AnnotationsVarAPI(Resource):
class DataVarAPI(Resource):
def put(self):
preferred_mimetype = request.accept_mimetypes.best_match(
["application/octet-stream"]
)
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
try:
if preferred_mimetype == "application/octet-stream":
filter_json = request.get_json()
filter = filter_json["filter"] if filter_json else None
return make_response(
current_app.data.data_frame_to_fbs_matrix(
filter, axis=Axis.VAR
),
current_app.data.data_frame_to_fbs_matrix(filter, axis=Axis.VAR),
HTTPStatus.OK,
{"Content-Type": "application/octet-stream"})
{"Content-Type": "application/octet-stream"},
)
else:
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
except FilterError as e:
@@ -175,35 +137,23 @@ class DiffExpObsAPI(Resource):
except KeyError:
return make_response("Error: mode is required", HTTPStatus.BAD_REQUEST)
except ValueError:
return make_response(
f"Error: invalid mode option {args['mode']}", HTTPStatus.BAD_REQUEST
)
return make_response(f"Error: invalid mode option {args['mode']}", HTTPStatus.BAD_REQUEST)
# Validate filters
if mode == DiffExpMode.VAR_FILTER or "varFilter" in args:
# not NOT_IMPLEMENTED
return make_response(
"mode=varfilter not implemented", HTTPStatus.NOT_IMPLEMENTED
)
return make_response("mode=varfilter not implemented", HTTPStatus.NOT_IMPLEMENTED)
if mode == DiffExpMode.TOP_N and "count" not in args:
return make_response(
"mode=topN requires a count parameter", HTTPStatus.BAD_REQUEST
)
return make_response("mode=topN requires a count parameter", HTTPStatus.BAD_REQUEST)
if "set1" not in args:
return make_response("set1 is required.", HTTPStatus.BAD_REQUEST)
if Axis.VAR in args["set1"]["filter"]:
return make_response(
"Var filter not allowed for set1", HTTPStatus.BAD_REQUEST
)
return make_response("Var filter not allowed for set1", HTTPStatus.BAD_REQUEST)
# set2
if "set2" not in args:
return make_response(
"Set2 as inverse of set1 is not implemented", HTTPStatus.NOT_IMPLEMENTED
)
return make_response("Set2 as inverse of set1 is not implemented", HTTPStatus.NOT_IMPLEMENTED)
if Axis.VAR in args["set2"]["filter"]:
return make_response(
"Var filter not allowed for set2", HTTPStatus.BAD_REQUEST
)
return make_response("Var filter not allowed for set2", HTTPStatus.BAD_REQUEST)
set1_filter = args["set1"]["filter"]
set2_filter = args.get("set2", {"filter": {}})["filter"]
@@ -214,14 +164,9 @@ class DiffExpObsAPI(Resource):
count = args.get("count", None)
try:
diffexp = current_app.data.diffexp_topN(
set1_filter,
set2_filter,
count,
current_app.data.features["diffexp"]["interactiveLimit"],
)
return make_response(
diffexp, HTTPStatus.OK, {"Content-Type": "application/json"}
set1_filter, set2_filter, count, current_app.data.features["diffexp"]["interactiveLimit"],
)
return make_response(diffexp, HTTPStatus.OK, {"Content-Type": "application/json"})
except (ValueError, FilterError) as e:
return make_response(e.message, HTTPStatus.BAD_REQUEST)
except InteractiveError:
@@ -236,14 +181,12 @@ class DiffExpObsAPI(Resource):
class LayoutObsAPI(Resource):
def get(self):
preferred_mimetype = request.accept_mimetypes.best_match(
["application/octet-stream"]
)
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
try:
if preferred_mimetype == "application/octet-stream":
return make_response(current_app.data.layout_to_fbs_matrix(),
HTTPStatus.OK,
{"Content-Type": "application/octet-stream"})
return make_response(
current_app.data.layout_to_fbs_matrix(), HTTPStatus.OK, {"Content-Type": "application/octet-stream"}
)
else:
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
except PrepareError as e:
@@ -278,7 +221,7 @@ def is_safe_collection_name(name):
"""
if name is None:
return False
return re.match(r'^\w+$', name) is not None
return re.match(r"^\w+$", name) is not None
def get_api_resources():
+2 -2
View File
@@ -32,8 +32,8 @@ def _mean_var_n(X):
v = sumsq / (n - 1)
if fp_err_occurred:
mean[np.isfinite(mean) == False] = 0 # noqa: E712
v[np.isfinite(v) == False] = 0 # noqa: E712
mean[np.isfinite(mean) == False] = 0 # noqa: E712
v[np.isfinite(v) == False] = 0 # noqa: E712
return mean, v, n
+4 -4
View File
@@ -9,7 +9,7 @@ import pandas as pd
def read_labels(fname):
if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
return pd.read_csv(fname, dtype='category', index_col=0, header=0, comment='#')
return pd.read_csv(fname, dtype="category", index_col=0, header=0, comment="#")
else:
return pd.DataFrame()
@@ -19,12 +19,12 @@ def write_labels(fname, df, header=None, backup_dir=None):
backup(fname, backup_dir)
# rotate_fname(fname, backup_dir)
if not df.empty:
with open(fname, 'w', newline="") as f:
with open(fname, "w", newline="") as f:
if header is not None:
f.write(header)
df.to_csv(f)
else:
open(fname, 'w').close()
open(fname, "w").close()
def backup(fname, backup_dir, max_backups=9):
@@ -46,7 +46,7 @@ def backup(fname, backup_dir, max_backups=9):
fname_base = os.path.basename(fname)
fname_base_root, fname_base_ext = os.path.splitext(fname_base)
# don't use ISO standard time format, as it contains characters illegal on some filesytems.
nowish = datetime.now().strftime('%Y-%m-%dT%H-%M-%S')
nowish = datetime.now().strftime("%Y-%m-%dT%H-%M-%S")
backup_fname = os.path.join(backup_dir, f"{fname_base_root}-{nowish}{fname_base_ext}")
if os.path.exists(backup_fname):
os.remove(backup_fname)
+8 -5
View File
@@ -1,4 +1,3 @@
from server.app.util.matrix_proxy import MatrixProxyView, ArrayProxyView
"""
@@ -17,6 +16,7 @@ class ArrayProxyView_anndata_h5py(ArrayProxyView):
override to handle sparse getitem semantics, which differ
from numpy.
"""
def toarray(self):
""" sadly, sparse indexing doesn't drop dimensions like numpy! """
arr = self.m[self._index[0], self._index[1]]
@@ -30,12 +30,15 @@ class MatrixProxy_anndata_h5py(MatrixProxyView):
AnnData sparse array stored in H5AD, or proxies for backed data.
None of these handle indexing very well, so we plop a proxy on top.
"""
@classmethod
def __supports__(cls):
return ("anndata.h5py.h5sparse.SparseDataset",
"anndata.h5py.h5sparse.backed_csc_matrix",
"anndata.h5py.h5sparse.backed_csr_matrix",
"h5py._hl.dataset.Dataset")
return (
"anndata.h5py.h5sparse.SparseDataset",
"anndata.h5py.h5sparse.backed_csc_matrix",
"anndata.h5py.h5sparse.backed_csr_matrix",
"h5py._hl.dataset.Dataset",
)
@classmethod
def create_array(cls, *args, **kwargs):
+65 -99
View File
@@ -62,7 +62,7 @@ class ScanpyEngine(CXGDriver):
"annotations_output_dir": None,
"backed": False,
"disable_diffexp": False,
"diffexp_may_be_slow": False
"diffexp_may_be_slow": False,
}
def get_config_parameters(self, uid=None, collection=None):
@@ -70,26 +70,25 @@ class ScanpyEngine(CXGDriver):
"max-category-items": self.config["max_category_items"],
"disable-diffexp": self.config["disable_diffexp"],
"diffexp-may-be-slow": self.config["diffexp_may_be_slow"],
"annotations": self.config["annotations"]
"annotations": self.config["annotations"],
}
if self.config["annotations"]:
if uid is not None:
params.update({
"annotations-user-data-idhash": self.get_userdata_idhash(uid)
})
if self.config['annotations_file'] is not None:
params.update({"annotations-user-data-idhash": self.get_userdata_idhash(uid)})
if self.config["annotations_file"] is not None:
# user has hard-wired the name of the annotation data collection
fname = os.path.basename(self.config['annotations_file'])
fname = os.path.basename(self.config["annotations_file"])
collection_fname = os.path.splitext(fname)[0]
params.update({
'annotations-data-collection-is-read-only': True,
'annotations-data-collection-name': collection_fname
})
params.update(
{
"annotations-data-collection-is-read-only": True,
"annotations-data-collection-name": collection_fname,
}
)
elif collection is not None:
params.update({
'annotations-data-collection-is-read-only': False,
'annotations-data-collection-name': collection
})
params.update(
{"annotations-data-collection-is-read-only": False, "annotations-data-collection-name": collection}
)
return params
@staticmethod
@@ -140,23 +139,18 @@ class ScanpyEngine(CXGDriver):
# User has specified alternative column for unique names, and it exists
if not df_axis[name].is_unique:
raise KeyError(
f"Values in {ax_name}.{name} must be unique. "
"Please prepare data to contain unique values."
f"Values in {ax_name}.{name} must be unique. " "Please prepare data to contain unique values."
)
df_axis.reset_index(drop=True, inplace=True)
else:
# user specified a non-existent column name
raise KeyError(
f"Annotation name {name}, specified in --{ax_name}-name does not exist."
)
raise KeyError(f"Annotation name {name}, specified in --{ax_name}-name does not exist.")
@staticmethod
def _can_cast_to_float32(ann):
if ann.dtype.kind == "f":
if not np.can_cast(ann.dtype, np.float32):
warnings.warn(
f"Annotation {ann.name} will be converted to 32 bit float and may lose precision."
)
warnings.warn(f"Annotation {ann.name} will be converted to 32 bit float and may lose precision.")
return True
return False
@@ -188,30 +182,18 @@ class ScanpyEngine(CXGDriver):
schema["type"] = "categorical"
schema["categories"] = dtype.categories.tolist()
else:
raise TypeError(
f"Annotations of type {dtype} are unsupported by cellxgene."
)
raise TypeError(f"Annotations of type {dtype} are unsupported by cellxgene.")
return schema
@requires_data
def _create_schema(self):
self.schema = {
"dataframe": {
"nObs": self.cell_count,
"nVar": self.gene_count,
"type": str(self.data.X.dtype),
},
"dataframe": {"nObs": self.cell_count, "nVar": self.gene_count, "type": str(self.data.X.dtype)},
"annotations": {
"obs": {
"index": self.config["obs_names"],
"columns": []
},
"var": {
"index": self.config["var_names"],
"columns": []
}
"obs": {"index": self.config["obs_names"], "columns": []},
"var": {"index": self.config["var_names"], "columns": []},
},
"layout": {"obs": []}
"layout": {"obs": []},
}
for ax in Axis:
curr_axis = getattr(self.data, str(ax))
@@ -220,12 +202,8 @@ class ScanpyEngine(CXGDriver):
ann_schema.update(self._get_col_type(curr_axis[ann]))
self.schema["annotations"][ax]["columns"].append(ann_schema)
for layout in self.config['layout']:
layout_schema = {
"name": layout,
"type": "float32",
"dims": [f"{layout}_0", f"{layout}_1"]
}
for layout in self.config["layout"]:
layout_schema = {"name": layout, "type": "float32", "dims": [f"{layout}_0", f"{layout}_1"]}
self.schema["layout"]["obs"].append(layout_schema)
@requires_data
@@ -250,7 +228,7 @@ class ScanpyEngine(CXGDriver):
Used to create safe annotations output file names.
"""
id = (uid + self.data_locator.abspath()).encode()
idhash = base64.b32encode(blake2b(id, digest_size=5).digest()).decode('utf-8')
idhash = base64.b32encode(blake2b(id, digest_size=5).digest()).decode("utf-8")
return idhash
def get_anno_fname(self, uid=None, collection=None):
@@ -272,11 +250,11 @@ class ScanpyEngine(CXGDriver):
if not self.config["annotations"]:
return None
if self.config['annotations_output_dir']:
return self.config['annotations_output_dir']
if self.config["annotations_output_dir"]:
return self.config["annotations_output_dir"]
if self.config['annotations_file']:
return os.path.dirname(os.path.abspath(self.config['annotations_file']))
if self.config["annotations_file"]:
return os.path.dirname(os.path.abspath(self.config["annotations_file"]))
return os.getcwd()
@@ -299,7 +277,7 @@ class ScanpyEngine(CXGDriver):
with data_locator.local_handle() as lh:
# as of AnnData 0.6.19, backed mode performs initial load fast, but at the
# cost of significantly slower access to X data.
backed = 'r' if self.config['backed'] else None
backed = "r" if self.config["backed"] else None
self.data = anndata.read_h5ad(lh, backed=backed)
except ValueError:
@@ -338,7 +316,7 @@ class ScanpyEngine(CXGDriver):
# heuristic
n_values = self.data.shape[0] * self.data.shape[1]
if (n_values > 1e8 and self.config['backed'] is True) or (n_values > 5e8):
if (n_values > 1e8 and self.config["backed"] is True) or (n_values > 5e8):
self.config.update({"diffexp_may_be_slow": True})
@requires_data
@@ -348,7 +326,7 @@ class ScanpyEngine(CXGDriver):
b) validate layouts are legal. remove/warn on any that are not
c) cap total list of layouts at global const MAX_LAYOUTS
"""
layouts = self.config['layout']
layouts = self.config["layout"]
# handle default
if layouts is None or len(layouts) == 0:
# load default layouts from the data.
@@ -372,7 +350,7 @@ class ScanpyEngine(CXGDriver):
raise PrepareError(f"No valid layout data.")
# cap layouts to MAX_LAYOUTS
self.config['layout'] = valid_layouts[0:MAX_LAYOUTS]
self.config["layout"] = valid_layouts[0:MAX_LAYOUTS]
@requires_data
def _is_valid_layout(self, arr):
@@ -394,8 +372,7 @@ class ScanpyEngine(CXGDriver):
)
if self.data.X.dtype != "float32":
warnings.warn(
f"Scanpy data matrix is in {self.data.X.dtype} format not float32. "
f"Precision may be truncated."
f"Scanpy data matrix is in {self.data.X.dtype} format not float32. " f"Precision may be truncated."
)
for ax in Axis:
curr_axis = getattr(self.data, str(ax))
@@ -414,7 +391,7 @@ class ScanpyEngine(CXGDriver):
)
if isinstance(datatype, CategoricalDtype):
category_num = len(curr_axis[ann].dtype.categories)
if category_num > 500 and category_num > self.config['max_category_items']:
if category_num > 500 and category_num > self.config["max_category_items"]:
warnings.warn(
f"{str(ax).title()} annotation '{ann}' has {category_num} categories, this may be "
f"cumbersome or slow to display. We recommend setting the "
@@ -439,14 +416,17 @@ class ScanpyEngine(CXGDriver):
raise KeyError(f"All row index values specified in user annotations must be unique.")
if not labels.index.equals(self.original_obs_index):
raise KeyError("Label file row index does not match H5AD file index. "
"Please ensure that column zero (0) in the label file contain the same "
"index values as the H5AD file.")
raise KeyError(
"Label file row index does not match H5AD file index. "
"Please ensure that column zero (0) in the label file contain the same "
"index values as the H5AD file."
)
duplicate_columns = list(set(labels.columns) & set(self.data.obs.columns))
if len(duplicate_columns) > 0:
raise KeyError(f"Labels file may not contain column names which overlap "
f"with h5ad obs columns {duplicate_columns}")
raise KeyError(
f"Labels file may not contain column names which overlap " f"with h5ad obs columns {duplicate_columns}"
)
# labels must have same count as obs annotations
if labels.shape[0] != self.data.obs.shape[0]:
@@ -475,7 +455,7 @@ class ScanpyEngine(CXGDriver):
mask = np.zeros((count,), dtype=bool)
for i in filter:
if type(i) == list:
mask[i[0]: i[1]] = True
mask[i[0] : i[1]] = True
else:
mask[i] = True
return mask
@@ -484,15 +464,10 @@ class ScanpyEngine(CXGDriver):
def _axis_filter_to_mask(filter, d_axis, count):
mask = np.ones((count,), dtype=bool)
if "index" in filter:
mask = np.logical_and(
mask, ScanpyEngine._index_filter_to_mask(filter["index"], count)
)
mask = np.logical_and(mask, ScanpyEngine._index_filter_to_mask(filter["index"], count))
if "annotation_value" in filter:
mask = np.logical_and(
mask,
ScanpyEngine._annotation_filter_to_mask(
filter["annotation_value"], d_axis, count
),
mask, ScanpyEngine._annotation_filter_to_mask(filter["annotation_value"], d_axis, count),
)
return mask
@@ -507,13 +482,9 @@ class ScanpyEngine(CXGDriver):
if filter is not None:
if Axis.OBS in filter:
obs_selector = self._axis_filter_to_mask(
filter["obs"], self.data.obs, self.data.n_obs
)
obs_selector = self._axis_filter_to_mask(filter["obs"], self.data.obs, self.data.n_obs)
if Axis.VAR in filter:
var_selector = self._axis_filter_to_mask(
filter["var"], self.data.var, self.data.n_vars
)
var_selector = self._axis_filter_to_mask(filter["var"], self.data.var, self.data.n_vars)
return obs_selector, var_selector
@requires_data
@@ -531,7 +502,7 @@ class ScanpyEngine(CXGDriver):
labels = None
if labels is not None and not labels.empty:
df = self.data.obs.join(labels, self.config['obs_names'])
df = self.data.obs.join(labels, self.config["obs_names"])
else:
df = self.data.obs
else:
@@ -560,18 +531,21 @@ class ScanpyEngine(CXGDriver):
# if any of the new column labels overlap with our existing labels, raise error
duplicate_columns = list(set(new_label_df.columns) & set(self.data.obs.columns))
if not new_label_df.columns.is_unique or len(duplicate_columns) > 0:
raise KeyError(f"Labels file may not contain column names which overlap "
f"with h5ad obs columns {duplicate_columns}")
raise KeyError(
f"Labels file may not contain column names which overlap " f"with h5ad obs columns {duplicate_columns}"
)
# update our internal state and save it. Multi-threading often enabled,
# so treat this as a critical section.
with self.label_lock:
lastmod = self.data_locator.lastmodtime()
lastmodstr = "'unknown'" if lastmod is None else lastmod.isoformat(timespec="seconds")
header = f"# Annotations generated on {datetime.now().isoformat(timespec='seconds')} " \
f"using cellxgene version {cellxgene_version}\n" \
f"# Input data file was {self.data_locator.uri_or_path}, " \
f"which was last modified on {lastmodstr}\n"
header = (
f"# Annotations generated on {datetime.now().isoformat(timespec='seconds')} "
f"using cellxgene version {cellxgene_version}\n"
f"# Input data file was {self.data_locator.uri_or_path}, "
f"which was last modified on {lastmodstr}\n"
)
write_labels(fname, new_label_df, header, backup_dir=self.get_anno_backup_dir(uid, collection))
return jsonify_scanpy({"status": "OK"})
@@ -598,8 +572,7 @@ class ScanpyEngine(CXGDriver):
raise FilterError("filtering on obs unsupported")
# Currently only handles VAR dimension
X = MatrixProxy.create(self.data.X if var_selector is None
else self.data.X[:, var_selector])
X = MatrixProxy.create(self.data.X if var_selector is None else self.data.X[:, var_selector])
return encode_matrix_fbs(X, col_idx=np.nonzero(var_selector)[0], row_idx=None)
@requires_data
@@ -607,25 +580,17 @@ class ScanpyEngine(CXGDriver):
if Axis.VAR in obsFilterA or Axis.VAR in obsFilterB:
raise FilterError("Observation filters may not contain vaiable conditions")
try:
obs_mask_A = self._axis_filter_to_mask(
obsFilterA["obs"], self.data.obs, self.data.n_obs
)
obs_mask_B = self._axis_filter_to_mask(
obsFilterB["obs"], self.data.obs, self.data.n_obs
)
obs_mask_A = self._axis_filter_to_mask(obsFilterA["obs"], self.data.obs, self.data.n_obs)
obs_mask_B = self._axis_filter_to_mask(obsFilterB["obs"], self.data.obs, self.data.n_obs)
except (KeyError, IndexError) as e:
raise FilterError(f"Error parsing filter: {e}") from e
if top_n is None:
top_n = DEFAULT_TOP_N
result = diffexp_ttest(
self.data, obs_mask_A, obs_mask_B, top_n, self.config['diffexp_lfc_cutoff']
)
result = diffexp_ttest(self.data, obs_mask_A, obs_mask_B, top_n, self.config["diffexp_lfc_cutoff"])
try:
return jsonify_scanpy(result)
except ValueError:
raise JSONEncodingValueError(
"Error encoding differential expression to JSON"
)
raise JSONEncodingValueError("Error encoding differential expression to JSON")
@requires_data
def layout_to_fbs_matrix(self):
@@ -661,7 +626,8 @@ class ScanpyEngine(CXGDriver):
except ValueError as e:
raise PrepareError(
f"Layout has not been calculated using {self.config['layout']}, "
f"please prepare your datafile and relaunch cellxgene") from e
f"please prepare your datafile and relaunch cellxgene"
) from e
df = pandas.concat(layout_data, axis=1, copy=False)
return encode_matrix_fbs(df, col_idx=df.columns, row_idx=None)
+1 -3
View File
@@ -27,9 +27,7 @@ class DiffExpMode(AugmentedEnum):
VAR_FILTER = "varFilter"
JSON_NaN_to_num_warning_msg = (
"JSON encoding failure - please verify all data are finite values (no NaN or Infinities)"
)
JSON_NaN_to_num_warning_msg = "JSON encoding failure - please verify all data are finite values (no NaN or Infinities)"
REACTIVE_LIMIT = 1_000_000
MAX_LAYOUTS = 30
+6 -6
View File
@@ -4,7 +4,7 @@ import fsspec
from datetime import datetime
class DataLocator():
class DataLocator:
"""
DataLocator is a simple wrapper around fsspec functionality, and provides a
set of functions to encapsulate a data location (URI or path), interogate
@@ -29,7 +29,7 @@ class DataLocator():
self.uri_or_path = uri_or_path
self.protocol, self.path = DataLocator._get_protocol_and_path(uri_or_path)
# work-around for LocalFileSystem not treating file: and None as the same scheme/protocol
self.cname = self.path if self.protocol == 'file' else self.uri_or_path
self.cname = self.path if self.protocol == "file" else self.uri_or_path
# will throw RuntimeError if the protocol is unsupported
self.fs = fsspec.filesystem(self.protocol)
@@ -53,9 +53,9 @@ class DataLocator():
""" return datetime object representing last modification time, or None if unavailable """
info = self.fs.info(self.cname)
if self.islocal() and info is not None:
return datetime.fromtimestamp(info['mtime'])
return datetime.fromtimestamp(info["mtime"])
else:
return getattr(info, 'LastModified', None)
return getattr(info, "LastModified", None)
def abspath(self):
"""
@@ -74,7 +74,7 @@ class DataLocator():
return self.fs.open(self.uri_or_path, *args)
def islocal(self):
return self.protocol is None or self.protocol == 'file'
return self.protocol is None or self.protocol == "file"
def local_handle(self):
if self.islocal():
@@ -90,7 +90,7 @@ class DataLocator():
return LocalFilePath(tmp_path, delete=True)
class LocalFilePath():
class LocalFilePath:
def __init__(self, tmp_path, delete=False):
self.tmp_path = tmp_path
self.delete = delete
+14 -17
View File
@@ -27,7 +27,7 @@ def CreateNumpyVector(builder, x):
if not isinstance(x, np.ndarray):
raise TypeError(f"non-numpy-ndarray passed to CreateNumpyVector ({type(x)}")
if x.dtype.kind not in ['b', 'i', 'u', 'f']:
if x.dtype.kind not in ["b", "i", "u", "f"]:
raise TypeError("numpy-ndarray holds elements of unsupported datatype")
if x.ndim > 1:
@@ -42,11 +42,11 @@ def CreateNumpyVector(builder, x):
x_little_endian = x.byteswap(inplace=False)
# Calculate total length
len = int(x_little_endian.itemsize * x_little_endian.size)
builder.head = int(builder.Head() - len)
length = int(x_little_endian.itemsize * x_little_endian.size)
builder.head = int(builder.Head() - length)
# tobytes ensures c_contiguous ordering
builder.Bytes[builder.Head():builder.Head() + len] = x_little_endian.tobytes(order='C')
builder.Bytes[builder.Head() : builder.Head() + length] = x_little_endian.tobytes(order="C")
return builder.EndVector(x.size)
@@ -88,9 +88,9 @@ def serialize_typed_array(builder, source_array, encoding_info):
arr = arr.to_series()
# convert to a simple ndarray
if as_type == 'json':
as_json = arr.to_json(orient='records')
arr = np.array(bytearray(as_json, 'utf-8'))
if as_type == "json":
as_json = arr.to_json(orient="records")
arr = np.array(bytearray(as_json, "utf-8"))
else:
if MatrixProxy.ismatrixproxy(arr) or sparse.issparse(arr):
arr = arr.toarray()
@@ -119,18 +119,16 @@ column_encoding_type_map = {
np.dtype(np.float64).str: (TypedArray.TypedArray.Float32Array, np.float32),
np.dtype(np.float32).str: (TypedArray.TypedArray.Float32Array, np.float32),
np.dtype(np.float16).str: (TypedArray.TypedArray.Float32Array, np.float32),
np.dtype(np.int8).str: (TypedArray.TypedArray.Int32Array, np.int32),
np.dtype(np.int16).str: (TypedArray.TypedArray.Int32Array, np.int32),
np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
np.dtype(np.uint8).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
np.dtype(np.uint16).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32)
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
}
column_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, 'json')
column_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
def column_encoding(arr):
@@ -141,11 +139,10 @@ index_encoding_type_map = {
# array protocol string: ( array_type, as_type )
np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32)
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
}
index_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, 'json')
index_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
def index_encoding(arr):
@@ -163,7 +160,7 @@ def guess_at_mem_needed(matrix):
guess = 1
# round up to nearest 1024 bytes
guess = (guess + 0x400) & (~0x3ff)
guess = (guess + 0x400) & (~0x3FF)
return guess
@@ -223,7 +220,7 @@ def deserialize_typed_array(tarr):
TypedArray.TypedArray.Int32Array: Int32Array.Int32Array,
TypedArray.TypedArray.Float32Array: Float32Array.Float32Array,
TypedArray.TypedArray.Float64Array: Float64Array.Float64Array,
TypedArray.TypedArray.JSONEncodedArray: JSONEncodedArray.JSONEncodedArray
TypedArray.TypedArray.JSONEncodedArray: JSONEncodedArray.JSONEncodedArray,
}
(u_type, u) = tarr
if u_type is TypedArray.TypedArray.NONE:
@@ -237,7 +234,7 @@ def deserialize_typed_array(tarr):
arr.Init(u.Bytes, u.Pos)
narr = arr.DataAsNumpy()
if u_type == TypedArray.TypedArray.JSONEncodedArray:
narr = json.loads(narr.tostring().decode('utf-8'))
narr = json.loads(narr.tostring().decode("utf-8"))
return narr
+30 -32
View File
@@ -18,6 +18,7 @@ class _ArrayProxyBase(abc.ABC):
Private base class for array or matrix proxy. This summarizes
the interface used by the rest of cellxgene.
"""
@property
@abc.abstractmethod
def dtype(self):
@@ -68,10 +69,10 @@ class MatrixProxy(_ArrayProxyBase):
Sub-classes automatically register.
"""
base_proxy_registry = {
'pandas.core.frame.DataFrame': True,
'numpy.ndarray': True,
'scipy.sparse.csc.csc_matrix': True,
'scipy.sparse.csr.csr_matrix': True,
"pandas.core.frame.DataFrame": True,
"numpy.ndarray": True,
"scipy.sparse.csc.csc_matrix": True,
"scipy.sparse.csr.csr_matrix": True,
}
proxy_registry = None
last_cache_token = None
@@ -103,7 +104,7 @@ class MatrixProxy(_ArrayProxyBase):
"""
cls.build_proxy_registry()
t = type(matrix)
fqtn = t.__module__ + '.' + t.__name__
fqtn = t.__module__ + "." + t.__name__
proxy_cls = cls.proxy_registry.get(fqtn, None)
if proxy_cls is None:
raise Exception(f"Matrix format `{fqtn}` is unsupported by proxy.")
@@ -128,21 +129,17 @@ class MatrixProxyView(MatrixProxy):
"""
2D matrix view to a 2D matrix
"""
def __init__(self, arg1, shape=None, index=(),
transposed=False, copy=False):
def __init__(self, arg1, shape=None, index=(), transposed=False, copy=False):
if not copy:
m = arg1
super().__init__(m)
if shape is None:
shape = m.shape
assert(len(shape) == 2)
assert len(shape) == 2
index = tuple(
map(lambda s_i:
slice(0, s_i[0], 1) if s_i[1] is None else s_i[1],
zip_longest(shape, index))
)
index = tuple(map(lambda s_i: slice(0, s_i[0], 1) if s_i[1] is None else s_i[1], zip_longest(shape, index)))
self._shape = shape
self._index = index
@@ -234,20 +231,20 @@ class MatrixProxyView(MatrixProxy):
NOTE: these follow the numpy rules for dimensionality reduction
when an integer index is specified.
"""
def _getitem_intXint(self, row, col):
return self.m[row, col]
def _getitem_intXslice(self, row, col):
shape = (_slice_length(col, self.m.shape[1]), )
shape = (_slice_length(col, self.m.shape[1]),)
return self.__class__.create_array(self.m, shape=shape, index=(row, col))
def _getitem_sliceXint(self, row, col):
shape = (_slice_length(row, self.m.shape[0]), )
shape = (_slice_length(row, self.m.shape[0]),)
return self.__class__.create_array(self.m, shape=shape, index=(row, col))
def _getitem_sliceXslice(self, row, col):
shape = (_slice_length(row, self.m.shape[0]),
_slice_length(col, self.m.shape[1]))
shape = (_slice_length(row, self.m.shape[0]), _slice_length(col, self.m.shape[1]))
return self.__class__(self.m, shape=shape, index=(row, col), transposed=self.transposed)
def toarray(self):
@@ -261,22 +258,23 @@ class ArrayProxyView(_ArrayProxyBase):
"""
1D array view to a 2D matrix
"""
def __init__(self, arg1, shape=None, index=None, copy=False):
super().__init__()
if not copy:
m = arg1
# one index MUST be an integer and the other MUST be a slice
assert(len(index) == 2)
assert(all(isinstance(idx, INT_TYPES + (slice, )) for idx in index))
assert(isinstance(index[0], INT_TYPES) != isinstance(index[1], INT_TYPES))
assert len(index) == 2
assert all(isinstance(idx, INT_TYPES + (slice,)) for idx in index)
assert isinstance(index[0], INT_TYPES) != isinstance(index[1], INT_TYPES)
if shape is None:
if isinstance(index[0], INT_TYPES):
shape = (m.shape[0], )
shape = (m.shape[0],)
else:
shape = (m.shape[1], )
assert(len(shape) == 1)
shape = (m.shape[1],)
assert len(shape) == 1
self._shape = shape
self.m = m
@@ -336,7 +334,7 @@ class ArrayProxyView(_ArrayProxyBase):
elif isinstance(col, slice):
return self._getitem_intXslice(row, col)
elif isinstance(row, slice):
assert(isinstance(col, INT_TYPES))
assert isinstance(col, INT_TYPES)
return self._getitem_sliceXint(row, col)
raise IndexError("unsupported column index types")
@@ -345,11 +343,11 @@ class ArrayProxyView(_ArrayProxyBase):
return self.m[row, col]
def _getitem_intXslice(self, row, col):
shape = (_slice_length(col, self.m.shape[1]), )
shape = (_slice_length(col, self.m.shape[1]),)
return self.__class__(self.m, shape=shape, index=(row, col))
def _getitem_sliceXint(self, row, col):
shape = (_slice_length(row, self.m.shape[0]), )
shape = (_slice_length(row, self.m.shape[0]),)
return self.__class__(self.m, shape=shape, index=(row, col))
def toarray(self):
@@ -358,7 +356,7 @@ class ArrayProxyView(_ArrayProxyBase):
def _unpack_index(index, shape):
if not isinstance(index, tuple):
index = (index, )
index = (index,)
if len(shape) < len(index):
raise IndexError("invalid index dimensionality - must be 2")
@@ -366,7 +364,7 @@ def _unpack_index(index, shape):
for shp, idx in zip_longest(shape, index):
idx = slice(None) if idx is None else idx
idx = _slice_defaults(idx, shp) if isinstance(idx, slice) else idx
unpacked += (idx, )
unpacked += (idx,)
return unpacked
@@ -376,7 +374,7 @@ def _slice_slice(outer, outer_len, inner, inner_len):
slice a slice - we take advantage of Python 3 range's support
for indexing.
"""
assert(outer_len >= inner_len)
assert outer_len >= inner_len
outer_rng = range(*outer.indices(outer_len))
rng = outer_rng[inner]
start, stop, step = rng.start, rng.stop, rng.step
@@ -387,8 +385,8 @@ def _slice_slice(outer, outer_len, inner, inner_len):
def _range_length(start, stop, step):
""" return length of range """
assert(step != 0)
assert(start is not None and stop is not None and step is not None)
assert step != 0
assert start is not None and stop is not None and step is not None
if step > 0 and start < stop:
return 1 + (stop - 1 - start) // step
elif step < 0 and start > stop:
@@ -404,7 +402,7 @@ def _slice_length(s, length):
def _slice_defaults(s, length):
""" apply slice defaulting conventions """
assert(length >= 0)
assert length >= 0
step = 1 if s.step is None else s.step
+1
View File
@@ -40,4 +40,5 @@ def requires_data(func):
if self.data is None:
raise DriverError(f"error data must be loaded before you call {func.__name__}")
return func(self, *args, **kwargs)
return wrapped_function