mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-02 14:18:12 +08:00
server refactor (#1140)
This PR contains a refactoring to make adding new features easier. The new features include supporting the tiledb format, and the multi dataset application. The refactoring includes Simplifying the directory structure and files. a class structure to handle annotations (currently one type: AnnotationsLocalFile). a class to handle application configuration a class structure to handle matrix data (currently AnndataAdaptor and CxgAdaptor). CxgAdaptor uses tiledb. Algorithms that were previously dependent on the scanpy anndata object are now generalized to work with an abstract interface. The multi dataset option is not fully supported yet, and so the option to use it is hidden. Use "cli launch --dataroot ..." To access this feature. All combinations of app single dataset/ app multi dataset and AnndataAdaptor/CxgAdaptor work with all the features, such as annotations, ontologies, diffexp.
This commit is contained in:
+201
-18
@@ -1,26 +1,196 @@
|
||||
import os
|
||||
import datetime
|
||||
|
||||
from flask import Flask
|
||||
from flask import Flask, redirect, current_app, make_response, render_template
|
||||
from flask import Blueprint, request, send_from_directory
|
||||
from flask_caching import Cache
|
||||
from flask_compress import Compress
|
||||
from flask_cors import CORS
|
||||
from flask_restful import Api, Resource
|
||||
|
||||
from server.app.rest_api.rest import get_api_resources
|
||||
from server.app.util.utils import Float32JSONEncoder
|
||||
from server.app.web import webapp
|
||||
from http import HTTPStatus
|
||||
|
||||
import server.common.rest as common_rest
|
||||
from server.common.errors import DatasetAccessError
|
||||
from server.common.utils import path_join, Float32JSONEncoder
|
||||
from server.common.data_locator import DataLocator
|
||||
from server.data_common.matrix_loader import MatrixDataLoader, MatrixDataType
|
||||
|
||||
from functools import wraps
|
||||
|
||||
webbp = Blueprint("webapp", "server.common.web", template_folder="templates")
|
||||
|
||||
|
||||
@webbp.route("/")
|
||||
def dataset_index(dataset=None):
|
||||
config = current_app.app_config
|
||||
if dataset is None:
|
||||
if config.datapath:
|
||||
location = config.datapath
|
||||
else:
|
||||
return dataroot_index()
|
||||
else:
|
||||
location = path_join(config.dataroot, dataset)
|
||||
|
||||
scripts = config.scripts
|
||||
|
||||
try:
|
||||
cache_manager = current_app.matrix_data_cache_manager
|
||||
with cache_manager.data_adaptor(location, config) as data_adaptor:
|
||||
dataset_title = config.get_title(data_adaptor)
|
||||
return render_template("index.html", datasetTitle=dataset_title, SCRIPTS=scripts)
|
||||
except DatasetAccessError as e:
|
||||
return make_response(f"Invalid dataset {dataset}: {str(e)}", HTTPStatus.BAD_REQUEST)
|
||||
|
||||
|
||||
@webbp.route("/favicon.png")
|
||||
def favicon():
|
||||
return send_from_directory(os.path.join(webbp.root_path, "static/img/"), "favicon.png")
|
||||
|
||||
|
||||
def get_data_adaptor(dataset=None):
|
||||
config = current_app.app_config
|
||||
|
||||
if dataset is None:
|
||||
datapath = config.datapath
|
||||
else:
|
||||
datapath = path_join(config.dataroot, dataset)
|
||||
# path_join returns a normalized path. Therefore it is
|
||||
# sufficient to check that the datapath starts with the
|
||||
# dataroot to determine that the datapath is under the dataroot.
|
||||
if not datapath.startswith(config.dataroot):
|
||||
raise DatasetAccessError("Invalid dataset {dataset}")
|
||||
|
||||
if datapath is None:
|
||||
return make_response("Dataset must be supplied", HTTPStatus.BAD_REQUEST)
|
||||
|
||||
cache_manager = current_app.matrix_data_cache_manager
|
||||
return cache_manager.data_adaptor(datapath, config)
|
||||
|
||||
|
||||
def rest_get_data_adaptor(func):
|
||||
@wraps(func)
|
||||
def wrapped_function(self, dataset=None):
|
||||
try:
|
||||
with get_data_adaptor(dataset) as data_adaptor:
|
||||
return func(self, data_adaptor)
|
||||
except DatasetAccessError as e:
|
||||
return make_response(f"Invalid dataset {dataset}: {str(e)}", HTTPStatus.BAD_REQUEST)
|
||||
|
||||
return wrapped_function
|
||||
|
||||
|
||||
def static_redirect(dataset, therest):
|
||||
""" redirect all static requests to the standard location """
|
||||
return redirect(f'/static/{therest}', code=301)
|
||||
|
||||
|
||||
def favicon_redirect(dataset):
|
||||
""" redirect favicon to static dir """
|
||||
return redirect('/static/favicon.png', code=301)
|
||||
|
||||
|
||||
def dataroot_index():
|
||||
# FIXME with a splash screen that includes a listing of all the datasets.
|
||||
# or perhaps a login screen if this is a hosted environment
|
||||
data = "<H1>Welcome to cellxgene</H1>"
|
||||
|
||||
# the following is just for demo purposes...
|
||||
try:
|
||||
config = current_app.app_config
|
||||
locator = DataLocator(config.dataroot)
|
||||
datasets = []
|
||||
for fname in locator.ls():
|
||||
location = path_join(config.dataroot, fname)
|
||||
matrix_data_loader = MatrixDataLoader(location)
|
||||
if matrix_data_loader.etype != MatrixDataType.UNKNOWN:
|
||||
datasets.append(fname)
|
||||
|
||||
data += "<br/>Select one of these datasets...<br/>"
|
||||
data += "<ul>"
|
||||
datasets.sort()
|
||||
for dataset in datasets:
|
||||
data += f"<li><a href={dataset}>{dataset}</a></li>"
|
||||
data += "</ul>"
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return make_response(data)
|
||||
|
||||
|
||||
class SchemaAPI(Resource):
|
||||
@rest_get_data_adaptor
|
||||
def get(self, data_adaptor):
|
||||
return common_rest.schema_get(data_adaptor, current_app.annotations)
|
||||
|
||||
|
||||
class ConfigAPI(Resource):
|
||||
@rest_get_data_adaptor
|
||||
def get(self, data_adaptor):
|
||||
return common_rest.config_get(
|
||||
current_app.app_config, data_adaptor, current_app.annotations)
|
||||
|
||||
|
||||
class AnnotationsObsAPI(Resource):
|
||||
@rest_get_data_adaptor
|
||||
def get(self, data_adaptor):
|
||||
return common_rest.annotations_obs_get(
|
||||
request, data_adaptor, current_app.annotations)
|
||||
|
||||
@rest_get_data_adaptor
|
||||
def put(self, data_adaptor):
|
||||
return common_rest.annotations_obs_put(
|
||||
request, data_adaptor, current_app.annotations)
|
||||
|
||||
|
||||
class AnnotationsVarAPI(Resource):
|
||||
@rest_get_data_adaptor
|
||||
def get(self, data_adaptor):
|
||||
return common_rest.annotations_var_get(request, data_adaptor, current_app.annotations)
|
||||
|
||||
|
||||
class DataVarAPI(Resource):
|
||||
@rest_get_data_adaptor
|
||||
def put(self, data_adaptor):
|
||||
return common_rest.data_var_put(request, data_adaptor)
|
||||
|
||||
|
||||
class DiffExpObsAPI(Resource):
|
||||
@rest_get_data_adaptor
|
||||
def post(self, data_adaptor):
|
||||
return common_rest.diffexp_obs_post(request, data_adaptor)
|
||||
|
||||
|
||||
class LayoutObsAPI(Resource):
|
||||
@rest_get_data_adaptor
|
||||
def get(self, data_adaptor):
|
||||
return common_rest.layout_obs_get(request, data_adaptor)
|
||||
|
||||
|
||||
def get_api_resources(bp_api):
|
||||
api = Api(bp_api)
|
||||
# Initialization routes
|
||||
api.add_resource(SchemaAPI, "/schema")
|
||||
api.add_resource(ConfigAPI, "/config")
|
||||
# Data routes
|
||||
api.add_resource(AnnotationsObsAPI, "/annotations/obs")
|
||||
api.add_resource(AnnotationsVarAPI, "/annotations/var")
|
||||
api.add_resource(DataVarAPI, "/data/var")
|
||||
# Computation routes
|
||||
api.add_resource(DiffExpObsAPI, "/diffexp/obs")
|
||||
api.add_resource(LayoutObsAPI, "/layout/obs")
|
||||
return api
|
||||
|
||||
|
||||
class Server:
|
||||
def __init__(self):
|
||||
self.data = None
|
||||
self.cache = Cache(config={"CACHE_TYPE": "simple", "CACHE_DEFAULT_TIMEOUT": 860_000})
|
||||
self.app = None
|
||||
def __init__(self, matrix_data_cache_manager, annotations, app_config):
|
||||
|
||||
def create_app(self):
|
||||
self.app = Flask(__name__, static_folder="web/static")
|
||||
self.app = Flask(__name__, static_folder="../common/web/static")
|
||||
self.app.json_encoder = Float32JSONEncoder
|
||||
|
||||
self.cache = Cache(config={"CACHE_TYPE": "simple", "CACHE_DEFAULT_TIMEOUT": 860_000})
|
||||
self.cache.init_app(self.app)
|
||||
|
||||
Compress(self.app)
|
||||
CORS(self.app, supports_credentials=True)
|
||||
|
||||
@@ -30,13 +200,26 @@ class Server:
|
||||
# Config
|
||||
SECRET_KEY = os.environ.get("CXG_SECRET_KEY", default="SparkleAndShine")
|
||||
self.app.config.update(SECRET_KEY=SECRET_KEY)
|
||||
self.app.config.update(SCRIPTS=[])
|
||||
|
||||
resources = get_api_resources()
|
||||
self.app.register_blueprint(webapp.bp)
|
||||
self.app.register_blueprint(resources.blueprint)
|
||||
self.app.add_url_rule("/", endpoint="index")
|
||||
self.app.register_blueprint(webbp)
|
||||
|
||||
def attach_data(self, data, title="Demo", about=""):
|
||||
self.app.config.update(DATASET_TITLE=title, ABOUT_DATASET=about)
|
||||
self.app.data = data
|
||||
api_version = "/api/v0.2"
|
||||
if app_config.datapath:
|
||||
bp_api = Blueprint("api", __name__, url_prefix=api_version)
|
||||
resources = get_api_resources(bp_api)
|
||||
self.app.register_blueprint(resources.blueprint)
|
||||
|
||||
else:
|
||||
# NOTE: These routes only allow the dataset to be in the directory
|
||||
# of the dataroot, and not a subdirectory. We may want to change
|
||||
# the route format at some point
|
||||
bp_api = Blueprint("api_dataset", __name__, url_prefix="/<dataset>" + api_version)
|
||||
resources = get_api_resources(bp_api)
|
||||
self.app.register_blueprint(resources.blueprint)
|
||||
self.app.add_url_rule("/<dataset>/", 'dataset_index', dataset_index)
|
||||
self.app.add_url_rule("/<dataset>/static/<path:therest>", "static_redirect", static_redirect)
|
||||
self.app.add_url_rule("/<dataset>/favicon.png", "favicon_redirect", favicon_redirect)
|
||||
|
||||
self.app.matrix_data_cache_manager = matrix_data_cache_manager
|
||||
self.app.annotations = annotations
|
||||
self.app.app_config = app_config
|
||||
|
||||
@@ -1,113 +0,0 @@
|
||||
from abc import ABCMeta, abstractmethod
|
||||
|
||||
"""
|
||||
Sort order for methods
|
||||
1. Initialize
|
||||
2. Helper
|
||||
3. Filter
|
||||
4. Data & Metadata
|
||||
5. Computation
|
||||
"""
|
||||
|
||||
|
||||
class CXGDriver(metaclass=ABCMeta):
|
||||
def __init__(self, data_locator=None, args={}):
|
||||
self.config = self._get_default_config()
|
||||
self.config.update(args)
|
||||
if data_locator:
|
||||
self._load_data(data_locator)
|
||||
self.data_locator = data_locator
|
||||
else:
|
||||
self.data = None
|
||||
|
||||
def update(self, data_locator=None, args={}):
|
||||
self.config.update(args)
|
||||
if data_locator:
|
||||
self._load_data(data_locator)
|
||||
self.data_locator = data_locator
|
||||
|
||||
@staticmethod
|
||||
def _get_default_config():
|
||||
return {
|
||||
"layout": None,
|
||||
"max_category_items": None,
|
||||
"diffexp_lfc_cutoff": None,
|
||||
"disable_diffexp": False,
|
||||
"diffexp_may_be_slow": False,
|
||||
}
|
||||
|
||||
@abstractmethod
|
||||
def get_config_parameters(self, uid=None):
|
||||
"""
|
||||
return a dict of properties that will be used to set the engine-specific
|
||||
"parameters" info for client-side configuration.
|
||||
|
||||
See rest.py /config route for use
|
||||
"""
|
||||
pass
|
||||
|
||||
@property
|
||||
def features(self):
|
||||
features = {
|
||||
"cluster": {"available": False},
|
||||
"layout": {"obs": {"available": False}, "var": {"available": False}},
|
||||
"diffexp": {"available": True, "interactiveLimit": 50000},
|
||||
}
|
||||
# TODO - Interactive limit should be generated from the actual available methods see GH issue #94
|
||||
if self.config["layout"]:
|
||||
# TODO handle "var" when gene layout becomes available
|
||||
features["layout"]["obs"] = {"available": True, "interactiveLimit": 50000}
|
||||
return features
|
||||
|
||||
@abstractmethod
|
||||
def get_schema(self):
|
||||
"""
|
||||
Return current schema
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def _load_data(self, data_locator):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def annotation_to_fbs_matrix(self, axis, field=None, uid=None):
|
||||
"""
|
||||
Gets annotation value for each observation
|
||||
:param axis: string obs or var
|
||||
:param fields: list of keys for annotation to return, returns all annotation values if not set.
|
||||
:return: flatbuffer: in fbs/matrix.fbs encoding
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def annotation_put_fbs(self, axis, fbs, uid=None):
|
||||
"""
|
||||
Put/save FBS as user-defined labels
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def data_frame_to_fbs_matrix(self, filter, axis):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def diffexp_topN(self, obsFilter1, obsFilter2, top_n=None, interactive_limit=None):
|
||||
"""
|
||||
Computes the top N differentially expressed variables between two observation sets. If mode
|
||||
is "TOP_N", then stats for the top N
|
||||
dataframes
|
||||
contain a subset of variables, then statistics for all variables will be returned, otherwise
|
||||
only the top N vars will be returned.
|
||||
:param obsFilter1: filter: dictionary with filter params for first set of observations
|
||||
:param obsFilter2: filter: dictionary with filter params for second set of observations
|
||||
:param top_n: Limit results to top N (Top var mode only)
|
||||
:param interactive_limit: -- don't compute if total # genes in dataframes are larger than this
|
||||
:return: top N genes and corresponding stats
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def layout_to_fbs_matrix(self, filter):
|
||||
""" same as layout, except returns a flatbuffer """
|
||||
pass
|
||||
@@ -1,240 +0,0 @@
|
||||
from http import HTTPStatus
|
||||
import warnings
|
||||
from uuid import uuid4
|
||||
import re
|
||||
|
||||
from flask import Blueprint, current_app, jsonify, make_response, request, session
|
||||
from flask_restful import Api, Resource
|
||||
from server import __version__ as cellxgene_version
|
||||
from anndata import __version__ as anndata_version
|
||||
|
||||
from server.app.util.constants import Axis, DiffExpMode, JSON_NaN_to_num_warning_msg, CXGUID, CXG_ANNO_COLLECTION
|
||||
from server.app.util.errors import (
|
||||
FilterError,
|
||||
InteractiveError,
|
||||
JSONEncodingValueError,
|
||||
PrepareError,
|
||||
DisabledFeatureError,
|
||||
)
|
||||
|
||||
|
||||
class SchemaAPI(Resource):
|
||||
def get(self):
|
||||
cxguid = get_userid(session)
|
||||
anno_collection = get_anno_collection(session)
|
||||
return make_response(
|
||||
jsonify({"schema": current_app.data.get_schema(uid=cxguid, collection=anno_collection)}), HTTPStatus.OK
|
||||
)
|
||||
|
||||
|
||||
class ConfigAPI(Resource):
|
||||
def get(self):
|
||||
cxguid = get_userid(session)
|
||||
anno_collection = get_anno_collection(session)
|
||||
config = {
|
||||
"config": {
|
||||
"features": [
|
||||
{"method": "POST", "path": "/cluster/", **current_app.data.features["cluster"]},
|
||||
{"method": "POST", "path": "/layout/obs", **current_app.data.features["layout"]["obs"]},
|
||||
{"method": "POST", "path": "/layout/var", **current_app.data.features["layout"]["var"]},
|
||||
{"method": "POST", "path": "/diffexp/", **current_app.data.features["diffexp"]},
|
||||
],
|
||||
"displayNames": {
|
||||
"engine": f"cellxgene Scanpy engine version ",
|
||||
"dataset": current_app.config["DATASET_TITLE"],
|
||||
},
|
||||
"links": {"about-dataset": current_app.config["ABOUT_DATASET"]},
|
||||
"parameters": {**current_app.data.get_config_parameters(uid=cxguid, collection=anno_collection)},
|
||||
"library_versions": {"cellxgene": cellxgene_version, "anndata": str(anndata_version)},
|
||||
}
|
||||
}
|
||||
|
||||
return make_response(jsonify(config), HTTPStatus.OK)
|
||||
|
||||
|
||||
class AnnotationsObsAPI(Resource):
|
||||
def get(self):
|
||||
fields = request.args.getlist("annotation-name", None)
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
cxguid = get_userid(session)
|
||||
anno_collection = get_anno_collection(session)
|
||||
try:
|
||||
if preferred_mimetype == "application/octet-stream":
|
||||
fbs = current_app.data.annotation_to_fbs_matrix("obs", fields, uid=cxguid, collection=anno_collection)
|
||||
return make_response(fbs, HTTPStatus.OK, {"Content-Type": "application/octet-stream"})
|
||||
else:
|
||||
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
|
||||
except KeyError:
|
||||
return make_response(f"Error bad key in {fields}", HTTPStatus.BAD_REQUEST)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
def put(self):
|
||||
cxguid = get_userid(session)
|
||||
anno_collection = request.args.get("annotation-collection-name", default=None)
|
||||
if anno_collection is not None:
|
||||
if not is_safe_collection_name(anno_collection):
|
||||
return make_response(f"Error, bad annotation collection name", HTTPStatus.BAD_REQUEST)
|
||||
set_anno_collection(session, anno_collection)
|
||||
else:
|
||||
anno_collection = get_anno_collection(session)
|
||||
|
||||
try:
|
||||
fbs = request.get_data()
|
||||
res = current_app.data.annotation_put_fbs("obs", fbs, uid=cxguid, collection=anno_collection)
|
||||
return make_response(res, HTTPStatus.OK, {"Content-Type": "application/json"})
|
||||
except (ValueError, DisabledFeatureError, KeyError) as e:
|
||||
return make_response(str(e), HTTPStatus.BAD_REQUEST)
|
||||
except Exception as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
class AnnotationsVarAPI(Resource):
|
||||
def get(self):
|
||||
fields = request.args.getlist("annotation-name", None)
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
try:
|
||||
if preferred_mimetype == "application/octet-stream":
|
||||
return make_response(
|
||||
current_app.data.annotation_to_fbs_matrix("var", fields),
|
||||
HTTPStatus.OK,
|
||||
{"Content-Type": "application/octet-stream"},
|
||||
)
|
||||
else:
|
||||
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
|
||||
except KeyError:
|
||||
return make_response(f"Error bad key in {fields}", HTTPStatus.BAD_REQUEST)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
class DataVarAPI(Resource):
|
||||
def put(self):
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
try:
|
||||
if preferred_mimetype == "application/octet-stream":
|
||||
filter_json = request.get_json()
|
||||
filter = filter_json["filter"] if filter_json else None
|
||||
return make_response(
|
||||
current_app.data.data_frame_to_fbs_matrix(filter, axis=Axis.VAR),
|
||||
HTTPStatus.OK,
|
||||
{"Content-Type": "application/octet-stream"},
|
||||
)
|
||||
else:
|
||||
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
|
||||
except FilterError as e:
|
||||
return make_response(e.message, HTTPStatus.BAD_REQUEST)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
class DiffExpObsAPI(Resource):
|
||||
def post(self):
|
||||
args = request.get_json()
|
||||
# confirm mode is present and legal
|
||||
try:
|
||||
mode = DiffExpMode(args["mode"])
|
||||
except KeyError:
|
||||
return make_response("Error: mode is required", HTTPStatus.BAD_REQUEST)
|
||||
except ValueError:
|
||||
return make_response(f"Error: invalid mode option {args['mode']}", HTTPStatus.BAD_REQUEST)
|
||||
# Validate filters
|
||||
if mode == DiffExpMode.VAR_FILTER or "varFilter" in args:
|
||||
# not NOT_IMPLEMENTED
|
||||
return make_response("mode=varfilter not implemented", HTTPStatus.NOT_IMPLEMENTED)
|
||||
if mode == DiffExpMode.TOP_N and "count" not in args:
|
||||
return make_response("mode=topN requires a count parameter", HTTPStatus.BAD_REQUEST)
|
||||
|
||||
if "set1" not in args:
|
||||
return make_response("set1 is required.", HTTPStatus.BAD_REQUEST)
|
||||
if Axis.VAR in args["set1"]["filter"]:
|
||||
return make_response("Var filter not allowed for set1", HTTPStatus.BAD_REQUEST)
|
||||
# set2
|
||||
if "set2" not in args:
|
||||
return make_response("Set2 as inverse of set1 is not implemented", HTTPStatus.NOT_IMPLEMENTED)
|
||||
if Axis.VAR in args["set2"]["filter"]:
|
||||
return make_response("Var filter not allowed for set2", HTTPStatus.BAD_REQUEST)
|
||||
|
||||
set1_filter = args["set1"]["filter"]
|
||||
set2_filter = args.get("set2", {"filter": {}})["filter"]
|
||||
|
||||
# TODO: implement varfilter mode
|
||||
|
||||
# mode=topN
|
||||
count = args.get("count", None)
|
||||
try:
|
||||
diffexp = current_app.data.diffexp_topN(
|
||||
set1_filter, set2_filter, count, current_app.data.features["diffexp"]["interactiveLimit"],
|
||||
)
|
||||
return make_response(diffexp, HTTPStatus.OK, {"Content-Type": "application/json"})
|
||||
except (ValueError, FilterError) as e:
|
||||
return make_response(e.message, HTTPStatus.BAD_REQUEST)
|
||||
except InteractiveError:
|
||||
return make_response("Non-interactive request", HTTPStatus.FORBIDDEN)
|
||||
except JSONEncodingValueError as e:
|
||||
# JSON encoding failure, usually due to bad data
|
||||
warnings.warn(JSON_NaN_to_num_warning_msg)
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
class LayoutObsAPI(Resource):
|
||||
def get(self):
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
try:
|
||||
if preferred_mimetype == "application/octet-stream":
|
||||
return make_response(
|
||||
current_app.data.layout_to_fbs_matrix(), HTTPStatus.OK, {"Content-Type": "application/octet-stream"}
|
||||
)
|
||||
else:
|
||||
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
|
||||
except PrepareError as e:
|
||||
return make_response(e.message, HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
def get_userid(ss):
|
||||
if CXGUID not in ss:
|
||||
ss[CXGUID] = uuid4().hex
|
||||
ss.permanent = True
|
||||
return ss[CXGUID]
|
||||
|
||||
|
||||
def get_anno_collection(ss):
|
||||
collection = ss[CXG_ANNO_COLLECTION] if CXG_ANNO_COLLECTION in ss else None
|
||||
return collection
|
||||
|
||||
|
||||
def set_anno_collection(ss, name):
|
||||
ss[CXG_ANNO_COLLECTION] = name
|
||||
ss.permanent = True
|
||||
|
||||
|
||||
def is_safe_collection_name(name):
|
||||
"""
|
||||
return true if this is a safe collection name
|
||||
|
||||
this is ultra convervative. If we want to allow full legal file name syntax,
|
||||
we could look at modules like `pathvalidate`
|
||||
"""
|
||||
if name is None:
|
||||
return False
|
||||
return re.match(r"^[\w\-]+$", name) is not None
|
||||
|
||||
|
||||
def get_api_resources():
|
||||
bp = Blueprint("api", __name__, url_prefix="/api/v0.2")
|
||||
api = Api(bp)
|
||||
# Initialization routes
|
||||
api.add_resource(SchemaAPI, "/schema")
|
||||
api.add_resource(ConfigAPI, "/config")
|
||||
# Data routes
|
||||
api.add_resource(AnnotationsObsAPI, "/annotations/obs")
|
||||
api.add_resource(AnnotationsVarAPI, "/annotations/var")
|
||||
api.add_resource(DataVarAPI, "/data/var")
|
||||
# Computation routes
|
||||
api.add_resource(DiffExpObsAPI, "/diffexp/obs")
|
||||
api.add_resource(LayoutObsAPI, "/layout/obs")
|
||||
return api
|
||||
@@ -1,121 +0,0 @@
|
||||
import numpy as np
|
||||
from scipy import sparse, stats
|
||||
|
||||
|
||||
# Convenience function which handles sparse data
|
||||
def _mean_var_n(X):
|
||||
"""
|
||||
Two-pass variance calculation. Numerically (more) stable
|
||||
than naive methods (and same method used by numpy.var())
|
||||
https://en.wikipedia.org/wiki/Algorithms_for_calculating_variance#Two-pass
|
||||
"""
|
||||
# fp_err_occurred is a flag indicating that a floating point error
|
||||
# occured somewhere in our compute. Used to trigger non-finite
|
||||
# number handling.
|
||||
fp_err_occurred = False
|
||||
|
||||
def fp_err_set(err, flag):
|
||||
nonlocal fp_err_occurred
|
||||
fp_err_occurred = True
|
||||
|
||||
with np.errstate(divide="call", invalid="call", call=fp_err_set):
|
||||
n = X.shape[0]
|
||||
if sparse.issparse(X):
|
||||
mean = X.mean(axis=0).A1
|
||||
dfm = X - mean
|
||||
sumsq = np.sum(np.multiply(dfm, dfm), axis=0).A1
|
||||
v = sumsq / (n - 1)
|
||||
else:
|
||||
mean = X.mean(axis=0)
|
||||
dfm = X - mean
|
||||
sumsq = np.sum(np.multiply(dfm, dfm), axis=0)
|
||||
v = sumsq / (n - 1)
|
||||
|
||||
if fp_err_occurred:
|
||||
mean[np.isfinite(mean) == False] = 0 # noqa: E712
|
||||
v[np.isfinite(v) == False] = 0 # noqa: E712
|
||||
return mean, v, n
|
||||
|
||||
|
||||
def diffexp_ttest(adata, maskA, maskB, top_n=8, diffexp_lfc_cutoff=0.01):
|
||||
"""
|
||||
Return differential expression statistics for top N variables.
|
||||
|
||||
Algorithm:
|
||||
- compute log fold change (log2(meanA/meanB))
|
||||
- compute Welch's t-test statistic and pvalue (w/ Bonferroni correction)
|
||||
- return top N abs(logfoldchange) where lfc > diffexp_lfc_cutoff
|
||||
|
||||
If there are not N which meet criteria, augment by removing the logfoldchange
|
||||
threshold requirement.
|
||||
|
||||
Notes on alogrithm:
|
||||
- Welch's ttest provides basic statistics test.
|
||||
https://en.wikipedia.org/wiki/Welch%27s_t-test
|
||||
- p-values adjusted with Bonferroni correction.
|
||||
https://en.wikipedia.org/wiki/Bonferroni_correction
|
||||
|
||||
:param adata: anndata dataframe
|
||||
:param maskA: observation selection mask for set 1
|
||||
:param maskB: observation selection mask for set 2
|
||||
:param top_n: number of variables to return stats for
|
||||
:param diffexp_lfc_cutoff: minimum
|
||||
:return: for top N genes, [ varindex, logfoldchange, pval, pval_adj ]
|
||||
"""
|
||||
if top_n > adata.n_obs:
|
||||
top_n = adata.n_obs
|
||||
|
||||
# mean, variance, N - calculate for both selections
|
||||
meanA, vA, nA = _mean_var_n(adata.X[maskA, :])
|
||||
meanB, vB, nB = _mean_var_n(adata.X[maskB, :])
|
||||
|
||||
# variance / N
|
||||
vnA = vA / min(nA, nB) # overestimate variance, would normally be nA
|
||||
vnB = vB / min(nA, nB) # overestimate variance, would normally be nB
|
||||
sum_vn = vnA + vnB
|
||||
|
||||
# degrees of freedom for Welch's t-test
|
||||
with np.errstate(divide="ignore", invalid="ignore"):
|
||||
dof = sum_vn ** 2 / (vnA ** 2 / (nA - 1) + vnB ** 2 / (nB - 1))
|
||||
dof[np.isnan(dof)] = 1
|
||||
|
||||
# Welch's t-test score calculation
|
||||
with np.errstate(divide="ignore", invalid="ignore"):
|
||||
tscores = (meanA - meanB) / np.sqrt(sum_vn)
|
||||
tscores[np.isnan(tscores)] = 0
|
||||
|
||||
# p-value
|
||||
pvals = stats.t.sf(np.abs(tscores), dof) * 2
|
||||
pvals_adj = pvals * adata.X.shape[1]
|
||||
pvals_adj[pvals_adj > 1] = 1 # cap adjusted p-value at 1
|
||||
|
||||
# logfoldchanges: log2(meanA / meanB)
|
||||
logfoldchanges = np.log2(np.abs((meanA + 1e-9) / (meanB + 1e-9)))
|
||||
|
||||
# find all with lfc > cutoff
|
||||
lfc_above_cutoff_idx = np.nonzero(np.abs(logfoldchanges) > diffexp_lfc_cutoff)[0]
|
||||
stats_to_sort = np.abs(tscores)
|
||||
|
||||
# derive sort order
|
||||
if lfc_above_cutoff_idx.shape[0] > top_n:
|
||||
# partition top N
|
||||
rel_t_partition = np.argpartition(stats_to_sort[lfc_above_cutoff_idx], -top_n)[-top_n:]
|
||||
t_partition = lfc_above_cutoff_idx[rel_t_partition]
|
||||
# sort the top N partition
|
||||
rel_sort_order = np.argsort(stats_to_sort[t_partition])[::-1]
|
||||
sort_order = t_partition[rel_sort_order]
|
||||
else:
|
||||
# partition and sort top N, ignoring lfc cutoff
|
||||
partition = np.argpartition(stats_to_sort, -top_n)[-top_n:]
|
||||
rel_sort_order = np.argsort(stats_to_sort[partition])[::-1]
|
||||
indices = np.indices(stats_to_sort.shape)[0]
|
||||
sort_order = indices[partition][rel_sort_order]
|
||||
|
||||
# top n slice based upon sort order
|
||||
logfoldchanges_top_n = logfoldchanges[sort_order]
|
||||
pvals_top_n = pvals[sort_order]
|
||||
pvals_adj_top_n = pvals_adj[sort_order]
|
||||
|
||||
# varIndex, logfoldchange, pval, pval_adj
|
||||
result = [[sort_order[i], logfoldchanges_top_n[i], pvals_top_n[i], pvals_adj_top_n[i]] for i in range(top_n)]
|
||||
return result
|
||||
@@ -1,60 +0,0 @@
|
||||
"""
|
||||
Helpers for user annotations
|
||||
"""
|
||||
import os
|
||||
import os.path
|
||||
from datetime import datetime
|
||||
import pandas as pd
|
||||
|
||||
|
||||
def read_labels(fname):
|
||||
if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
|
||||
return pd.read_csv(fname, dtype="category", index_col=0, header=0, comment="#", keep_default_na=False)
|
||||
else:
|
||||
return pd.DataFrame()
|
||||
|
||||
|
||||
def write_labels(fname, df, header=None, backup_dir=None):
|
||||
if backup_dir is not None:
|
||||
backup(fname, backup_dir)
|
||||
if not df.empty:
|
||||
with open(fname, "w", newline="") as f:
|
||||
if header is not None:
|
||||
f.write(header)
|
||||
df.to_csv(f)
|
||||
else:
|
||||
open(fname, "w").close()
|
||||
|
||||
|
||||
def backup(fname, backup_dir, max_backups=9):
|
||||
"""
|
||||
save N backups of file to backup_dir.
|
||||
1. fname -> backup_dir/fname-TIME
|
||||
2. delete excess files in backup_dir
|
||||
"""
|
||||
|
||||
# Make sure there is work to do
|
||||
if not os.path.exists(fname):
|
||||
return
|
||||
|
||||
# Ensure backup_dir exists
|
||||
if not os.path.exists(backup_dir):
|
||||
os.mkdir(backup_dir)
|
||||
|
||||
# Save current file to backup_dir
|
||||
fname_base = os.path.basename(fname)
|
||||
fname_base_root, fname_base_ext = os.path.splitext(fname_base)
|
||||
# don't use ISO standard time format, as it contains characters illegal on some filesytems.
|
||||
nowish = datetime.now().strftime("%Y-%m-%dT%H-%M-%S")
|
||||
backup_fname = os.path.join(backup_dir, f"{fname_base_root}-{nowish}{fname_base_ext}")
|
||||
if os.path.exists(backup_fname):
|
||||
os.remove(backup_fname)
|
||||
os.rename(fname, backup_fname)
|
||||
|
||||
# prune the backup_dir to max number of backup files, keeping the most recent backups
|
||||
backups = list(filter(lambda s: s.startswith(fname_base_root), os.listdir(backup_dir)))
|
||||
excess_count = len(backups) - max_backups
|
||||
if excess_count > 0:
|
||||
backups.sort()
|
||||
for bu in backups[0:excess_count]:
|
||||
os.remove(os.path.join(backup_dir, bu))
|
||||
@@ -1,658 +0,0 @@
|
||||
import warnings
|
||||
import copy
|
||||
import threading
|
||||
from datetime import datetime
|
||||
import os.path
|
||||
from hashlib import blake2b
|
||||
import base64
|
||||
from packaging import version
|
||||
|
||||
import numpy as np
|
||||
import pandas
|
||||
from pandas.core.dtypes.dtypes import CategoricalDtype
|
||||
import anndata
|
||||
from scipy import sparse
|
||||
|
||||
from server import __version__ as cellxgene_version
|
||||
from server.app.driver.driver import CXGDriver
|
||||
from server.app.util.constants import Axis, DEFAULT_TOP_N, MAX_LAYOUTS
|
||||
from server.app.util.errors import (
|
||||
FilterError,
|
||||
JSONEncodingValueError,
|
||||
PrepareError,
|
||||
ScanpyFileError,
|
||||
DisabledFeatureError,
|
||||
)
|
||||
from server.app.util.utils import jsonify_scanpy, requires_data
|
||||
from server.app.scanpy_engine.diffexp import diffexp_ttest
|
||||
from server.app.util.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
|
||||
from server.app.scanpy_engine.labels import read_labels, write_labels
|
||||
|
||||
|
||||
anndata_version = version.parse(str(anndata.__version__)).release
|
||||
|
||||
|
||||
def anndata_version_is_pre_070():
|
||||
major = anndata_version[0]
|
||||
minor = anndata_version[1] if len(anndata_version) > 1 else 0
|
||||
return major == 0 and minor < 7
|
||||
|
||||
|
||||
def has_method(o, name):
|
||||
""" return True if `o` has callable method `name` """
|
||||
op = getattr(o, name, None)
|
||||
return op is not None and callable(op)
|
||||
|
||||
|
||||
class ScanpyEngine(CXGDriver):
|
||||
def __init__(self, data_locator=None, args={}):
|
||||
super().__init__(data_locator, args)
|
||||
# lock used to protect label file write ops
|
||||
self.label_lock = threading.RLock()
|
||||
if self.data:
|
||||
self._validate_and_initialize()
|
||||
|
||||
def update(self, data_locator=None, args={}):
|
||||
super().__init__(data_locator, args)
|
||||
if self.data:
|
||||
self._validate_and_initialize()
|
||||
|
||||
@staticmethod
|
||||
def _get_default_config():
|
||||
return {
|
||||
"layout": [],
|
||||
"max_category_items": 100,
|
||||
"obs_names": None,
|
||||
"var_names": None,
|
||||
"diffexp_lfc_cutoff": 0.01,
|
||||
"annotations": False,
|
||||
"annotations_file": None,
|
||||
"annotations_output_dir": None,
|
||||
"annotations_cell_ontology_enabled": False,
|
||||
"annotations_cell_ontology_obopath": None,
|
||||
"annotations_cell_ontology_terms": None,
|
||||
"backed": False,
|
||||
"disable_diffexp": False,
|
||||
"diffexp_may_be_slow": False,
|
||||
}
|
||||
|
||||
def get_config_parameters(self, uid=None, collection=None):
|
||||
params = {
|
||||
"max-category-items": self.config["max_category_items"],
|
||||
"disable-diffexp": self.config["disable_diffexp"],
|
||||
"diffexp-may-be-slow": self.config["diffexp_may_be_slow"],
|
||||
"annotations": self.config["annotations"],
|
||||
"annotations_cell_ontology_enabled": self.config["annotations_cell_ontology_enabled"],
|
||||
"annotations_cell_ontology_terms": self.config["annotations_cell_ontology_terms"],
|
||||
}
|
||||
if self.config["annotations"]:
|
||||
if uid is not None:
|
||||
params.update({"annotations-user-data-idhash": self.get_userdata_idhash(uid)})
|
||||
if self.config["annotations_file"] is not None:
|
||||
# user has hard-wired the name of the annotation data collection
|
||||
fname = os.path.basename(self.config["annotations_file"])
|
||||
collection_fname = os.path.splitext(fname)[0]
|
||||
params.update(
|
||||
{
|
||||
"annotations-data-collection-is-read-only": True,
|
||||
"annotations-data-collection-name": collection_fname,
|
||||
}
|
||||
)
|
||||
elif collection is not None:
|
||||
params.update(
|
||||
{"annotations-data-collection-is-read-only": False, "annotations-data-collection-name": collection}
|
||||
)
|
||||
return params
|
||||
|
||||
@staticmethod
|
||||
def _create_unique_column_name(df, col_name_prefix):
|
||||
""" given the columns of a dataframe, and a name prefix, return a column name which
|
||||
does not exist in the dataframe, AND which is prefixed by `prefix`
|
||||
|
||||
The approach is to append a numeric suffix, starting at zero and increasing by
|
||||
one, until an unused name is found (eg, prefix_0, prefix_1, ...).
|
||||
"""
|
||||
suffix = 0
|
||||
while f"{col_name_prefix}{suffix}" in df:
|
||||
suffix += 1
|
||||
return f"{col_name_prefix}{suffix}"
|
||||
|
||||
def _alias_annotation_names(self):
|
||||
"""
|
||||
The front-end relies on the existance of a unique, human-readable
|
||||
index for obs & var (eg, var is typically gene name, obs the cell name).
|
||||
The user can specify these via the --obs-names and --var-names config.
|
||||
If they are not specified, use the existing index to create them, giving
|
||||
the resulting column a unique name (eg, "name").
|
||||
|
||||
In both cases, enforce that the result is unique, and communicate the
|
||||
index column name to the front-end via the obs_names and var_names config
|
||||
(which is incorporated into the schema).
|
||||
"""
|
||||
self.original_obs_index = self.data.obs.index
|
||||
|
||||
for (ax_name, config_name) in ((Axis.OBS, "obs_names"), (Axis.VAR, "var_names")):
|
||||
name = self.config[config_name]
|
||||
df_axis = getattr(self.data, str(ax_name))
|
||||
if name is None:
|
||||
# Default: create unique names from index
|
||||
if not df_axis.index.is_unique:
|
||||
raise KeyError(
|
||||
f"Values in {ax_name}.index must be unique. "
|
||||
"Please prepare data to contain unique index values, or specify an "
|
||||
"alternative with --{ax_name}-name."
|
||||
)
|
||||
name = self._create_unique_column_name(df_axis.columns, "name_")
|
||||
self.config[config_name] = name
|
||||
# reset index to simple range; alias name to point at the
|
||||
# previously specified index.
|
||||
df_axis.rename_axis(name, inplace=True)
|
||||
df_axis.reset_index(inplace=True)
|
||||
elif name in df_axis.columns:
|
||||
# User has specified alternative column for unique names, and it exists
|
||||
if not df_axis[name].is_unique:
|
||||
raise KeyError(
|
||||
f"Values in {ax_name}.{name} must be unique. " "Please prepare data to contain unique values."
|
||||
)
|
||||
df_axis.reset_index(drop=True, inplace=True)
|
||||
else:
|
||||
# user specified a non-existent column name
|
||||
raise KeyError(f"Annotation name {name}, specified in --{ax_name}-name does not exist.")
|
||||
|
||||
@staticmethod
|
||||
def _can_cast_to_float32(ann):
|
||||
if ann.dtype.kind == "f":
|
||||
if not np.can_cast(ann.dtype, np.float32):
|
||||
warnings.warn(f"Annotation {ann.name} will be converted to 32 bit float and may lose precision.")
|
||||
return True
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def _can_cast_to_int32(ann):
|
||||
if ann.dtype.kind in ["i", "u"]:
|
||||
if np.can_cast(ann.dtype, np.int32):
|
||||
return True
|
||||
ii32 = np.iinfo(np.int32)
|
||||
if ann.min() >= ii32.min and ann.max() <= ii32.max:
|
||||
return True
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def _get_col_type(col):
|
||||
dtype = col.dtype
|
||||
data_kind = dtype.kind
|
||||
schema = {}
|
||||
|
||||
if ScanpyEngine._can_cast_to_float32(col):
|
||||
schema["type"] = "float32"
|
||||
elif ScanpyEngine._can_cast_to_int32(col):
|
||||
schema["type"] = "int32"
|
||||
elif dtype == np.bool_:
|
||||
schema["type"] = "boolean"
|
||||
elif data_kind == "O" and dtype == "object":
|
||||
schema["type"] = "string"
|
||||
elif data_kind == "O" and dtype == "category":
|
||||
schema["type"] = "categorical"
|
||||
schema["categories"] = dtype.categories.tolist()
|
||||
else:
|
||||
raise TypeError(f"Annotations of type {dtype} are unsupported by cellxgene.")
|
||||
return schema
|
||||
|
||||
@requires_data
|
||||
def _create_schema(self):
|
||||
self.schema = {
|
||||
"dataframe": {"nObs": self.cell_count, "nVar": self.gene_count, "type": str(self.data.X.dtype)},
|
||||
"annotations": {
|
||||
"obs": {"index": self.config["obs_names"], "columns": []},
|
||||
"var": {"index": self.config["var_names"], "columns": []},
|
||||
},
|
||||
"layout": {"obs": []},
|
||||
}
|
||||
for ax in Axis:
|
||||
curr_axis = getattr(self.data, str(ax))
|
||||
for ann in curr_axis:
|
||||
ann_schema = {"name": ann, "writable": False}
|
||||
ann_schema.update(self._get_col_type(curr_axis[ann]))
|
||||
self.schema["annotations"][ax]["columns"].append(ann_schema)
|
||||
|
||||
for layout in self.config["layout"]:
|
||||
layout_schema = {"name": layout, "type": "float32", "dims": [f"{layout}_0", f"{layout}_1"]}
|
||||
self.schema["layout"]["obs"].append(layout_schema)
|
||||
|
||||
@requires_data
|
||||
def get_schema(self, uid=None, collection=None):
|
||||
schema = self.schema # base schema
|
||||
# add label obs annotations as needed
|
||||
labels = read_labels(self.get_anno_fname(uid, collection))
|
||||
if labels is not None and not labels.empty:
|
||||
schema = copy.deepcopy(schema)
|
||||
for col in labels.columns:
|
||||
col_schema = {
|
||||
"name": col,
|
||||
"writable": True,
|
||||
}
|
||||
col_schema.update(self._get_col_type(labels[col]))
|
||||
schema["annotations"]["obs"]["columns"].append(col_schema)
|
||||
return schema
|
||||
|
||||
def get_userdata_idhash(self, uid):
|
||||
"""
|
||||
Return a short hash that weakly identifies the user and dataset.
|
||||
Used to create safe annotations output file names.
|
||||
"""
|
||||
id = (uid + self.data_locator.abspath()).encode()
|
||||
idhash = base64.b32encode(blake2b(id, digest_size=5).digest()).decode("utf-8")
|
||||
return idhash
|
||||
|
||||
def get_anno_fname(self, uid=None, collection=None):
|
||||
""" return the current annotation file name """
|
||||
if not self.config["annotations"]:
|
||||
return None
|
||||
|
||||
if self.config["annotations_file"] is not None:
|
||||
return self.config["annotations_file"]
|
||||
|
||||
# we need to generate a file name, which we can only do if we have a UID and collection name
|
||||
if uid is None or collection is None:
|
||||
return None
|
||||
idhash = self.get_userdata_idhash(uid)
|
||||
return os.path.join(self.get_anno_output_dir(), f"{collection}-{idhash}.csv")
|
||||
|
||||
def get_anno_output_dir(self):
|
||||
""" return the current annotation output directory """
|
||||
if not self.config["annotations"]:
|
||||
return None
|
||||
|
||||
if self.config["annotations_output_dir"]:
|
||||
return self.config["annotations_output_dir"]
|
||||
|
||||
if self.config["annotations_file"]:
|
||||
return os.path.dirname(os.path.abspath(self.config["annotations_file"]))
|
||||
|
||||
return os.getcwd()
|
||||
|
||||
def get_anno_backup_dir(self, uid, collection=None):
|
||||
""" return the current annotation backup directory """
|
||||
if not self.config["annotations"]:
|
||||
return None
|
||||
|
||||
fname = self.get_anno_fname(uid, collection)
|
||||
root, ext = os.path.splitext(fname)
|
||||
return f"{root}-backups"
|
||||
|
||||
def _load_data(self, data_locator):
|
||||
# as of AnnData 0.6.19, backed mode performs initial load fast, but at the
|
||||
# cost of significantly slower access to X data.
|
||||
try:
|
||||
# there is no guarantee data_locator indicates a local file. The AnnData
|
||||
# API will only consume local file objects. If we get a non-local object,
|
||||
# make a copy in tmp, and delete it after we load into memory.
|
||||
with data_locator.local_handle() as lh:
|
||||
# as of AnnData 0.6.19, backed mode performs initial load fast, but at the
|
||||
# cost of significantly slower access to X data.
|
||||
backed = "r" if self.config["backed"] else None
|
||||
self.data = anndata.read_h5ad(lh, backed=backed)
|
||||
|
||||
except ValueError:
|
||||
raise ScanpyFileError(
|
||||
"File must be in the .h5ad format. Please read "
|
||||
"https://github.com/theislab/scanpy_usage/blob/master/170505_seurat/info_h5ad.md to "
|
||||
"learn more about this format. You may be able to convert your file into this format "
|
||||
"using `cellxgene prepare`, please run `cellxgene prepare --help` for more "
|
||||
"information."
|
||||
)
|
||||
except MemoryError:
|
||||
raise ScanpyFileError("Out of memory - file is too large for available memory.")
|
||||
except Exception as e:
|
||||
raise ScanpyFileError(
|
||||
f"{e} - file not found or is inaccessible. File must be an .h5ad object. "
|
||||
f"Please check your input and try again."
|
||||
)
|
||||
|
||||
@requires_data
|
||||
def _validate_and_initialize(self):
|
||||
if anndata_version_is_pre_070() and self.config['backed']:
|
||||
warnings.warn(f"Use of --backed mode with anndata versions older than 0.7 will have serious "
|
||||
"performance issues. Please update to at least anndata 0.7 or later.")
|
||||
|
||||
# var and obs column names must be unique
|
||||
if not self.data.obs.columns.is_unique or not self.data.var.columns.is_unique:
|
||||
raise KeyError(f"All annotation column names must be unique.")
|
||||
|
||||
self._alias_annotation_names()
|
||||
self._validate_data_types()
|
||||
self.cell_count = self.data.shape[0]
|
||||
self.gene_count = self.data.shape[1]
|
||||
self._default_and_validate_layouts()
|
||||
self._create_schema()
|
||||
|
||||
# if the user has specified a fixed label file, go ahead and validate it
|
||||
# so that we can remove errors early in the process.
|
||||
if self.config["annotations_file"]:
|
||||
self._validate_label_data(read_labels(self.get_anno_fname()))
|
||||
|
||||
# heuristic
|
||||
n_values = self.data.shape[0] * self.data.shape[1]
|
||||
if (n_values > 1e8 and self.config["backed"] is True) or (n_values > 5e8):
|
||||
self.config.update({"diffexp_may_be_slow": True})
|
||||
|
||||
@requires_data
|
||||
def _default_and_validate_layouts(self):
|
||||
""" function:
|
||||
a) generate list of default layouts, if not already user specified
|
||||
b) validate layouts are legal. remove/warn on any that are not
|
||||
c) cap total list of layouts at global const MAX_LAYOUTS
|
||||
"""
|
||||
layouts = self.config["layout"]
|
||||
# handle default
|
||||
if layouts is None or len(layouts) == 0:
|
||||
# load default layouts from the data.
|
||||
layouts = [key[2:] for key in self.data.obsm_keys() if type(key) == str and key.startswith("X_")]
|
||||
if len(layouts) == 0:
|
||||
raise PrepareError(f"Unable to find any precomputed layouts within the dataset.")
|
||||
|
||||
# remove invalid layouts
|
||||
valid_layouts = []
|
||||
obsm_keys = self.data.obsm_keys()
|
||||
for layout in layouts:
|
||||
layout_name = f"X_{layout}"
|
||||
if layout_name not in obsm_keys:
|
||||
warnings.warn(f"Ignoring unknown layout name: {layout}.")
|
||||
elif not self._is_valid_layout(self.data.obsm[layout_name]):
|
||||
warnings.warn(f"Ignoring layout due to malformed shape or data type: {layout}")
|
||||
else:
|
||||
valid_layouts.append(layout)
|
||||
|
||||
if len(valid_layouts) == 0:
|
||||
raise PrepareError(f"No valid layout data.")
|
||||
|
||||
# cap layouts to MAX_LAYOUTS
|
||||
self.config["layout"] = valid_layouts[0:MAX_LAYOUTS]
|
||||
|
||||
@requires_data
|
||||
def _is_valid_layout(self, arr):
|
||||
""" return True if this layout data is a valid array for front-end presentation:
|
||||
* ndarray, with shape (n_obs, >= 2), dtype float/int/uint
|
||||
* contains only finite values
|
||||
"""
|
||||
is_valid = type(arr) == np.ndarray and arr.dtype.kind in "fiu"
|
||||
is_valid = is_valid and arr.shape[0] == self.data.n_obs and arr.shape[1] >= 2
|
||||
is_valid = is_valid and np.all(np.isfinite(arr))
|
||||
return is_valid
|
||||
|
||||
@requires_data
|
||||
def _validate_data_types(self):
|
||||
# The backed API does not support interrogation of the underlying sparsity or sparse matrix type
|
||||
# Fake it by asking for a small subarray and testing it. NOTE: if the user has ignored our
|
||||
# anndata <= 0.7 warning, opted for the --backed option, and specified a large, sparse dataset,
|
||||
# this "small" indexing request will load the entire X array. This is due to a bug in anndata<=0.7
|
||||
# which will load the entire X matrix to fullfill any slicing request if X is sparse. See
|
||||
# user warning in _load_data().
|
||||
X0 = self.data.X[0, 0:1]
|
||||
if sparse.isspmatrix(X0) and not sparse.isspmatrix_csc(X0):
|
||||
warnings.warn(
|
||||
f"Scanpy data matrix is sparse, but not a CSC (columnar) matrix. "
|
||||
f"Performance may be improved by using CSC."
|
||||
)
|
||||
if self.data.X.dtype != "float32":
|
||||
warnings.warn(
|
||||
f"Scanpy data matrix is in {self.data.X.dtype} format not float32. " f"Precision may be truncated."
|
||||
)
|
||||
for ax in Axis:
|
||||
curr_axis = getattr(self.data, str(ax))
|
||||
for ann in curr_axis:
|
||||
datatype = curr_axis[ann].dtype
|
||||
downcast_map = {
|
||||
"int64": "int32",
|
||||
"uint32": "int32",
|
||||
"uint64": "int32",
|
||||
"float64": "float32",
|
||||
}
|
||||
if datatype in downcast_map:
|
||||
warnings.warn(
|
||||
f"Scanpy annotation {ax}:{ann} is in unsupported format: {datatype}. "
|
||||
f"Data will be downcast to {downcast_map[datatype]}."
|
||||
)
|
||||
if isinstance(datatype, CategoricalDtype):
|
||||
category_num = len(curr_axis[ann].dtype.categories)
|
||||
if category_num > 500 and category_num > self.config["max_category_items"]:
|
||||
warnings.warn(
|
||||
f"{str(ax).title()} annotation '{ann}' has {category_num} categories, this may be "
|
||||
f"cumbersome or slow to display. We recommend setting the "
|
||||
f"--max-category-items option to 500, this will hide categorical "
|
||||
f"annotations with more than 500 categories in the UI"
|
||||
)
|
||||
|
||||
@requires_data
|
||||
def _validate_label_data(self, labels):
|
||||
"""
|
||||
labels is None if disabled, empty if enabled by no data
|
||||
"""
|
||||
if labels is None or labels.empty:
|
||||
return
|
||||
|
||||
# all lables must have a name, which must be unique and not used in obs column names
|
||||
if not labels.columns.is_unique:
|
||||
raise KeyError(f"All column names specified in user annotations must be unique.")
|
||||
|
||||
# the label index must be unique, and must have same values the anndata obs index
|
||||
if not labels.index.is_unique:
|
||||
raise KeyError(f"All row index values specified in user annotations must be unique.")
|
||||
|
||||
if not labels.index.equals(self.original_obs_index):
|
||||
raise KeyError(
|
||||
"Label file row index does not match H5AD file index. "
|
||||
"Please ensure that column zero (0) in the label file contain the same "
|
||||
"index values as the H5AD file."
|
||||
)
|
||||
|
||||
duplicate_columns = list(set(labels.columns) & set(self.data.obs.columns))
|
||||
if len(duplicate_columns) > 0:
|
||||
raise KeyError(
|
||||
f"Labels file may not contain column names which overlap " f"with h5ad obs columns {duplicate_columns}"
|
||||
)
|
||||
|
||||
# labels must have same count as obs annotations
|
||||
if labels.shape[0] != self.data.obs.shape[0]:
|
||||
raise ValueError("Labels file must have same number of rows as h5ad file.")
|
||||
|
||||
@staticmethod
|
||||
def _annotation_filter_to_mask(filter, d_axis, count):
|
||||
mask = np.ones((count,), dtype=bool)
|
||||
for v in filter:
|
||||
if d_axis[v["name"]].dtype.name in ["boolean", "category", "object"]:
|
||||
key_idx = np.in1d(getattr(d_axis, v["name"]), v["values"])
|
||||
mask = np.logical_and(mask, key_idx)
|
||||
else:
|
||||
min_ = v.get("min", None)
|
||||
max_ = v.get("max", None)
|
||||
if min_ is not None:
|
||||
key_idx = (getattr(d_axis, v["name"]) >= min_).ravel()
|
||||
mask = np.logical_and(mask, key_idx)
|
||||
if max_ is not None:
|
||||
key_idx = (getattr(d_axis, v["name"]) <= max_).ravel()
|
||||
mask = np.logical_and(mask, key_idx)
|
||||
return mask
|
||||
|
||||
@staticmethod
|
||||
def _index_filter_to_mask(filter, count):
|
||||
mask = np.zeros((count,), dtype=bool)
|
||||
for i in filter:
|
||||
if type(i) == list:
|
||||
mask[i[0] : i[1]] = True
|
||||
else:
|
||||
mask[i] = True
|
||||
return mask
|
||||
|
||||
@staticmethod
|
||||
def _axis_filter_to_mask(filter, d_axis, count):
|
||||
mask = np.ones((count,), dtype=bool)
|
||||
if "index" in filter:
|
||||
mask = np.logical_and(mask, ScanpyEngine._index_filter_to_mask(filter["index"], count))
|
||||
if "annotation_value" in filter:
|
||||
mask = np.logical_and(
|
||||
mask, ScanpyEngine._annotation_filter_to_mask(filter["annotation_value"], d_axis, count),
|
||||
)
|
||||
return mask
|
||||
|
||||
@requires_data
|
||||
def _filter_to_mask(self, filter, use_slices=True):
|
||||
if use_slices:
|
||||
obs_selector = slice(0, self.data.n_obs)
|
||||
var_selector = slice(0, self.data.n_vars)
|
||||
else:
|
||||
obs_selector = None
|
||||
var_selector = None
|
||||
|
||||
if filter is not None:
|
||||
if Axis.OBS in filter:
|
||||
obs_selector = self._axis_filter_to_mask(filter["obs"], self.data.obs, self.data.n_obs)
|
||||
if Axis.VAR in filter:
|
||||
var_selector = self._axis_filter_to_mask(filter["var"], self.data.var, self.data.n_vars)
|
||||
return obs_selector, var_selector
|
||||
|
||||
@requires_data
|
||||
def annotation_to_fbs_matrix(self, axis, fields=None, uid=None, collection=None):
|
||||
if axis == Axis.OBS:
|
||||
if self.config["annotations"]:
|
||||
try:
|
||||
labels = read_labels(self.get_anno_fname(uid, collection))
|
||||
except Exception as e:
|
||||
raise ScanpyFileError(
|
||||
f"Error while loading label file: {e}, File must be in the .csv format, please check "
|
||||
f"your input and try again."
|
||||
)
|
||||
else:
|
||||
labels = None
|
||||
|
||||
if labels is not None and not labels.empty:
|
||||
df = self.data.obs.join(labels, self.config["obs_names"])
|
||||
else:
|
||||
df = self.data.obs
|
||||
else:
|
||||
df = self.data.var
|
||||
if fields is not None and len(fields) > 0:
|
||||
df = df[fields]
|
||||
return encode_matrix_fbs(df, col_idx=df.columns)
|
||||
|
||||
@requires_data
|
||||
def annotation_put_fbs(self, axis, fbs, uid=None, collection=None):
|
||||
if not self.config["annotations"]:
|
||||
raise DisabledFeatureError("Writable annotations are not enabled")
|
||||
|
||||
fname = self.get_anno_fname(uid, collection)
|
||||
if not fname:
|
||||
raise ScanpyFileError("Writable annotations - unable to determine file name for annotations")
|
||||
|
||||
if axis != Axis.OBS:
|
||||
raise ValueError("Only OBS dimension access is supported")
|
||||
|
||||
new_label_df = decode_matrix_fbs(fbs)
|
||||
if not new_label_df.empty:
|
||||
new_label_df.index = self.original_obs_index
|
||||
self._validate_label_data(new_label_df) # paranoia
|
||||
|
||||
# if any of the new column labels overlap with our existing labels, raise error
|
||||
duplicate_columns = list(set(new_label_df.columns) & set(self.data.obs.columns))
|
||||
if not new_label_df.columns.is_unique or len(duplicate_columns) > 0:
|
||||
raise KeyError(
|
||||
f"Labels file may not contain column names which overlap " f"with h5ad obs columns {duplicate_columns}"
|
||||
)
|
||||
|
||||
# update our internal state and save it. Multi-threading often enabled,
|
||||
# so treat this as a critical section.
|
||||
with self.label_lock:
|
||||
lastmod = self.data_locator.lastmodtime()
|
||||
lastmodstr = "'unknown'" if lastmod is None else lastmod.isoformat(timespec="seconds")
|
||||
header = (
|
||||
f"# Annotations generated on {datetime.now().isoformat(timespec='seconds')} "
|
||||
f"using cellxgene version {cellxgene_version}\n"
|
||||
f"# Input data file was {self.data_locator.uri_or_path}, "
|
||||
f"which was last modified on {lastmodstr}\n"
|
||||
)
|
||||
write_labels(fname, new_label_df, header, backup_dir=self.get_anno_backup_dir(uid, collection))
|
||||
|
||||
return jsonify_scanpy({"status": "OK"})
|
||||
|
||||
@requires_data
|
||||
def data_frame_to_fbs_matrix(self, filter, axis):
|
||||
"""
|
||||
Retrieves data 'X' and returns in a flatbuffer Matrix.
|
||||
:param filter: filter: dictionary with filter params
|
||||
:param axis: string obs or var
|
||||
:return: flatbuffer Matrix
|
||||
|
||||
Caveats:
|
||||
* currently only supports access on VAR axis
|
||||
* currently only supports filtering on VAR axis
|
||||
"""
|
||||
if axis != Axis.VAR:
|
||||
raise ValueError("Only VAR dimension access is supported")
|
||||
try:
|
||||
obs_selector, var_selector = self._filter_to_mask(filter, use_slices=False)
|
||||
except (KeyError, IndexError, TypeError) as e:
|
||||
raise FilterError(f"Error parsing filter: {e}") from e
|
||||
if obs_selector is not None:
|
||||
raise FilterError("filtering on obs unsupported")
|
||||
|
||||
# Currently only handles VAR dimension
|
||||
X = self.data.X[:, slice(None) if var_selector is None else var_selector]
|
||||
col_idx = np.nonzero([] if var_selector is None else var_selector)[0]
|
||||
return encode_matrix_fbs(X, col_idx=col_idx, row_idx=None)
|
||||
|
||||
@requires_data
|
||||
def diffexp_topN(self, obsFilterA, obsFilterB, top_n=None, interactive_limit=None):
|
||||
if Axis.VAR in obsFilterA or Axis.VAR in obsFilterB:
|
||||
raise FilterError("Observation filters may not contain vaiable conditions")
|
||||
try:
|
||||
obs_mask_A = self._axis_filter_to_mask(obsFilterA["obs"], self.data.obs, self.data.n_obs)
|
||||
obs_mask_B = self._axis_filter_to_mask(obsFilterB["obs"], self.data.obs, self.data.n_obs)
|
||||
except (KeyError, IndexError) as e:
|
||||
raise FilterError(f"Error parsing filter: {e}") from e
|
||||
if top_n is None:
|
||||
top_n = DEFAULT_TOP_N
|
||||
result = diffexp_ttest(self.data, obs_mask_A, obs_mask_B, top_n, self.config["diffexp_lfc_cutoff"])
|
||||
try:
|
||||
return jsonify_scanpy(result)
|
||||
except ValueError:
|
||||
raise JSONEncodingValueError("Error encoding differential expression to JSON")
|
||||
|
||||
@requires_data
|
||||
def layout_to_fbs_matrix(self):
|
||||
"""
|
||||
Return the default 2-D layout for cells as a FBS Matrix.
|
||||
|
||||
Caveats:
|
||||
* does not support filtering
|
||||
* only returns Matrix in columnar layout
|
||||
|
||||
All embeddings must be individually centered & scaled (isotropically)
|
||||
to a [0, 1] range.
|
||||
"""
|
||||
try:
|
||||
layout_data = []
|
||||
for layout in self.config["layout"]:
|
||||
full_embedding = self.data.obsm[f"X_{layout}"]
|
||||
embedding = full_embedding[:, :2]
|
||||
|
||||
# scale isotropically
|
||||
min = embedding.min(axis=0)
|
||||
max = embedding.max(axis=0)
|
||||
scale = np.amax(max - min)
|
||||
normalized_layout = (embedding - min) / scale
|
||||
|
||||
# translate to center on both axis
|
||||
translate = 0.5 - ((max - min) / scale / 2)
|
||||
normalized_layout = normalized_layout + translate
|
||||
|
||||
normalized_layout = normalized_layout.astype(dtype=np.float32)
|
||||
layout_data.append(pandas.DataFrame(normalized_layout, columns=[f"{layout}_0", f"{layout}_1"]))
|
||||
|
||||
except ValueError as e:
|
||||
raise PrepareError(
|
||||
f"Layout has not been calculated using {self.config['layout']}, "
|
||||
f"please prepare your datafile and relaunch cellxgene"
|
||||
) from e
|
||||
|
||||
df = pandas.concat(layout_data, axis=1, copy=False)
|
||||
return encode_matrix_fbs(df, col_idx=df.columns, row_idx=None)
|
||||
@@ -1,36 +0,0 @@
|
||||
from enum import Enum
|
||||
|
||||
|
||||
DEFAULT_TOP_N = 10
|
||||
|
||||
|
||||
class AugmentedEnum(Enum):
|
||||
def __hash__(self):
|
||||
return self.value.__hash__()
|
||||
|
||||
def __eq__(self, other):
|
||||
if isinstance(other, type(self)) or isinstance(other, str):
|
||||
return self.value == other
|
||||
return False
|
||||
|
||||
def __str__(self) -> str:
|
||||
return self.value
|
||||
|
||||
|
||||
class Axis(AugmentedEnum):
|
||||
OBS = "obs"
|
||||
VAR = "var"
|
||||
|
||||
|
||||
class DiffExpMode(AugmentedEnum):
|
||||
TOP_N = "topN"
|
||||
VAR_FILTER = "varFilter"
|
||||
|
||||
|
||||
JSON_NaN_to_num_warning_msg = "JSON encoding failure - please verify all data are finite values (no NaN or Infinities)"
|
||||
REACTIVE_LIMIT = 1_000_000
|
||||
|
||||
MAX_LAYOUTS = 30
|
||||
|
||||
CXGUID = "cxguid"
|
||||
CXG_ANNO_COLLECTION = "cxg_anno_collection"
|
||||
@@ -1,106 +0,0 @@
|
||||
import os
|
||||
import tempfile
|
||||
import fsspec
|
||||
from datetime import datetime
|
||||
|
||||
|
||||
class DataLocator:
|
||||
"""
|
||||
DataLocator is a simple wrapper around fsspec functionality, and provides a
|
||||
set of functions to encapsulate a data location (URI or path), interogate
|
||||
metadata about the object at that location (size, existance, etc) and
|
||||
access the underlying data.
|
||||
|
||||
https://filesystem-spec.readthedocs.io/en/latest/index.html
|
||||
|
||||
Example:
|
||||
dl = DataLocator("/tmp/foo.h5ad")
|
||||
if dl.exists():
|
||||
print(dl.size())
|
||||
with dl.open() as f:
|
||||
thecontents = f.read()
|
||||
|
||||
DataLocator will accept a URI or native path. Error handling is as defined
|
||||
in fsspec.
|
||||
|
||||
"""
|
||||
|
||||
def __init__(self, uri_or_path):
|
||||
self.uri_or_path = uri_or_path
|
||||
self.protocol, self.path = DataLocator._get_protocol_and_path(uri_or_path)
|
||||
# work-around for LocalFileSystem not treating file: and None as the same scheme/protocol
|
||||
self.cname = self.path if self.protocol == "file" else self.uri_or_path
|
||||
# will throw RuntimeError if the protocol is unsupported
|
||||
self.fs = fsspec.filesystem(self.protocol)
|
||||
|
||||
@staticmethod
|
||||
def _get_protocol_and_path(uri_or_path):
|
||||
if "://" in uri_or_path:
|
||||
protocol, path = uri_or_path.split("://", 1)
|
||||
# windows!!! Ignore single letter drive identifiers,
|
||||
# eg, G:\foo.txt
|
||||
if len(protocol) > 1:
|
||||
return protocol, path
|
||||
return None, uri_or_path
|
||||
|
||||
def exists(self):
|
||||
return self.fs.exists(self.cname)
|
||||
|
||||
def size(self):
|
||||
return self.fs.size(self.cname)
|
||||
|
||||
def lastmodtime(self):
|
||||
""" return datetime object representing last modification time, or None if unavailable """
|
||||
info = self.fs.info(self.cname)
|
||||
if self.islocal() and info is not None:
|
||||
return datetime.fromtimestamp(info["mtime"])
|
||||
else:
|
||||
return getattr(info, "LastModified", None)
|
||||
|
||||
def abspath(self):
|
||||
"""
|
||||
return the absolute path for the locator - only really does something
|
||||
for file: protocol, as all others are already absolute
|
||||
"""
|
||||
if self.islocal():
|
||||
return os.path.abspath(self.path)
|
||||
else:
|
||||
return self.uri_or_path
|
||||
|
||||
def isfile(self):
|
||||
return self.fs.isfile(self.cname)
|
||||
|
||||
def open(self, *args):
|
||||
return self.fs.open(self.uri_or_path, *args)
|
||||
|
||||
def islocal(self):
|
||||
return self.protocol is None or self.protocol == "file"
|
||||
|
||||
def local_handle(self):
|
||||
if self.islocal():
|
||||
return LocalFilePath(self.path)
|
||||
|
||||
# if not local, create a tmp file system object to contain the data,
|
||||
# and clean it up when done. If the path has a suffix/extension,
|
||||
# do our best to create a file with the same.
|
||||
ext = os.path.splitext(self.path)
|
||||
suffix = None if ext[1] == '' else ext[1]
|
||||
with self.open() as src, tempfile.NamedTemporaryFile(prefix="cellxgene_", suffix=suffix, delete=False) as tmp:
|
||||
tmp.write(src.read())
|
||||
tmp.close()
|
||||
src.close()
|
||||
tmp_path = tmp.name
|
||||
return LocalFilePath(tmp_path, delete=True)
|
||||
|
||||
|
||||
class LocalFilePath:
|
||||
def __init__(self, tmp_path, delete=False):
|
||||
self.tmp_path = tmp_path
|
||||
self.delete = delete
|
||||
|
||||
def __enter__(self):
|
||||
return self.tmp_path
|
||||
|
||||
def __exit__(self, *args):
|
||||
if self.delete:
|
||||
os.unlink(self.tmp_path)
|
||||
@@ -1,70 +0,0 @@
|
||||
class FilterError(Exception):
|
||||
"""
|
||||
Raised when filter is malformed
|
||||
"""
|
||||
|
||||
def __init__(self, message):
|
||||
self.message = message
|
||||
|
||||
|
||||
class InteractiveError(Exception):
|
||||
"""
|
||||
Raised when computation would exceed interactive time
|
||||
"""
|
||||
|
||||
def __init__(self, message):
|
||||
self.message = message
|
||||
|
||||
|
||||
class JSONEncodingValueError(Exception):
|
||||
"""
|
||||
Raised when file loaded into scanpy is misformatted
|
||||
"""
|
||||
|
||||
def __init__(self, message):
|
||||
self.message = message
|
||||
|
||||
|
||||
class MimeTypeError(Exception):
|
||||
"""
|
||||
Raised when incompatible MIME type selected
|
||||
"""
|
||||
|
||||
def __init__(self, message):
|
||||
self.message = message
|
||||
|
||||
|
||||
class PrepareError(Exception):
|
||||
"""
|
||||
Raised when data is misprepared
|
||||
"""
|
||||
|
||||
def __init__(self, message):
|
||||
self.message = message
|
||||
|
||||
|
||||
class ScanpyFileError(Exception):
|
||||
"""
|
||||
Raised when file loaded into scanpy is misformatted
|
||||
"""
|
||||
|
||||
def __init__(self, message):
|
||||
self.message = message
|
||||
|
||||
|
||||
class DriverError(Exception):
|
||||
"""
|
||||
Raised when file loaded into scanpy is misformatted
|
||||
"""
|
||||
|
||||
def __init__(self, message):
|
||||
self.message = message
|
||||
|
||||
|
||||
class DisabledFeatureError(Exception):
|
||||
"""
|
||||
Raised when an attempt to use a disabled feature occurs
|
||||
"""
|
||||
|
||||
def __init__(self, message):
|
||||
self.message = message
|
||||
@@ -1,41 +0,0 @@
|
||||
# automatically generated by the FlatBuffers compiler, do not modify
|
||||
|
||||
# namespace: NetEncoding
|
||||
|
||||
import flatbuffers
|
||||
|
||||
class Column(object):
|
||||
__slots__ = ['_tab']
|
||||
|
||||
@classmethod
|
||||
def GetRootAsColumn(cls, buf, offset):
|
||||
n = flatbuffers.encode.Get(flatbuffers.packer.uoffset, buf, offset)
|
||||
x = Column()
|
||||
x.Init(buf, n + offset)
|
||||
return x
|
||||
|
||||
# Column
|
||||
def Init(self, buf, pos):
|
||||
self._tab = flatbuffers.table.Table(buf, pos)
|
||||
|
||||
# Column
|
||||
def UType(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.Get(flatbuffers.number_types.Uint8Flags, o + self._tab.Pos)
|
||||
return 0
|
||||
|
||||
# Column
|
||||
def U(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(6))
|
||||
if o != 0:
|
||||
from flatbuffers.table import Table
|
||||
obj = Table(bytearray(), 0)
|
||||
self._tab.Union(obj, o)
|
||||
return obj
|
||||
return None
|
||||
|
||||
def ColumnStart(builder): builder.StartObject(2)
|
||||
def ColumnAddUType(builder, uType): builder.PrependUint8Slot(0, uType, 0)
|
||||
def ColumnAddU(builder, u): builder.PrependUOffsetTRelativeSlot(1, flatbuffers.number_types.UOffsetTFlags.py_type(u), 0)
|
||||
def ColumnEnd(builder): return builder.EndObject()
|
||||
@@ -1,46 +0,0 @@
|
||||
# automatically generated by the FlatBuffers compiler, do not modify
|
||||
|
||||
# namespace: NetEncoding
|
||||
|
||||
import flatbuffers
|
||||
|
||||
class Float32Array(object):
|
||||
__slots__ = ['_tab']
|
||||
|
||||
@classmethod
|
||||
def GetRootAsFloat32Array(cls, buf, offset):
|
||||
n = flatbuffers.encode.Get(flatbuffers.packer.uoffset, buf, offset)
|
||||
x = Float32Array()
|
||||
x.Init(buf, n + offset)
|
||||
return x
|
||||
|
||||
# Float32Array
|
||||
def Init(self, buf, pos):
|
||||
self._tab = flatbuffers.table.Table(buf, pos)
|
||||
|
||||
# Float32Array
|
||||
def Data(self, j):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
a = self._tab.Vector(o)
|
||||
return self._tab.Get(flatbuffers.number_types.Float32Flags, a + flatbuffers.number_types.UOffsetTFlags.py_type(j * 4))
|
||||
return 0
|
||||
|
||||
# Float32Array
|
||||
def DataAsNumpy(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.GetVectorAsNumpy(flatbuffers.number_types.Float32Flags, o)
|
||||
return 0
|
||||
|
||||
# Float32Array
|
||||
def DataLength(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.VectorLen(o)
|
||||
return 0
|
||||
|
||||
def Float32ArrayStart(builder): builder.StartObject(1)
|
||||
def Float32ArrayAddData(builder, data): builder.PrependUOffsetTRelativeSlot(0, flatbuffers.number_types.UOffsetTFlags.py_type(data), 0)
|
||||
def Float32ArrayStartDataVector(builder, numElems): return builder.StartVector(4, numElems, 4)
|
||||
def Float32ArrayEnd(builder): return builder.EndObject()
|
||||
@@ -1,46 +0,0 @@
|
||||
# automatically generated by the FlatBuffers compiler, do not modify
|
||||
|
||||
# namespace: NetEncoding
|
||||
|
||||
import flatbuffers
|
||||
|
||||
class Float64Array(object):
|
||||
__slots__ = ['_tab']
|
||||
|
||||
@classmethod
|
||||
def GetRootAsFloat64Array(cls, buf, offset):
|
||||
n = flatbuffers.encode.Get(flatbuffers.packer.uoffset, buf, offset)
|
||||
x = Float64Array()
|
||||
x.Init(buf, n + offset)
|
||||
return x
|
||||
|
||||
# Float64Array
|
||||
def Init(self, buf, pos):
|
||||
self._tab = flatbuffers.table.Table(buf, pos)
|
||||
|
||||
# Float64Array
|
||||
def Data(self, j):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
a = self._tab.Vector(o)
|
||||
return self._tab.Get(flatbuffers.number_types.Float64Flags, a + flatbuffers.number_types.UOffsetTFlags.py_type(j * 8))
|
||||
return 0
|
||||
|
||||
# Float64Array
|
||||
def DataAsNumpy(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.GetVectorAsNumpy(flatbuffers.number_types.Float64Flags, o)
|
||||
return 0
|
||||
|
||||
# Float64Array
|
||||
def DataLength(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.VectorLen(o)
|
||||
return 0
|
||||
|
||||
def Float64ArrayStart(builder): builder.StartObject(1)
|
||||
def Float64ArrayAddData(builder, data): builder.PrependUOffsetTRelativeSlot(0, flatbuffers.number_types.UOffsetTFlags.py_type(data), 0)
|
||||
def Float64ArrayStartDataVector(builder, numElems): return builder.StartVector(8, numElems, 8)
|
||||
def Float64ArrayEnd(builder): return builder.EndObject()
|
||||
@@ -1,46 +0,0 @@
|
||||
# automatically generated by the FlatBuffers compiler, do not modify
|
||||
|
||||
# namespace: NetEncoding
|
||||
|
||||
import flatbuffers
|
||||
|
||||
class Int32Array(object):
|
||||
__slots__ = ['_tab']
|
||||
|
||||
@classmethod
|
||||
def GetRootAsInt32Array(cls, buf, offset):
|
||||
n = flatbuffers.encode.Get(flatbuffers.packer.uoffset, buf, offset)
|
||||
x = Int32Array()
|
||||
x.Init(buf, n + offset)
|
||||
return x
|
||||
|
||||
# Int32Array
|
||||
def Init(self, buf, pos):
|
||||
self._tab = flatbuffers.table.Table(buf, pos)
|
||||
|
||||
# Int32Array
|
||||
def Data(self, j):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
a = self._tab.Vector(o)
|
||||
return self._tab.Get(flatbuffers.number_types.Int32Flags, a + flatbuffers.number_types.UOffsetTFlags.py_type(j * 4))
|
||||
return 0
|
||||
|
||||
# Int32Array
|
||||
def DataAsNumpy(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.GetVectorAsNumpy(flatbuffers.number_types.Int32Flags, o)
|
||||
return 0
|
||||
|
||||
# Int32Array
|
||||
def DataLength(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.VectorLen(o)
|
||||
return 0
|
||||
|
||||
def Int32ArrayStart(builder): builder.StartObject(1)
|
||||
def Int32ArrayAddData(builder, data): builder.PrependUOffsetTRelativeSlot(0, flatbuffers.number_types.UOffsetTFlags.py_type(data), 0)
|
||||
def Int32ArrayStartDataVector(builder, numElems): return builder.StartVector(4, numElems, 4)
|
||||
def Int32ArrayEnd(builder): return builder.EndObject()
|
||||
@@ -1,46 +0,0 @@
|
||||
# automatically generated by the FlatBuffers compiler, do not modify
|
||||
|
||||
# namespace: NetEncoding
|
||||
|
||||
import flatbuffers
|
||||
|
||||
class JSONEncodedArray(object):
|
||||
__slots__ = ['_tab']
|
||||
|
||||
@classmethod
|
||||
def GetRootAsJSONEncodedArray(cls, buf, offset):
|
||||
n = flatbuffers.encode.Get(flatbuffers.packer.uoffset, buf, offset)
|
||||
x = JSONEncodedArray()
|
||||
x.Init(buf, n + offset)
|
||||
return x
|
||||
|
||||
# JSONEncodedArray
|
||||
def Init(self, buf, pos):
|
||||
self._tab = flatbuffers.table.Table(buf, pos)
|
||||
|
||||
# JSONEncodedArray
|
||||
def Data(self, j):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
a = self._tab.Vector(o)
|
||||
return self._tab.Get(flatbuffers.number_types.Uint8Flags, a + flatbuffers.number_types.UOffsetTFlags.py_type(j * 1))
|
||||
return 0
|
||||
|
||||
# JSONEncodedArray
|
||||
def DataAsNumpy(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.GetVectorAsNumpy(flatbuffers.number_types.Uint8Flags, o)
|
||||
return 0
|
||||
|
||||
# JSONEncodedArray
|
||||
def DataLength(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.VectorLen(o)
|
||||
return 0
|
||||
|
||||
def JSONEncodedArrayStart(builder): builder.StartObject(1)
|
||||
def JSONEncodedArrayAddData(builder, data): builder.PrependUOffsetTRelativeSlot(0, flatbuffers.number_types.UOffsetTFlags.py_type(data), 0)
|
||||
def JSONEncodedArrayStartDataVector(builder, numElems): return builder.StartVector(1, numElems, 1)
|
||||
def JSONEncodedArrayEnd(builder): return builder.EndObject()
|
||||
@@ -1,98 +0,0 @@
|
||||
# automatically generated by the FlatBuffers compiler, do not modify
|
||||
|
||||
# namespace: NetEncoding
|
||||
|
||||
import flatbuffers
|
||||
|
||||
class Matrix(object):
|
||||
__slots__ = ['_tab']
|
||||
|
||||
@classmethod
|
||||
def GetRootAsMatrix(cls, buf, offset):
|
||||
n = flatbuffers.encode.Get(flatbuffers.packer.uoffset, buf, offset)
|
||||
x = Matrix()
|
||||
x.Init(buf, n + offset)
|
||||
return x
|
||||
|
||||
# Matrix
|
||||
def Init(self, buf, pos):
|
||||
self._tab = flatbuffers.table.Table(buf, pos)
|
||||
|
||||
# Matrix
|
||||
def NRows(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.Get(flatbuffers.number_types.Uint32Flags, o + self._tab.Pos)
|
||||
return 0
|
||||
|
||||
# Matrix
|
||||
def NCols(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(6))
|
||||
if o != 0:
|
||||
return self._tab.Get(flatbuffers.number_types.Uint32Flags, o + self._tab.Pos)
|
||||
return 0
|
||||
|
||||
# Matrix
|
||||
def Columns(self, j):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(8))
|
||||
if o != 0:
|
||||
x = self._tab.Vector(o)
|
||||
x += flatbuffers.number_types.UOffsetTFlags.py_type(j) * 4
|
||||
x = self._tab.Indirect(x)
|
||||
from .Column import Column
|
||||
obj = Column()
|
||||
obj.Init(self._tab.Bytes, x)
|
||||
return obj
|
||||
return None
|
||||
|
||||
# Matrix
|
||||
def ColumnsLength(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(8))
|
||||
if o != 0:
|
||||
return self._tab.VectorLen(o)
|
||||
return 0
|
||||
|
||||
# Matrix
|
||||
def ColIndexType(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(10))
|
||||
if o != 0:
|
||||
return self._tab.Get(flatbuffers.number_types.Uint8Flags, o + self._tab.Pos)
|
||||
return 0
|
||||
|
||||
# Matrix
|
||||
def ColIndex(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(12))
|
||||
if o != 0:
|
||||
from flatbuffers.table import Table
|
||||
obj = Table(bytearray(), 0)
|
||||
self._tab.Union(obj, o)
|
||||
return obj
|
||||
return None
|
||||
|
||||
# Matrix
|
||||
def RowIndexType(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(14))
|
||||
if o != 0:
|
||||
return self._tab.Get(flatbuffers.number_types.Uint8Flags, o + self._tab.Pos)
|
||||
return 0
|
||||
|
||||
# Matrix
|
||||
def RowIndex(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(16))
|
||||
if o != 0:
|
||||
from flatbuffers.table import Table
|
||||
obj = Table(bytearray(), 0)
|
||||
self._tab.Union(obj, o)
|
||||
return obj
|
||||
return None
|
||||
|
||||
def MatrixStart(builder): builder.StartObject(7)
|
||||
def MatrixAddNRows(builder, nRows): builder.PrependUint32Slot(0, nRows, 0)
|
||||
def MatrixAddNCols(builder, nCols): builder.PrependUint32Slot(1, nCols, 0)
|
||||
def MatrixAddColumns(builder, columns): builder.PrependUOffsetTRelativeSlot(2, flatbuffers.number_types.UOffsetTFlags.py_type(columns), 0)
|
||||
def MatrixStartColumnsVector(builder, numElems): return builder.StartVector(4, numElems, 4)
|
||||
def MatrixAddColIndexType(builder, colIndexType): builder.PrependUint8Slot(3, colIndexType, 0)
|
||||
def MatrixAddColIndex(builder, colIndex): builder.PrependUOffsetTRelativeSlot(4, flatbuffers.number_types.UOffsetTFlags.py_type(colIndex), 0)
|
||||
def MatrixAddRowIndexType(builder, rowIndexType): builder.PrependUint8Slot(5, rowIndexType, 0)
|
||||
def MatrixAddRowIndex(builder, rowIndex): builder.PrependUOffsetTRelativeSlot(6, flatbuffers.number_types.UOffsetTFlags.py_type(rowIndex), 0)
|
||||
def MatrixEnd(builder): return builder.EndObject()
|
||||
@@ -1,12 +0,0 @@
|
||||
# automatically generated by the FlatBuffers compiler, do not modify
|
||||
|
||||
# namespace: NetEncoding
|
||||
|
||||
class TypedArray(object):
|
||||
NONE = 0
|
||||
Float32Array = 1
|
||||
Int32Array = 2
|
||||
Uint32Array = 3
|
||||
Float64Array = 4
|
||||
JSONEncodedArray = 5
|
||||
|
||||
@@ -1,46 +0,0 @@
|
||||
# automatically generated by the FlatBuffers compiler, do not modify
|
||||
|
||||
# namespace: NetEncoding
|
||||
|
||||
import flatbuffers
|
||||
|
||||
class Uint32Array(object):
|
||||
__slots__ = ['_tab']
|
||||
|
||||
@classmethod
|
||||
def GetRootAsUint32Array(cls, buf, offset):
|
||||
n = flatbuffers.encode.Get(flatbuffers.packer.uoffset, buf, offset)
|
||||
x = Uint32Array()
|
||||
x.Init(buf, n + offset)
|
||||
return x
|
||||
|
||||
# Uint32Array
|
||||
def Init(self, buf, pos):
|
||||
self._tab = flatbuffers.table.Table(buf, pos)
|
||||
|
||||
# Uint32Array
|
||||
def Data(self, j):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
a = self._tab.Vector(o)
|
||||
return self._tab.Get(flatbuffers.number_types.Uint32Flags, a + flatbuffers.number_types.UOffsetTFlags.py_type(j * 4))
|
||||
return 0
|
||||
|
||||
# Uint32Array
|
||||
def DataAsNumpy(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.GetVectorAsNumpy(flatbuffers.number_types.Uint32Flags, o)
|
||||
return 0
|
||||
|
||||
# Uint32Array
|
||||
def DataLength(self):
|
||||
o = flatbuffers.number_types.UOffsetTFlags.py_type(self._tab.Offset(4))
|
||||
if o != 0:
|
||||
return self._tab.VectorLen(o)
|
||||
return 0
|
||||
|
||||
def Uint32ArrayStart(builder): builder.StartObject(1)
|
||||
def Uint32ArrayAddData(builder, data): builder.PrependUOffsetTRelativeSlot(0, flatbuffers.number_types.UOffsetTFlags.py_type(data), 0)
|
||||
def Uint32ArrayStartDataVector(builder, numElems): return builder.StartVector(4, numElems, 4)
|
||||
def Uint32ArrayEnd(builder): return builder.EndObject()
|
||||
@@ -1,282 +0,0 @@
|
||||
import flatbuffers
|
||||
import numpy as np
|
||||
from scipy import sparse
|
||||
import pandas as pd
|
||||
import json
|
||||
|
||||
import server.app.util.fbs.NetEncoding.Column as Column
|
||||
import server.app.util.fbs.NetEncoding.TypedArray as TypedArray
|
||||
import server.app.util.fbs.NetEncoding.Matrix as Matrix
|
||||
import server.app.util.fbs.NetEncoding.Int32Array as Int32Array
|
||||
import server.app.util.fbs.NetEncoding.Uint32Array as Uint32Array
|
||||
import server.app.util.fbs.NetEncoding.Float32Array as Float32Array
|
||||
import server.app.util.fbs.NetEncoding.Float64Array as Float64Array
|
||||
import server.app.util.fbs.NetEncoding.JSONEncodedArray as JSONEncodedArray
|
||||
|
||||
|
||||
# Placeholder until recent enhancements to flatbuffers Python
|
||||
# runtime are released, at which point we can use the default
|
||||
# version. This code is a port of the head. See:
|
||||
#
|
||||
# https://github.com/google/flatbuffers/pull/4829
|
||||
#
|
||||
def CreateNumpyVector(builder, x):
|
||||
"""CreateNumpyVector writes a numpy array into the buffer."""
|
||||
|
||||
if not isinstance(x, np.ndarray):
|
||||
raise TypeError(f"non-numpy-ndarray passed to CreateNumpyVector ({type(x)}")
|
||||
|
||||
if x.dtype.kind not in ["b", "i", "u", "f"]:
|
||||
raise TypeError("numpy-ndarray holds elements of unsupported datatype")
|
||||
|
||||
if x.ndim > 1:
|
||||
raise TypeError("multidimensional-ndarray passed to CreateNumpyVector")
|
||||
|
||||
builder.StartVector(x.itemsize, x.size, x.dtype.alignment)
|
||||
|
||||
# Ensure little endian byte ordering
|
||||
if x.dtype.str[0] == "<":
|
||||
x_little_endian = x
|
||||
else:
|
||||
x_little_endian = x.byteswap(inplace=False)
|
||||
|
||||
# Calculate total length
|
||||
length = int(x_little_endian.itemsize * x_little_endian.size)
|
||||
builder.head = int(builder.Head() - length)
|
||||
|
||||
# tobytes ensures c_contiguous ordering
|
||||
builder.Bytes[builder.Head() : builder.Head() + length] = x_little_endian.tobytes(order="C")
|
||||
|
||||
return builder.EndVector(x.size)
|
||||
|
||||
|
||||
# Serialization helper
|
||||
def serialize_column(builder, typed_arr):
|
||||
""" Serialize NetEncoding.Column """
|
||||
(u_type, u_value) = typed_arr
|
||||
Column.ColumnStart(builder)
|
||||
Column.ColumnAddUType(builder, u_type)
|
||||
Column.ColumnAddU(builder, u_value)
|
||||
return Column.ColumnEnd(builder)
|
||||
|
||||
|
||||
# Serialization helper
|
||||
def serialize_matrix(builder, n_rows, n_cols, columns, col_idx):
|
||||
""" Serialize NetEncoding.Matrix """
|
||||
Matrix.MatrixStart(builder)
|
||||
Matrix.MatrixAddNRows(builder, n_rows)
|
||||
Matrix.MatrixAddNCols(builder, n_cols)
|
||||
Matrix.MatrixAddColumns(builder, columns)
|
||||
if col_idx is not None:
|
||||
(u_type, u_val) = col_idx
|
||||
Matrix.MatrixAddColIndexType(builder, u_type)
|
||||
Matrix.MatrixAddColIndex(builder, u_val)
|
||||
return Matrix.MatrixEnd(builder)
|
||||
|
||||
|
||||
# Serialization helper
|
||||
def serialize_typed_array(builder, source_array, encoding_info):
|
||||
"""
|
||||
Serialize any of the various typed arrays, eg, Float32Array. Specific
|
||||
means of serialization and type conversion are provided by type_info.
|
||||
"""
|
||||
arr = source_array
|
||||
(array_type, as_type) = encoding_info(source_array)
|
||||
|
||||
if isinstance(arr, pd.Index):
|
||||
arr = arr.to_series()
|
||||
|
||||
# convert to a simple ndarray
|
||||
if as_type == "json":
|
||||
as_json = arr.to_json(orient="records")
|
||||
arr = np.array(bytearray(as_json, "utf-8"))
|
||||
else:
|
||||
if sparse.issparse(arr):
|
||||
arr = arr.toarray()
|
||||
elif isinstance(arr, pd.Series):
|
||||
arr = arr.to_numpy()
|
||||
if arr.dtype != as_type:
|
||||
arr = arr.astype(as_type)
|
||||
|
||||
# serialize the ndarray into a vector
|
||||
if arr.ndim == 2:
|
||||
if arr.shape[0] == 1:
|
||||
arr = arr[0]
|
||||
elif arr.shape[1] == 1:
|
||||
arr = arr.T[0]
|
||||
vec = CreateNumpyVector(builder, arr)
|
||||
|
||||
# serialize the typed array table
|
||||
builder.StartObject(1)
|
||||
builder.PrependUOffsetTRelativeSlot(0, vec, 0)
|
||||
array_value = builder.EndObject()
|
||||
return (array_type, array_value)
|
||||
|
||||
|
||||
column_encoding_type_map = {
|
||||
# array protocol string: ( array_type, as_type )
|
||||
np.dtype(np.float64).str: (TypedArray.TypedArray.Float32Array, np.float32),
|
||||
np.dtype(np.float32).str: (TypedArray.TypedArray.Float32Array, np.float32),
|
||||
np.dtype(np.float16).str: (TypedArray.TypedArray.Float32Array, np.float32),
|
||||
np.dtype(np.int8).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.int16).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.uint8).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
np.dtype(np.uint16).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
}
|
||||
column_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
|
||||
|
||||
|
||||
def column_encoding(arr):
|
||||
return column_encoding_type_map.get(arr.dtype.str, column_encoding_default)
|
||||
|
||||
|
||||
index_encoding_type_map = {
|
||||
# array protocol string: ( array_type, as_type )
|
||||
np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
|
||||
np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
|
||||
}
|
||||
index_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
|
||||
|
||||
|
||||
def index_encoding(arr):
|
||||
return index_encoding_type_map.get(arr.dtype.str, index_encoding_default)
|
||||
|
||||
|
||||
def guess_at_mem_needed(matrix):
|
||||
(n_rows, n_cols) = matrix.shape
|
||||
if isinstance(matrix, np.ndarray) or sparse.issparse(matrix):
|
||||
guess = (n_rows * n_cols * matrix.dtype.itemsize) + 1024
|
||||
elif isinstance(matrix, pd.DataFrame):
|
||||
# XXX TODO - DataFrame type estimate
|
||||
guess = 1
|
||||
else:
|
||||
guess = 1
|
||||
|
||||
# round up to nearest 1024 bytes
|
||||
guess = (guess + 0x400) & (~0x3FF)
|
||||
return guess
|
||||
|
||||
|
||||
def encode_matrix_fbs(matrix, row_idx=None, col_idx=None):
|
||||
"""
|
||||
Given a 2D DataFrame, ndarray or sparse equivalent, create and return a
|
||||
Matrix flatbuffer.
|
||||
|
||||
:param matrix: 2D DataFrame, ndarray or sparse equivalent
|
||||
:param row_idx: index for row dimension, Index or ndarray
|
||||
:param col_idx: index for col dimension, Index or ndarray
|
||||
|
||||
NOTE: row indices are (currently) unsupported and must be None
|
||||
"""
|
||||
|
||||
if row_idx is not None:
|
||||
raise ValueError("row indexing not supported for FBS Matrix")
|
||||
if matrix.ndim != 2:
|
||||
raise ValueError("FBS Matrix must be 2D")
|
||||
|
||||
(n_rows, n_cols) = matrix.shape
|
||||
|
||||
# estimate size needed, so we don't unnecessarily realloc.
|
||||
builder = flatbuffers.Builder(guess_at_mem_needed(matrix))
|
||||
|
||||
columns = []
|
||||
for cidx in range(n_cols - 1, -1, -1):
|
||||
# serialize the typed array
|
||||
col = matrix.iloc[:, cidx] if isinstance(matrix, pd.DataFrame) else matrix[:, cidx]
|
||||
typed_arr = serialize_typed_array(builder, col, column_encoding)
|
||||
|
||||
# serialize the Column union
|
||||
columns.append(serialize_column(builder, typed_arr))
|
||||
|
||||
# Serialize Matrix.columns[]
|
||||
Matrix.MatrixStartColumnsVector(builder, n_cols)
|
||||
for c in columns:
|
||||
builder.PrependUOffsetTRelative(c)
|
||||
matrix_column_vec = builder.EndVector(n_cols)
|
||||
|
||||
# serialize the colIndex if provided
|
||||
cidx = None
|
||||
if col_idx is not None:
|
||||
cidx = serialize_typed_array(builder, col_idx, index_encoding)
|
||||
|
||||
# Serialize Matrix
|
||||
matrix = serialize_matrix(builder, n_rows, n_cols, matrix_column_vec, cidx)
|
||||
|
||||
builder.Finish(matrix)
|
||||
return builder.Output()
|
||||
|
||||
|
||||
def deserialize_typed_array(tarr):
|
||||
type_map = {
|
||||
TypedArray.TypedArray.NONE: None,
|
||||
TypedArray.TypedArray.Uint32Array: Uint32Array.Uint32Array,
|
||||
TypedArray.TypedArray.Int32Array: Int32Array.Int32Array,
|
||||
TypedArray.TypedArray.Float32Array: Float32Array.Float32Array,
|
||||
TypedArray.TypedArray.Float64Array: Float64Array.Float64Array,
|
||||
TypedArray.TypedArray.JSONEncodedArray: JSONEncodedArray.JSONEncodedArray,
|
||||
}
|
||||
(u_type, u) = tarr
|
||||
if u_type is TypedArray.TypedArray.NONE:
|
||||
return None
|
||||
|
||||
TarType = type_map.get(u_type, None)
|
||||
if TarType is None:
|
||||
raise TypeError(f"FBS contains unknown data type: {u_type}")
|
||||
|
||||
arr = TarType()
|
||||
arr.Init(u.Bytes, u.Pos)
|
||||
narr = arr.DataAsNumpy()
|
||||
if u_type == TypedArray.TypedArray.JSONEncodedArray:
|
||||
narr = json.loads(narr.tostring().decode("utf-8"))
|
||||
return narr
|
||||
|
||||
|
||||
def decode_matrix_fbs(fbs):
|
||||
"""
|
||||
Given an FBS-encoded Matrix, return a Pandas DataFrame the contains the data
|
||||
and indices.
|
||||
"""
|
||||
matrix = Matrix.Matrix.GetRootAsMatrix(fbs, 0)
|
||||
n_rows = matrix.NRows()
|
||||
n_cols = matrix.NCols()
|
||||
if n_rows == 0 or n_cols == 0:
|
||||
return pd.DataFrame()
|
||||
|
||||
if matrix.RowIndexType() is not TypedArray.TypedArray.NONE:
|
||||
raise ValueError("row indexing not supported for FBS Matrix")
|
||||
|
||||
columns_length = matrix.ColumnsLength()
|
||||
|
||||
columns_index = deserialize_typed_array((matrix.ColIndexType(), matrix.ColIndex()))
|
||||
if columns_index is None:
|
||||
columns_index = range(0, n_cols)
|
||||
|
||||
# sanity checks
|
||||
if len(columns_index) != n_cols or columns_length != n_cols:
|
||||
raise ValueError("FBS column count does not match number of columns in underlying matrix")
|
||||
|
||||
columns_data = {}
|
||||
columns_type = {}
|
||||
for col_idx in range(0, columns_length):
|
||||
col = matrix.Columns(col_idx)
|
||||
tarr = (col.UType(), col.U())
|
||||
data = deserialize_typed_array(tarr)
|
||||
columns_data[columns_index[col_idx]] = data
|
||||
if len(data) != n_rows:
|
||||
raise ValueError("FBS column length does not match number of rows")
|
||||
if col.UType() is TypedArray.TypedArray.JSONEncodedArray:
|
||||
columns_type[columns_index[col_idx]] = "category"
|
||||
|
||||
df = pd.DataFrame.from_dict(data=columns_data).astype(columns_type, copy=False)
|
||||
|
||||
# more sanity checks
|
||||
if not df.columns.is_unique or len(df.columns) != n_cols:
|
||||
raise KeyError("FBS column indices are not unique")
|
||||
|
||||
return df
|
||||
@@ -1,37 +0,0 @@
|
||||
"""
|
||||
Load and parse ontologies - currently support OBO files only.
|
||||
"""
|
||||
import fsspec
|
||||
import fastobo
|
||||
import traceback # use built-in formatter for SyntaxError
|
||||
|
||||
|
||||
""" our default ontology is the PURL for the Cell Ontology. See http://www.obofoundry.org/ontology/cl.html """
|
||||
DefaultOnotology = "http://purl.obolibrary.org/obo/cl.obo"
|
||||
|
||||
|
||||
class OntologyLoadFailure(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def load_obo(path):
|
||||
""" given a URI or path, return an array of term names """
|
||||
if path is None:
|
||||
path = DefaultOnotology
|
||||
|
||||
try:
|
||||
with fsspec.open(path) as f:
|
||||
obo = fastobo.iter(f)
|
||||
terms = filter(lambda stanza: type(stanza) is fastobo.term.TermFrame, obo)
|
||||
names = [tag.name for term in terms for tag in term if type(tag) is fastobo.term.NameClause]
|
||||
return names
|
||||
|
||||
except FileNotFoundError as e:
|
||||
raise OntologyLoadFailure(f"Unable to find OBO ontology path: {path}") from e
|
||||
|
||||
except SyntaxError as e:
|
||||
msg = ''.join(traceback.format_exception_only(SyntaxError, e))
|
||||
raise OntologyLoadFailure(msg) from e
|
||||
|
||||
except Exception as e:
|
||||
raise OntologyLoadFailure(f"Error loading OBO file {path}") from e
|
||||
@@ -1,44 +0,0 @@
|
||||
from functools import wraps
|
||||
|
||||
from flask import json
|
||||
from numpy import float32, integer
|
||||
|
||||
from server.app.util.errors import DriverError
|
||||
|
||||
|
||||
class Float32JSONEncoder(json.JSONEncoder):
|
||||
def __init__(self, *args, **kwargs):
|
||||
"""
|
||||
NaN/Infinities are illegal in standard JSON. Python extends JSON with
|
||||
non-standard symbols that most JavaScript JSON parsers do not understand.
|
||||
The `allow_nan` parameter will force Python simplejson to throw an ValueError
|
||||
if it runs into non-finite floating point values which are unsupported by
|
||||
standard JSON.
|
||||
"""
|
||||
kwargs["allow_nan"] = False
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
def default(self, obj):
|
||||
if isinstance(obj, float32):
|
||||
return float(obj)
|
||||
elif isinstance(obj, integer):
|
||||
return int(obj)
|
||||
return json.JSONEncoder.default(self, obj)
|
||||
|
||||
|
||||
def custom_format_warning(msg, *args, **kwargs):
|
||||
return f"[cellxgene] Warning: {msg} \n"
|
||||
|
||||
|
||||
def jsonify_scanpy(data):
|
||||
return json.dumps(data, cls=Float32JSONEncoder, allow_nan=False)
|
||||
|
||||
|
||||
def requires_data(func):
|
||||
@wraps(func)
|
||||
def wrapped_function(self, *args, **kwargs):
|
||||
if self.data is None:
|
||||
raise DriverError(f"error data must be loaded before you call {func.__name__}")
|
||||
return func(self, *args, **kwargs)
|
||||
|
||||
return wrapped_function
|
||||
@@ -1,17 +0,0 @@
|
||||
import os
|
||||
from flask import Blueprint, render_template, send_from_directory, current_app
|
||||
|
||||
|
||||
bp = Blueprint("webapp", __name__, template_folder="templates")
|
||||
|
||||
|
||||
@bp.route("/")
|
||||
def index():
|
||||
dataset_title = current_app.config["DATASET_TITLE"]
|
||||
scripts = current_app.config["SCRIPTS"]
|
||||
return render_template("index.html", datasetTitle=dataset_title, SCRIPTS=scripts)
|
||||
|
||||
|
||||
@bp.route("/favicon.png")
|
||||
def favicon():
|
||||
return send_from_directory(os.path.join(bp.root_path, "static/img/"), "favicon.png")
|
||||
Reference in New Issue
Block a user