mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-30 16:08:12 +08:00
server refactor (#1140)
This PR contains a refactoring to make adding new features easier. The new features include supporting the tiledb format, and the multi dataset application. The refactoring includes Simplifying the directory structure and files. a class structure to handle annotations (currently one type: AnnotationsLocalFile). a class to handle application configuration a class structure to handle matrix data (currently AnndataAdaptor and CxgAdaptor). CxgAdaptor uses tiledb. Algorithms that were previously dependent on the scanpy anndata object are now generalized to work with an abstract interface. The multi dataset option is not fully supported yet, and so the option to use it is hidden. Use "cli launch --dataroot ..." To access this feature. All combinations of app single dataset/ app multi dataset and AnndataAdaptor/CxgAdaptor work with all the features, such as annotations, ontologies, diffexp.
This commit is contained in:
@@ -0,0 +1,247 @@
|
||||
from datetime import datetime
|
||||
import re
|
||||
from uuid import uuid4
|
||||
import os
|
||||
import pandas as pd
|
||||
from hashlib import blake2b
|
||||
import base64
|
||||
from server import __version__ as cellxgene_version
|
||||
import threading
|
||||
from server.common.errors import AnnotationsError, OntologyLoadFailure
|
||||
from server.common.utils import series_to_schema
|
||||
import fsspec
|
||||
import fastobo
|
||||
import traceback # use built-in formatter for SyntaxError
|
||||
from flask import session
|
||||
from abc import ABCMeta, abstractmethod
|
||||
|
||||
|
||||
class Annotations(metaclass=ABCMeta):
|
||||
""" baseclass for annotations, including ontologies"""
|
||||
|
||||
""" our default ontology is the PURL for the Cell Ontology.
|
||||
See http://www.obofoundry.org/ontology/cl.html """
|
||||
DefaultOnotology = "http://purl.obolibrary.org/obo/cl.obo"
|
||||
|
||||
def __init__(self):
|
||||
self.ontology_data = None
|
||||
|
||||
def load_ontology(self, path):
|
||||
"""Load and parse ontologies - currently support OBO files only."""
|
||||
if path is None:
|
||||
path = self.DefaultOnotology
|
||||
|
||||
try:
|
||||
with fsspec.open(path) as f:
|
||||
obo = fastobo.iter(f)
|
||||
terms = filter(lambda stanza: type(stanza) is fastobo.term.TermFrame, obo)
|
||||
names = [tag.name for term in terms for tag in term if type(tag) is fastobo.term.NameClause]
|
||||
self.ontology_data = names
|
||||
|
||||
except FileNotFoundError as e:
|
||||
raise OntologyLoadFailure(f"Unable to find OBO ontology path: {path}") from e
|
||||
|
||||
except SyntaxError as e:
|
||||
msg = ''.join(traceback.format_exception_only(SyntaxError, e))
|
||||
raise OntologyLoadFailure(msg) from e
|
||||
|
||||
except Exception as e:
|
||||
raise OntologyLoadFailure(f"Error loading OBO file {path}") from e
|
||||
|
||||
def get_schema(self, data_adaptor):
|
||||
labels = self.read_labels(data_adaptor)
|
||||
schema = []
|
||||
if labels is not None and not labels.empty:
|
||||
for col in labels.columns:
|
||||
col_schema = dict(name=col, writable=True)
|
||||
col_schema.update(series_to_schema(labels[col]))
|
||||
schema.append(col_schema)
|
||||
|
||||
return schema
|
||||
|
||||
@abstractmethod
|
||||
def set_collection(self, name):
|
||||
"""set or create a new annotation collection"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def read_labels(self, data_adaptor):
|
||||
"""Return the labels as a pandas.DataFrame"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def write_labels(self, df, data_adaptor):
|
||||
"""Write the labels (df) to a persistent storage such that it can later be read"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def update_parameters(self, parameters, data_adaptor):
|
||||
"""Update configuration parameters that describe information about the annotations feature"""
|
||||
pass
|
||||
|
||||
|
||||
class AnnotationsLocalFile(Annotations):
|
||||
|
||||
CXGUID = "cxguid"
|
||||
CXG_ANNO_COLLECTION = "cxg_anno_collection"
|
||||
|
||||
def __init__(self, output_dir, output_file):
|
||||
super().__init__()
|
||||
self.output_dir = output_dir
|
||||
self.output_file = output_file
|
||||
# lock used to protect label file write ops
|
||||
self.label_lock = threading.RLock()
|
||||
|
||||
def is_safe_collection_name(self, name):
|
||||
"""
|
||||
return true if this is a safe collection name
|
||||
|
||||
this is ultra conservative. If we want to allow full legal file name syntax,
|
||||
we could look at modules like `pathvalidate`
|
||||
"""
|
||||
if name is None:
|
||||
return False
|
||||
return re.match(r"^[\w\-]+$", name) is not None
|
||||
|
||||
def set_collection(self, name):
|
||||
session[self.CXG_ANNO_COLLECTION] = name
|
||||
session.permanent = True
|
||||
|
||||
def get_collection(self):
|
||||
if session is None:
|
||||
return None
|
||||
return session.get(self.CXG_ANNO_COLLECTION)
|
||||
|
||||
def read_labels(self, data_adaptor):
|
||||
fname = self._get_filename(data_adaptor)
|
||||
if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
|
||||
return pd.read_csv(fname, dtype="category", index_col=0, header=0, comment="#", keep_default_na=False)
|
||||
else:
|
||||
return pd.DataFrame()
|
||||
|
||||
def write_labels(self, df, data_adaptor):
|
||||
# update our internal state and save it. Multi-threading often enabled,
|
||||
# so treat this as a critical section.
|
||||
with self.label_lock:
|
||||
lastmod = data_adaptor.get_last_mod_time()
|
||||
lastmodstr = "'unknown'" if lastmod is None else lastmod.isoformat(timespec="seconds")
|
||||
header = (
|
||||
f"# Annotations generated on {datetime.now().isoformat(timespec='seconds')} "
|
||||
f"using cellxgene version {cellxgene_version}\n"
|
||||
f"# Input data file was {data_adaptor.get_location()}, "
|
||||
f"which was last modified on {lastmodstr}\n"
|
||||
)
|
||||
|
||||
fname = self._get_filename(data_adaptor)
|
||||
self._backup(fname)
|
||||
if not df.empty:
|
||||
with open(fname, "w", newline="") as f:
|
||||
if header is not None:
|
||||
f.write(header)
|
||||
df.to_csv(f)
|
||||
else:
|
||||
open(fname, "w").close()
|
||||
|
||||
def _get_userid(self):
|
||||
if self.CXGUID not in session:
|
||||
session[self.CXGUID] = uuid4().hex
|
||||
session.permanent = True
|
||||
return session[self.CXGUID]
|
||||
|
||||
def _get_userdata_idhash(self, data_adaptor):
|
||||
"""
|
||||
Return a short hash that weakly identifies the user and dataset.
|
||||
Used to create safe annotations output file names.
|
||||
"""
|
||||
uid = self._get_userid()
|
||||
id = (uid + data_adaptor.get_location()).encode()
|
||||
idhash = base64.b32encode(blake2b(id, digest_size=5).digest()).decode("utf-8")
|
||||
return idhash
|
||||
|
||||
def _get_output_dir(self):
|
||||
if self.output_dir:
|
||||
return self.output_dir
|
||||
|
||||
if self.output_file:
|
||||
return os.path.dirname(self.path.abspath(self.output_dir))
|
||||
|
||||
return os.getcwd()
|
||||
|
||||
def _get_filename(self, data_adaptor):
|
||||
""" return the current annotation file name """
|
||||
if self.output_file:
|
||||
return self.output_file
|
||||
|
||||
# we need to generate a file name, which we can only do if we have a UID and collection name
|
||||
if session is None:
|
||||
raise AnnotationsError("unable to determine file name for annotations")
|
||||
|
||||
collection = self.get_collection()
|
||||
if collection is None:
|
||||
return None
|
||||
|
||||
if data_adaptor is None:
|
||||
raise AnnotationsError("unable to determine file name for annotations")
|
||||
|
||||
idhash = self._get_userdata_idhash(data_adaptor)
|
||||
return os.path.join(self._get_output_dir(), f"{collection}-{idhash}.csv")
|
||||
|
||||
def _backup(self, fname, max_backups=9):
|
||||
"""
|
||||
save N backups of file to backup_dir.
|
||||
1. fname -> backup_dir/fname-TIME
|
||||
2. delete excess files in backup_dir
|
||||
"""
|
||||
root, ext = os.path.splitext(fname)
|
||||
backup_dir = f"{root}-backups"
|
||||
|
||||
# Make sure there is work to do
|
||||
if not os.path.exists(fname):
|
||||
return
|
||||
|
||||
# Ensure backup_dir exists
|
||||
if not os.path.exists(backup_dir):
|
||||
os.mkdir(backup_dir)
|
||||
|
||||
# Save current file to backup_dir
|
||||
fname_base = os.path.basename(fname)
|
||||
fname_base_root, fname_base_ext = os.path.splitext(fname_base)
|
||||
# don't use ISO standard time format, as it contains characters illegal on some filesytems.
|
||||
nowish = datetime.now().strftime("%Y-%m-%dT%H-%M-%S")
|
||||
backup_fname = os.path.join(backup_dir, f"{fname_base_root}-{nowish}{fname_base_ext}")
|
||||
if os.path.exists(backup_fname):
|
||||
os.remove(backup_fname)
|
||||
os.rename(fname, backup_fname)
|
||||
|
||||
# prune the backup_dir to max number of backup files, keeping the most recent backups
|
||||
backups = list(filter(lambda s: s.startswith(fname_base_root), os.listdir(backup_dir)))
|
||||
excess_count = len(backups) - max_backups
|
||||
if excess_count > 0:
|
||||
backups.sort()
|
||||
for bu in backups[0:excess_count]:
|
||||
os.remove(os.path.join(backup_dir, bu))
|
||||
|
||||
def update_parameters(self, parameters, data_adaptor):
|
||||
params = {}
|
||||
params["annotations"] = True
|
||||
|
||||
if self.ontology_data:
|
||||
params["annotations_cell_ontology_enabled"] = True
|
||||
params["annotations_cell_ontology_terms"] = self.ontology_data
|
||||
else:
|
||||
params["annotations_cell_ontology_enabled"] = False
|
||||
|
||||
if self.output_file is not None:
|
||||
# user has hard-wired the name of the annotation data collection
|
||||
fname = os.path.basename(self.output_file)
|
||||
collection_fname = os.path.splitext(fname)[0]
|
||||
params["annotations-data-collection-is-read-only"] = True
|
||||
params["annotations-data-collection-name"] = collection_fname
|
||||
|
||||
elif session is not None:
|
||||
collection = self.get_collection()
|
||||
params["annotations-user-data-idhash"] = self._get_userdata_idhash(data_adaptor)
|
||||
params["annotations-data-collection-is-read-only"] = False
|
||||
params["annotations-data-collection-name"] = collection
|
||||
|
||||
parameters.update(params)
|
||||
@@ -0,0 +1,138 @@
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
from server import __version__ as cellxgene_version
|
||||
from os.path import basename, splitext
|
||||
|
||||
|
||||
class AppFeature(object):
|
||||
def __init__(self, path, available=False, method="POST", extra={}):
|
||||
self.path = path
|
||||
self.available = available
|
||||
self.method = method
|
||||
self.extra = extra
|
||||
for k, v in extra.items():
|
||||
setattr(self, k, v)
|
||||
|
||||
def todict(self):
|
||||
d = dict(
|
||||
available=self.available,
|
||||
method=self.method,
|
||||
path=self.path)
|
||||
d.update(self.extra)
|
||||
return d
|
||||
|
||||
|
||||
class AppConfig(object):
|
||||
|
||||
def __init__(self, **kw):
|
||||
super().__init__()
|
||||
|
||||
# app inputs
|
||||
self.datapath = None
|
||||
self.dataroot = None
|
||||
self.title = ""
|
||||
self.about = None
|
||||
self.scripts = []
|
||||
self.layout = None
|
||||
self.max_category_items = 100
|
||||
self.diffexp_lfc_cutoff = 0.01
|
||||
self.disable_diffexp = False
|
||||
self.anndata_backed = False
|
||||
|
||||
# TODO these options may not apply to all datasets in the multi dataset.
|
||||
# may need to invent a way to associate these config parameters with
|
||||
# specific datasets.
|
||||
self.obs_names = None
|
||||
self.var_names = None
|
||||
|
||||
# parameters
|
||||
self.diffexp_may_be_slow = False
|
||||
|
||||
inputs = ["datapath", "dataroot", "title", "about", "scripts", "layout",
|
||||
"max_category_items", "diffexp_lfc_cutoff",
|
||||
"obs_names", "var_names",
|
||||
"anndata_backed", "disable_diffexp"]
|
||||
|
||||
self.update(inputs, kw)
|
||||
|
||||
def update(self, inputs, kw):
|
||||
|
||||
for k, v in kw.items():
|
||||
if k in inputs:
|
||||
setattr(self, k, v)
|
||||
else:
|
||||
raise RuntimeError(f"unknown config parameter {k}.")
|
||||
|
||||
def get_title(self, data_adaptor):
|
||||
if self.title:
|
||||
return self.title
|
||||
|
||||
# TODO: find a place to stash the dataset title, such as a
|
||||
# json file at the same location as the data matrix.
|
||||
# for example, if the dataset is at abc.cxg then a file with
|
||||
# the title and about info could be at abc.cxg.metadata.
|
||||
# for now just return the basename
|
||||
location = data_adaptor.get_location()
|
||||
if location.endswith("/"):
|
||||
location = location[:-1]
|
||||
return splitext(basename(location))[0]
|
||||
|
||||
def get_about(self, data_adaptor):
|
||||
return self.about
|
||||
|
||||
def get_config(self, data_adaptor, annotation=None):
|
||||
|
||||
# FIXME The current set of config is not consistently presented:
|
||||
# we have camalCase, hyphen-text, and underscore_text
|
||||
|
||||
# features
|
||||
features = [f.todict() for f in data_adaptor.get_features().values()]
|
||||
|
||||
# display_names
|
||||
title = self.get_title(data_adaptor)
|
||||
about = self.get_about(data_adaptor)
|
||||
|
||||
display_names = dict(
|
||||
engine=data_adaptor.get_name(),
|
||||
dataset=title)
|
||||
|
||||
# library_versions
|
||||
library_versions = {}
|
||||
library_versions.update(data_adaptor.get_library_versions())
|
||||
library_versions["cellxgene"] = cellxgene_version
|
||||
|
||||
# links
|
||||
links = {"about-dataset" : about}
|
||||
|
||||
# parameters
|
||||
parameters = {
|
||||
"layout": self.layout,
|
||||
"max-category-items": self.max_category_items,
|
||||
"obs_names": self.obs_names,
|
||||
"var_names": self.var_names,
|
||||
"diffexp_lfc_cutoff": self.diffexp_lfc_cutoff,
|
||||
"backed": self.anndata_backed,
|
||||
"disable-diffexp": self.disable_diffexp,
|
||||
"annotations": False,
|
||||
"annotations_file": None,
|
||||
"annotations_output_dir": None,
|
||||
"annotations_cell_ontology_enabled": False,
|
||||
"annotations_cell_ontology_obopath": None,
|
||||
"annotations_cell_ontology_terms": None,
|
||||
"diffexp-may-be-slow": False,
|
||||
}
|
||||
|
||||
data_adaptor.update_parameters(parameters)
|
||||
if annotation:
|
||||
annotation.update_parameters(parameters, data_adaptor)
|
||||
|
||||
# gather it all together
|
||||
c = {}
|
||||
config = c["config"] = {}
|
||||
config["features"] = features
|
||||
config["displayNames"] = display_names
|
||||
config["library_versions"] = library_versions
|
||||
config["links"] = links
|
||||
config["parameters"] = parameters
|
||||
|
||||
return c
|
||||
@@ -0,0 +1,33 @@
|
||||
from enum import Enum
|
||||
|
||||
|
||||
DEFAULT_TOP_N = 10
|
||||
|
||||
|
||||
class AugmentedEnum(Enum):
|
||||
def __hash__(self):
|
||||
return self.value.__hash__()
|
||||
|
||||
def __eq__(self, other):
|
||||
if isinstance(other, type(self)) or isinstance(other, str):
|
||||
return self.value == other
|
||||
return False
|
||||
|
||||
def __str__(self) -> str:
|
||||
return self.value
|
||||
|
||||
|
||||
class Axis(AugmentedEnum):
|
||||
OBS = "obs"
|
||||
VAR = "var"
|
||||
|
||||
|
||||
class DiffExpMode(AugmentedEnum):
|
||||
TOP_N = "topN"
|
||||
VAR_FILTER = "varFilter"
|
||||
|
||||
|
||||
JSON_NaN_to_num_warning_msg = "JSON encoding failure - please verify all data are finite values (no NaN or Infinities)"
|
||||
REACTIVE_LIMIT = 1_000_000
|
||||
|
||||
MAX_LAYOUTS = 30
|
||||
@@ -0,0 +1,110 @@
|
||||
import os
|
||||
import tempfile
|
||||
import fsspec
|
||||
from datetime import datetime
|
||||
|
||||
|
||||
class DataLocator:
|
||||
"""
|
||||
DataLocator is a simple wrapper around fsspec functionality, and provides a
|
||||
set of functions to encapsulate a data location (URI or path), interogate
|
||||
metadata about the object at that location (size, existance, etc) and
|
||||
access the underlying data.
|
||||
|
||||
https://filesystem-spec.readthedocs.io/en/latest/index.html
|
||||
|
||||
Example:
|
||||
dl = DataLocator("/tmp/foo.h5ad")
|
||||
if dl.exists():
|
||||
print(dl.size())
|
||||
with dl.open() as f:
|
||||
thecontents = f.read()
|
||||
|
||||
DataLocator will accept a URI or native path. Error handling is as defined
|
||||
in fsspec.
|
||||
|
||||
"""
|
||||
|
||||
def __init__(self, uri_or_path):
|
||||
self.uri_or_path = uri_or_path
|
||||
self.protocol, self.path = DataLocator._get_protocol_and_path(uri_or_path)
|
||||
# work-around for LocalFileSystem not treating file: and None as the same scheme/protocol
|
||||
self.cname = self.path if self.protocol == "file" else self.uri_or_path
|
||||
# will throw RuntimeError if the protocol is unsupported
|
||||
self.fs = fsspec.filesystem(self.protocol)
|
||||
|
||||
@staticmethod
|
||||
def _get_protocol_and_path(uri_or_path):
|
||||
if "://" in uri_or_path:
|
||||
protocol, path = uri_or_path.split("://", 1)
|
||||
# windows!!! Ignore single letter drive identifiers,
|
||||
# eg, G:\foo.txt
|
||||
if len(protocol) > 1:
|
||||
return protocol, path
|
||||
return None, uri_or_path
|
||||
|
||||
def exists(self):
|
||||
return self.fs.exists(self.cname)
|
||||
|
||||
def size(self):
|
||||
return self.fs.size(self.cname)
|
||||
|
||||
def lastmodtime(self):
|
||||
""" return datetime object representing last modification time, or None if unavailable """
|
||||
info = self.fs.info(self.cname)
|
||||
if self.islocal() and info is not None:
|
||||
return datetime.fromtimestamp(info["mtime"])
|
||||
else:
|
||||
return getattr(info, "LastModified", None)
|
||||
|
||||
def abspath(self):
|
||||
"""
|
||||
return the absolute path for the locator - only really does something
|
||||
for file: protocol, as all others are already absolute
|
||||
"""
|
||||
if self.islocal():
|
||||
return os.path.abspath(self.path)
|
||||
else:
|
||||
return self.uri_or_path
|
||||
|
||||
def isfile(self):
|
||||
return self.fs.isfile(self.cname)
|
||||
|
||||
def open(self, *args):
|
||||
return self.fs.open(self.uri_or_path, *args)
|
||||
|
||||
def islocal(self):
|
||||
return self.protocol is None or self.protocol == "file"
|
||||
|
||||
def local_handle(self):
|
||||
if self.islocal():
|
||||
return LocalFilePath(self.path)
|
||||
|
||||
# if not local, create a tmp file system object to contain the data,
|
||||
# and clean it up when done. If the path has a suffix/extension,
|
||||
# do our best to create a file with the same.
|
||||
ext = os.path.splitext(self.path)
|
||||
suffix = None if ext[1] == '' else ext[1]
|
||||
with self.open() as src, tempfile.NamedTemporaryFile(prefix="cellxgene_", suffix=suffix, delete=False) as tmp:
|
||||
tmp.write(src.read())
|
||||
tmp.close()
|
||||
src.close()
|
||||
tmp_path = tmp.name
|
||||
return LocalFilePath(tmp_path, delete=True)
|
||||
|
||||
def ls(self):
|
||||
paths = self.fs.ls(self.uri_or_path)
|
||||
return [os.path.basename(p) for p in paths]
|
||||
|
||||
|
||||
class LocalFilePath:
|
||||
def __init__(self, tmp_path, delete=False):
|
||||
self.tmp_path = tmp_path
|
||||
self.delete = delete
|
||||
|
||||
def __enter__(self):
|
||||
return self.tmp_path
|
||||
|
||||
def __exit__(self, *args):
|
||||
if self.delete:
|
||||
os.unlink(self.tmp_path)
|
||||
@@ -0,0 +1,54 @@
|
||||
class FilterError(Exception):
|
||||
"""
|
||||
Raised when filter is malformed
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class JSONEncodingValueError(Exception):
|
||||
"""
|
||||
Raised when data cannot be encoded into json
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class MimeTypeError(Exception):
|
||||
"""
|
||||
Raised when incompatible MIME type selected
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class PrepareError(Exception):
|
||||
"""
|
||||
Raised when data is misprepared
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class DatasetAccessError(Exception):
|
||||
"""
|
||||
Raised when file loaded into a DataAdaptor is misformatted
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class DisabledFeatureError(Exception):
|
||||
"""
|
||||
Raised when an attempt to use a disabled feature occurs
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class AnnotationsError(Exception):
|
||||
"""
|
||||
Raised when an attempt to use the annotations feature fails
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class OntologyLoadFailure(Exception):
|
||||
"""
|
||||
Raised when reading the ontology file fails
|
||||
"""
|
||||
pass
|
||||
@@ -0,0 +1,192 @@
|
||||
from http import HTTPStatus
|
||||
import warnings
|
||||
import copy
|
||||
from flask import make_response, jsonify
|
||||
from server.common.constants import Axis, DiffExpMode, JSON_NaN_to_num_warning_msg
|
||||
from server.common.errors import (
|
||||
FilterError,
|
||||
JSONEncodingValueError,
|
||||
PrepareError,
|
||||
DisabledFeatureError,
|
||||
)
|
||||
|
||||
import json
|
||||
from server.data_common.fbs.matrix import decode_matrix_fbs
|
||||
|
||||
|
||||
def schema_get_helper(data_adaptor, annotations):
|
||||
"""helper function to gather the schema from the data source and annotations"""
|
||||
schema = data_adaptor.get_schema()
|
||||
schema = copy.deepcopy(schema)
|
||||
|
||||
# add label obs annotations as needed
|
||||
if annotations is not None:
|
||||
label_schema = annotations.get_schema(data_adaptor)
|
||||
schema["annotations"]["obs"]["columns"].extend(label_schema)
|
||||
|
||||
return schema
|
||||
|
||||
|
||||
def schema_get(data_adaptor, annotations):
|
||||
schema = schema_get_helper(data_adaptor, annotations)
|
||||
return make_response(
|
||||
jsonify({"schema": schema}), HTTPStatus.OK
|
||||
)
|
||||
|
||||
|
||||
def config_get(app_config, data_adaptor, annotations):
|
||||
config = app_config.get_config(data_adaptor, annotations)
|
||||
return make_response(make_response(jsonify(config), HTTPStatus.OK))
|
||||
|
||||
|
||||
def annotations_obs_get(request, data_adaptor, annotations):
|
||||
fields = request.args.getlist("annotation-name", None)
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
if preferred_mimetype != "application/octet-stream":
|
||||
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
|
||||
try:
|
||||
labels = None
|
||||
if annotations:
|
||||
labels = annotations.read_labels(data_adaptor)
|
||||
fbs = data_adaptor.annotation_to_fbs_matrix(Axis.OBS, fields, labels)
|
||||
return make_response(fbs, HTTPStatus.OK, {"Content-Type": "application/octet-stream"})
|
||||
except KeyError:
|
||||
return make_response(f"Error bad key in {fields}", HTTPStatus.BAD_REQUEST)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
except Exception as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
def annotations_put_fbs_helper(data_adaptor, annotations, fbs):
|
||||
"""helper function to write annotations from fbs"""
|
||||
if annotations is None:
|
||||
raise DisabledFeatureError("Writable annotations are not enabled")
|
||||
|
||||
new_label_df = decode_matrix_fbs(fbs)
|
||||
if not new_label_df.empty:
|
||||
data_adaptor.check_new_labels(new_label_df)
|
||||
annotations.write_labels(new_label_df, data_adaptor)
|
||||
|
||||
|
||||
def annotations_obs_put(request, data_adaptor, annotations):
|
||||
anno_collection = request.args.get("annotation-collection-name", default=None)
|
||||
fbs = request.get_data()
|
||||
if annotations is None:
|
||||
return make_response("Error, annotations are not configured", HTTPStatus.BAD_REQUEST)
|
||||
|
||||
if anno_collection is not None:
|
||||
if not annotations.is_safe_collection_name(anno_collection):
|
||||
return make_response(f"Error, bad annotation collection name", HTTPStatus.BAD_REQUEST)
|
||||
annotations.set_collection(anno_collection)
|
||||
|
||||
try:
|
||||
annotations_put_fbs_helper(data_adaptor, annotations, fbs)
|
||||
res = json.dumps({"status": "OK"})
|
||||
return make_response(res, HTTPStatus.OK, {"Content-Type": "application/json"})
|
||||
except (ValueError, DisabledFeatureError, KeyError) as e:
|
||||
return make_response(str(e), HTTPStatus.BAD_REQUEST)
|
||||
except Exception as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
def annotations_var_get(request, data_adaptor, annotations):
|
||||
fields = request.args.getlist("annotation-name", None)
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
if preferred_mimetype != "application/octet-stream":
|
||||
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
|
||||
try:
|
||||
labels = None
|
||||
if annotations is not None:
|
||||
labels = annotations.read_labels(data_adaptor)
|
||||
return make_response(
|
||||
data_adaptor.annotation_to_fbs_matrix(Axis.VAR, fields, labels),
|
||||
HTTPStatus.OK,
|
||||
{"Content-Type": "application/octet-stream"},
|
||||
)
|
||||
except KeyError:
|
||||
return make_response(f"Error bad key in {fields}", HTTPStatus.BAD_REQUEST)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
except Exception as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
def data_var_put(request, data_adaptor):
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
if preferred_mimetype != "application/octet-stream":
|
||||
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
|
||||
|
||||
filter_json = request.get_json()
|
||||
filter = filter_json["filter"] if filter_json else None
|
||||
try:
|
||||
return make_response(
|
||||
data_adaptor.data_frame_to_fbs_matrix(filter, axis=Axis.VAR),
|
||||
HTTPStatus.OK,
|
||||
{"Content-Type": "application/octet-stream"},
|
||||
)
|
||||
except FilterError as e:
|
||||
return make_response(str(e), HTTPStatus.BAD_REQUEST)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
def diffexp_obs_post(request, data_adaptor):
|
||||
args = request.get_json()
|
||||
# confirm mode is present and legal
|
||||
try:
|
||||
mode = DiffExpMode(args["mode"])
|
||||
except KeyError:
|
||||
return make_response("Error: mode is required", HTTPStatus.BAD_REQUEST)
|
||||
except ValueError:
|
||||
return make_response(f"Error: invalid mode option {args['mode']}", HTTPStatus.BAD_REQUEST)
|
||||
# Validate filters
|
||||
if mode == DiffExpMode.VAR_FILTER or "varFilter" in args:
|
||||
# not NOT_IMPLEMENTED
|
||||
return make_response("mode=varfilter not implemented", HTTPStatus.NOT_IMPLEMENTED)
|
||||
if mode == DiffExpMode.TOP_N and "count" not in args:
|
||||
return make_response("mode=topN requires a count parameter", HTTPStatus.BAD_REQUEST)
|
||||
|
||||
if "set1" not in args:
|
||||
return make_response("set1 is required.", HTTPStatus.BAD_REQUEST)
|
||||
if Axis.VAR in args["set1"]["filter"]:
|
||||
return make_response("Var filter not allowed for set1", HTTPStatus.BAD_REQUEST)
|
||||
# set2
|
||||
if "set2" not in args:
|
||||
return make_response("Set2 as inverse of set1 is not implemented", HTTPStatus.NOT_IMPLEMENTED)
|
||||
if Axis.VAR in args["set2"]["filter"]:
|
||||
return make_response("Var filter not allowed for set2", HTTPStatus.BAD_REQUEST)
|
||||
|
||||
set1_filter = args["set1"]["filter"]
|
||||
set2_filter = args.get("set2", {"filter": {}})["filter"]
|
||||
|
||||
# TODO: implement varfilter mode
|
||||
|
||||
# mode=topN
|
||||
count = args.get("count", None)
|
||||
try:
|
||||
diffexp = data_adaptor.diffexp_topN(set1_filter, set2_filter, count)
|
||||
return make_response(diffexp, HTTPStatus.OK, {"Content-Type": "application/json"})
|
||||
except (ValueError, FilterError) as e:
|
||||
return make_response(str(e), HTTPStatus.BAD_REQUEST)
|
||||
except JSONEncodingValueError as e:
|
||||
# JSON encoding failure, usually due to bad data
|
||||
warnings.warn(JSON_NaN_to_num_warning_msg)
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
|
||||
|
||||
def layout_obs_get(request, data_adaptor):
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
try:
|
||||
if preferred_mimetype == "application/octet-stream":
|
||||
return make_response(
|
||||
data_adaptor.layout_to_fbs_matrix(), HTTPStatus.OK, {"Content-Type": "application/octet-stream"}
|
||||
)
|
||||
else:
|
||||
return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
|
||||
except PrepareError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
except ValueError as e:
|
||||
return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
|
||||
@@ -0,0 +1,146 @@
|
||||
import contextlib
|
||||
import errno
|
||||
import socket
|
||||
from urllib.parse import urlsplit, urljoin
|
||||
import os
|
||||
from flask import json
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import warnings
|
||||
|
||||
|
||||
def find_available_port(host, port=5005):
|
||||
"""
|
||||
Helper method to find open port on host. Tries 5000 ports incremented from the specified port
|
||||
"""
|
||||
# Takes approx 2 seconds to do a scan of 5000 ports on my laptop
|
||||
num_ports_to_try = 5000
|
||||
for port_to_try in range(port, port + num_ports_to_try):
|
||||
if is_port_available(host, port_to_try):
|
||||
return port_to_try
|
||||
raise socket.error(errno.EADDRINUSE, f"No port in range {port} - {port + num_ports_to_try - 1} available.")
|
||||
|
||||
|
||||
def is_port_available(host, port):
|
||||
is_available = False
|
||||
with contextlib.closing(socket.socket(socket.AF_INET, socket.SOCK_STREAM)) as s:
|
||||
try:
|
||||
s.bind((host, port))
|
||||
is_available = True
|
||||
except socket.error:
|
||||
pass
|
||||
return is_available
|
||||
|
||||
|
||||
def sort_options(command):
|
||||
"""
|
||||
Helper for the click options - will sort options in a command, and can
|
||||
be used as a decorator.
|
||||
"""
|
||||
command.params.sort(key=lambda p: p.name)
|
||||
return command
|
||||
|
||||
|
||||
def path_join(base, *urls):
|
||||
"""
|
||||
this is like urllib.parse.urljoin, except it works around the scheme-specific
|
||||
cleverness in the aforementioned code, ignores anything in the url except the path,
|
||||
and accepts more than one url.
|
||||
"""
|
||||
if not base.endswith("/"):
|
||||
base += "/"
|
||||
btpl = urlsplit(base)
|
||||
path = btpl.path
|
||||
for url in urls:
|
||||
utpl = urlsplit(url)
|
||||
if btpl.scheme == "":
|
||||
path = os.path.join(path, utpl.path)
|
||||
path = os.path.normpath(path)
|
||||
else:
|
||||
path = urljoin(path, utpl.path)
|
||||
return btpl._replace(path=path).geturl()
|
||||
|
||||
|
||||
class Float32JSONEncoder(json.JSONEncoder):
|
||||
def __init__(self, *args, **kwargs):
|
||||
"""
|
||||
NaN/Infinities are illegal in standard JSON. Python extends JSON with
|
||||
non-standard symbols that most JavaScript JSON parsers do not understand.
|
||||
The `allow_nan` parameter will force Python simplejson to throw an ValueError
|
||||
if it runs into non-finite floating point values which are unsupported by
|
||||
standard JSON.
|
||||
"""
|
||||
kwargs["allow_nan"] = False
|
||||
super().__init__(*args, **kwargs)
|
||||
|
||||
def default(self, obj):
|
||||
if isinstance(obj, np.float32):
|
||||
return float(obj)
|
||||
elif isinstance(obj, np.integer):
|
||||
return int(obj)
|
||||
return json.JSONEncoder.default(self, obj)
|
||||
|
||||
|
||||
def custom_format_warning(msg, *args, **kwargs):
|
||||
return f"[cellxgene] Warning: {msg} \n"
|
||||
|
||||
|
||||
def jsonify_numpy(data):
|
||||
return json.dumps(data, cls=Float32JSONEncoder, allow_nan=False)
|
||||
|
||||
|
||||
def dtype_to_schema(dtype):
|
||||
schema = {}
|
||||
if dtype == np.float32:
|
||||
schema['type'] = 'float32'
|
||||
elif dtype == np.int32:
|
||||
schema['type'] = 'int32'
|
||||
elif dtype == np.bool_:
|
||||
schema['type'] = 'boolean'
|
||||
elif dtype == np.str:
|
||||
schema['type'] = 'string'
|
||||
elif dtype == "category":
|
||||
schema["type"] = "categorical"
|
||||
schema["categories"] = dtype.categories.tolist()
|
||||
else:
|
||||
raise TypeError(
|
||||
f"Annotations of type {dtype} are unsupported."
|
||||
)
|
||||
return schema
|
||||
|
||||
|
||||
def can_cast_to_float32(array):
|
||||
if array.dtype.kind == "f":
|
||||
if not np.can_cast(array.dtype, np.float32):
|
||||
warnings.warn(f"Annotation {array.name} will be converted to 32 bit float and may lose precision.")
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def can_cast_to_int32(array):
|
||||
if array.dtype.kind in ["i", "u"]:
|
||||
if np.can_cast(array.dtype, np.int32):
|
||||
return True
|
||||
ii32 = np.iinfo(np.int32)
|
||||
if array.min() >= ii32.min and array.max() <= ii32.max:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def series_to_schema(array):
|
||||
assert type(array) == pd.Series
|
||||
try:
|
||||
return dtype_to_schema(array.dtype)
|
||||
except TypeError:
|
||||
dtype = array.dtype
|
||||
data_kind = dtype.kind
|
||||
schema = {}
|
||||
if can_cast_to_float32(array):
|
||||
schema["type"] = "float32"
|
||||
elif can_cast_to_int32(array):
|
||||
schema["type"] = "int32"
|
||||
elif data_kind == "O" and dtype == "object":
|
||||
schema["type"] = "string"
|
||||
else:
|
||||
raise TypeError(f"Annotations of type {dtype} are unsupported.")
|
||||
return schema
|
||||
Reference in New Issue
Block a user