wire e2e gene set loading prototype

This commit is contained in:
bkmartinjr
2021-01-21 15:01:10 -08:00
parent 7f1bb3e21e
commit 0001928317
11 changed files with 279 additions and 3280 deletions
+11 -1
View File
@@ -8,7 +8,7 @@ from server.common.utils.type_conversion_utils import get_schema_type_hint_of_ar
class Annotations(metaclass=ABCMeta):
""" baseclass for annotations, including ontologies"""
""" baseclass for annotations, including ontologies and gene sets"""
""" our default ontology is the PURL for the Cell Ontology.
See http://www.obofoundry.org/ontology/cl.html """
@@ -64,6 +64,16 @@ class Annotations(metaclass=ABCMeta):
"""Write the labels (df) to a persistent storage such that it can later be read"""
pass
@abstractmethod
def read_genesets(self, data_adaptor):
"""Return the genesets as a list of list, ie, [['gsname', ['gene1', 'gene2']], ...] """
pass
@abstractmethod
def write_genesets(self, gs, data_adaptor):
"""Write the genesets (gs) to a persistent storage such that it can later be read"""
pass
@abstractmethod
def update_parameters(self, parameters, data_adaptor):
"""Update configuration parameters that describe information about the annotations feature"""
+83 -14
View File
@@ -4,6 +4,7 @@ import re
import threading
from datetime import datetime
from hashlib import blake2b
import csv
import pandas as pd
from flask import session, has_request_context, current_app
@@ -16,14 +17,16 @@ from server.common.errors import AnnotationsError
class AnnotationsLocalFile(Annotations):
CXG_ANNO_COLLECTION = "cxg_anno_collection"
def __init__(self, output_dir, output_file):
def __init__(self, output_dir, label_output_file, genesets_output_file):
super().__init__()
self.output_dir = output_dir
self.output_file = output_file
# lock used to protect label file write ops
self.label_output_file = label_output_file
self.genesets_output_file = genesets_output_file
# lock used to protect label and geneset file write ops
self.label_lock = threading.RLock()
self.genesets_lock = threading.RLock()
# cache the most recent annotations
# cache the most recent annotations. We don't cache genesets as they are small
self.last_fname = None
self.last_labels = None
@@ -51,7 +54,7 @@ class AnnotationsLocalFile(Annotations):
if not current_app.auth.is_user_authenticated():
return pd.DataFrame()
fname = self._get_filename(data_adaptor)
fname = self._get_celllabels_filename(data_adaptor)
with self.label_lock:
if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
# returned the cached labels if possible, otherwise read them from the file
@@ -81,7 +84,7 @@ class AnnotationsLocalFile(Annotations):
f"which was last modified on {lastmodstr}\n"
)
fname = self._get_filename(data_adaptor)
fname = self._get_celllabels_filename(data_adaptor)
self._backup(fname)
if not df.empty:
with open(fname, "w", newline="") as f:
@@ -95,6 +98,62 @@ class AnnotationsLocalFile(Annotations):
self.last_fname = fname
self.last_labels = df
def read_genesets(self, data_adaptor):
if has_request_context():
if not current_app.auth.is_user_authenticated():
return {}
fname = self._get_genesets_filename(data_adaptor)
gs = []
with self.genesets_lock:
if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
with open(fname, newline="") as f:
sample = f.read(1024)
f.seek(0)
sniffer = csv.Sniffer()
dialect = sniffer.sniff(sample)
dialect.skipinitialspace = True
reader = csv.reader(f, dialect)
haveReadHeader = False
for row in reader:
if len(row) == 0:
continue
# if row starts with '#' it is a comment
if row[0].startswith("#"):
continue
# if this is the first non-comment row, assume it is a header
if not haveReadHeader:
haveReadHeader = True
continue
gs.append([row[0], row[1:]])
return gs
def write_genesets(self, genesets, data_adaptor):
with self.genesets_lock:
lastmod = data_adaptor.get_last_mod_time()
lastmodstr = "'unknown'" if lastmod is None else lastmod.isoformat(timespec="seconds")
header = (
f"# Geneset generated on {datetime.now().isoformat(timespec='seconds')} "
f"using cellxgene version {cellxgene_version}\n"
f"# Input data file was {data_adaptor.get_location()}, "
f"which was last modified on {lastmodstr}\n"
)
fname = self._get_genesets_filename(data_adaptor)
self._backup(fname)
if len(genesets) > 0:
gsRows = [[gs[0]] + gs[1] for gs in genesets]
with open(fname, "w", newline="") as f:
if header is not None:
f.write(header)
writer = csv.writer(f)
writer.writerow(["name", "genes..."]) # CSV column header row
writer.writerows(gsRows)
else:
open(fname, "w").close()
def _get_userdata_idhash(self, data_adaptor):
"""
Return a short hash that weakly identifies the user and dataset.
@@ -109,16 +168,26 @@ class AnnotationsLocalFile(Annotations):
if self.output_dir:
return self.output_dir
if self.output_file:
if self.label_output_file:
return os.path.dirname(self.path.abspath(self.output_dir))
return os.getcwd()
def _get_filename(self, data_adaptor):
def _get_celllabels_filename(self, data_adaptor):
""" return the current annotation file name """
if self.output_file:
return self.output_file
if self.label_output_file:
return self.label_output_file
return self._get_filename(data_adaptor, "celllabels")
def _get_genesets_filename(self, data_adaptor):
""" return the current annotation file name """
if self.genesets_output_file:
return self.genesets_output_file
return self._get_filename(data_adaptor, "genesets")
def _get_filename(self, data_adaptor, anno_name):
# we need to generate a file name, which we can only do if we have a UID and collection name
if session is None:
raise AnnotationsError("unable to determine file name for annotations")
@@ -131,7 +200,7 @@ class AnnotationsLocalFile(Annotations):
raise AnnotationsError("unable to determine file name for annotations")
idhash = self._get_userdata_idhash(data_adaptor)
return os.path.join(self._get_output_dir(), f"{collection}-{idhash}.csv")
return os.path.join(self._get_output_dir(), f"{collection}-{anno_name}-{idhash}.csv")
def _backup(self, fname, max_backups=9):
"""
@@ -179,9 +248,9 @@ class AnnotationsLocalFile(Annotations):
else:
params["annotations_cell_ontology_enabled"] = False
if self.output_file is not None:
# user has hard-wired the name of the annotation data collection
fname = os.path.basename(self.output_file)
if self.label_output_file is not None:
# user has hard-wired the name of the annotation cell label data collection
fname = os.path.basename(self.label_output_file)
collection_fname = os.path.splitext(fname)[0]
params["annotations-data-collection-is-read-only"] = True
params["annotations-data-collection-name"] = collection_fname
+1
View File
@@ -44,6 +44,7 @@ def get_client_config(app_config, data_adaptor):
"annotations": False,
"annotations_file": None,
"annotations_dir": None,
"annotations_genesets_readonly": dataset_config.user_annotations__genesets__readonly,
"annotations_cell_ontology_enabled": False,
"annotations_cell_ontology_obopath": None,
"annotations_cell_ontology_terms": None,
+35 -11
View File
@@ -42,6 +42,10 @@ class DatasetConfig(BaseConfig):
self.user_annotations__hosted_tiledb_array__hosted_file_directory = default_config["user_annotations"][
"hosted_tiledb_array"
]["hosted_file_directory"]
self.user_annotations__genesets__readonly = default_config["user_annotations"]["genesets"]["readonly"]
self.user_annotations__local_file_csv__genesets_file = default_config["user_annotations"]["local_file_csv"][
"genesets_file"
]
self.embeddings__names = default_config["embeddings"]["names"]
self.embeddings__enable_reembedding = default_config["embeddings"]["enable_reembedding"]
@@ -98,6 +102,9 @@ class DatasetConfig(BaseConfig):
self.validate_correct_type_of_configuration_attribute(
"user_annotations__local_file_csv__file", (type(None), str)
)
self.validate_correct_type_of_configuration_attribute(
"user_annotations__local_file_csv__genesets_file", (type(None), str)
)
self.validate_correct_type_of_configuration_attribute("user_annotations__ontology__enable", bool)
self.validate_correct_type_of_configuration_attribute(
"user_annotations__ontology__obo_location", (type(None), str)
@@ -108,6 +115,8 @@ class DatasetConfig(BaseConfig):
self.validate_correct_type_of_configuration_attribute(
"user_annotations__hosted_tiledb_array__hosted_file_directory", (type(None), str)
)
self.validate_correct_type_of_configuration_attribute("user_annotations__genesets__readonly", bool)
if self.user_annotations__enable:
server_config = self.app_config.server_config
if not self.app__authentication_enable:
@@ -132,14 +141,22 @@ class DatasetConfig(BaseConfig):
def handle_local_file_csv_annotations(self):
dirname = self.user_annotations__local_file_csv__directory
filename = self.user_annotations__local_file_csv__file
if filename is not None and dirname is not None:
annotation_filename = self.user_annotations__local_file_csv__file
if annotation_filename is not None and dirname is not None:
raise ConfigurationError("'annotations-file' and 'annotations-dir' may not be used together.")
genesets_filename = self.user_annotations__local_file_csv__genesets_file
if genesets_filename is not None and dirname is not None:
raise ConfigurationError("'genesets-file' and 'annotations-dir' may not be used together.")
if filename is not None:
lf_name, lf_ext = splitext(filename)
if annotation_filename is not None:
lf_name, lf_ext = splitext(annotation_filename)
if lf_ext and lf_ext != ".csv":
raise ConfigurationError(f"annotation file type must be .csv: {filename}")
raise ConfigurationError(f"annotation file type must be .csv: {annotation_filename}")
if genesets_filename is not None:
lf_name, lf_ext = splitext(genesets_filename)
if lf_ext and lf_ext != ".csv":
raise ConfigurationError(f"genesets file type must be .csv: {genesets_filename}")
if dirname is not None and not isdir(dirname):
try:
@@ -147,16 +164,23 @@ class DatasetConfig(BaseConfig):
except OSError:
raise ConfigurationError("Unable to create directory specified by --annotations-dir")
self.user_annotations = AnnotationsLocalFile(dirname, filename)
self.user_annotations = AnnotationsLocalFile(dirname, annotation_filename, genesets_filename)
# if the user has specified a fixed label file, go ahead and validate it
# so that we can remove errors early in the process.
server_config = self.app_config.server_config
if server_config.single_dataset__datapath and self.user_annotations__local_file_csv__file:
with server_config.matrix_data_cache_manager.data_adaptor(
self.tag, server_config.single_dataset__datapath, self.app_config
) as data_adaptor:
data_adaptor.check_new_labels(self.user_annotations.read_labels(data_adaptor))
if server_config.single_dataset__datapath:
if self.user_annotations__local_file_csv__file:
with server_config.matrix_data_cache_manager.data_adaptor(
self.tag, server_config.single_dataset__datapath, self.app_config
) as data_adaptor:
data_adaptor.check_new_labels(self.user_annotations.read_labels(data_adaptor))
if self.user_annotations__local_file_csv__genesets_file:
with server_config.matrix_data_cache_manager.data_adaptor(
self.tag, server_config.single_dataset__datapath, self.app_config
) as data_adaptor:
data_adaptor.check_new_genesets(self.user_annotations.read_genesets(data_adaptor))
def handle_hosted_tiledb_annotations(self):
self.validate_correct_type_of_configuration_attribute("user_annotations__hosted_tiledb_array__db_uri", str)
+40 -3
View File
@@ -196,9 +196,6 @@ def annotations_var_get(request, data_adaptor):
try:
labels = None
annotations = data_adaptor.dataset_config.user_annotations
if annotations is not None:
labels = annotations.read_labels(data_adaptor)
return make_response(
data_adaptor.annotation_to_fbs_matrix(Axis.VAR, fields, labels),
HTTPStatus.OK,
@@ -328,3 +325,43 @@ def layout_obs_put(request, data_adaptor):
return abort_and_log(HTTPStatus.NOT_IMPLEMENTED, str(e))
except (ValueError, DisabledFeatureError, FilterError) as e:
return abort_and_log(HTTPStatus.BAD_REQUEST, str(e), include_exc_info=True)
def genesets_get(request, data_adaptor):
preferred_mimetype = request.accept_mimetypes.best_match(["application/json"])
if preferred_mimetype != "application/json":
return abort(HTTPStatus.NOT_ACCEPTABLE)
annotations = data_adaptor.dataset_config.user_annotations
genesets = annotations.read_genesets(data_adaptor)
return make_response(jsonify({"genesets": genesets}), HTTPStatus.OK)
def genesets_put(request, data_adaptor):
annotations = data_adaptor.dataset_config.user_annotations
if annotations is None:
return abort(HTTPStatus.NOT_IMPLEMENTED)
if data_adaptor.dataset_config.user_annotations__genesets__readonly:
return abort(HTTPStatus.NOT_IMPLEMENTED)
anno_collection = request.args.get("annotation-collection-name", default=None)
if anno_collection is not None:
if not annotations.is_safe_collection_name(anno_collection):
return abort(HTTPStatus.BAD_REQUEST, "Bad annotation collection name")
annotations.set_collection(anno_collection)
args = request.get_json()
try:
genesets = args["genesets"]
if type(genesets) is list:
gs = data_adaptor.check_new_genesets(genesets)
annotations.write_genesets(gs, data_adaptor)
else:
pass
annotations.write_genesets([], data_adaptor)
res = json.dumps({"status": "OK"})
return make_response(res, HTTPStatus.OK, {"Content-Type": "application/json"})
except (ValueError, DisabledFeatureError, KeyError) as e:
return abort_and_log(HTTPStatus.BAD_REQUEST, str(e), include_exc_info=True)