wire e2e gene set loading prototype

This commit is contained in:
bkmartinjr
2021-01-21 15:01:10 -08:00
parent 7f1bb3e21e
commit 0001928317
11 changed files with 279 additions and 3280 deletions
+11
View File
@@ -52,6 +52,16 @@ async function userInfoFetch(dispatch) {
}); });
} }
async function genesetsFetch(dispatch) {
return fetchJson("genesets").then((response) => {
const genesets = response?.genesets ?? {};
dispatch({
type: "geneset: initial load",
init: genesets,
});
})
}
function prefetchEmbeddings(annoMatrix) { function prefetchEmbeddings(annoMatrix) {
/* /*
prefetch requests for all embeddings prefetch requests for all embeddings
@@ -74,6 +84,7 @@ const doInitialDataLoad = () =>
schemaFetch(dispatch), schemaFetch(dispatch),
userColorsFetchAndLoad(dispatch), userColorsFetchAndLoad(dispatch),
userInfoFetch(dispatch), userInfoFetch(dispatch),
genesetsFetch(dispatch),
]); ]);
const baseDataUrl = `${globals.API.prefix}${globals.API.version}`; const baseDataUrl = `${globals.API.prefix}${globals.API.version}`;
File diff suppressed because it is too large Load Diff
+14
View File
@@ -318,6 +318,19 @@ class LayoutObsAPI(DatasetResource):
return common_rest.layout_obs_put(request, data_adaptor) return common_rest.layout_obs_put(request, data_adaptor)
class GenesetsAPI(DatasetResource):
@cache_control(public=True, no_store=True)
@rest_get_data_adaptor
def get(self, data_adaptor):
return common_rest.genesets_get(request, data_adaptor)
@requires_authentication
@cache_control(no_store=True)
@rest_get_data_adaptor
def put(self, data_adaptor):
return common_rest.genesets_put(request, data_adaptor)
def get_api_base_resources(bp_base): def get_api_base_resources(bp_base):
"""Add resources that are accessed from the api_base_url""" """Add resources that are accessed from the api_base_url"""
api = Api(bp_base) api = Api(bp_base)
@@ -343,6 +356,7 @@ def get_api_dataroot_resources(bp_dataroot, url_dataroot=None):
add_resource(AnnotationsObsAPI, "/annotations/obs") add_resource(AnnotationsObsAPI, "/annotations/obs")
add_resource(AnnotationsVarAPI, "/annotations/var") add_resource(AnnotationsVarAPI, "/annotations/var")
add_resource(DataVarAPI, "/data/var") add_resource(DataVarAPI, "/data/var")
add_resource(GenesetsAPI, "/genesets")
# Display routes # Display routes
add_resource(ColorsAPI, "/colors") add_resource(ColorsAPI, "/colors")
# Computation routes # Computation routes
+21 -2
View File
@@ -36,12 +36,12 @@ def annotation_args(func):
) )
@click.option( @click.option(
"--annotations-dir", "--annotations-dir",
"--data-dir",
default=DEFAULT_CONFIG.default_dataset_config.user_annotations__local_file_csv__directory, default=DEFAULT_CONFIG.default_dataset_config.user_annotations__local_file_csv__directory,
show_default=False, show_default=False,
multiple=False, multiple=False,
metavar="<directory path>", metavar="<directory path>",
help="Directory of where to save output annotations; filename will be specified in the application. " help="Directory of where to save annotations and gene sets; filename will be specified in the application. Incompatible with --annotations-file and --genesets-file.",
"Incompatible with --annotations-file.",
) )
@click.option( @click.option(
"--experimental-annotations-ontology", "--experimental-annotations-ontology",
@@ -57,6 +57,21 @@ def annotation_args(func):
metavar="<path or url>", metavar="<path or url>",
help="Location of OBO file defining cell annotation autosuggest terms.", help="Location of OBO file defining cell annotation autosuggest terms.",
) )
@click.option(
"--disable-genesets-save",
is_flag=True,
default=not DEFAULT_CONFIG.default_dataset_config.user_annotations__genesets__readonly,
show_default=False,
help="Disable saving gene sets. If disabled, users will be able to make changes to gene sets but all changes will be lost on browser refresh.",
)
@click.option(
"--genesets-file",
default=DEFAULT_CONFIG.default_dataset_config.user_annotations__local_file_csv__genesets_file,
show_default=True,
multiple=False,
metavar="<path>",
help="CSV file to initialize editing of gene sets; will be altered in-place. Incompatible with --data-dir.",
)
@functools.wraps(func) @functools.wraps(func)
def wrapper(*args, **kwargs): def wrapper(*args, **kwargs):
return func(*args, **kwargs) return func(*args, **kwargs)
@@ -325,6 +340,8 @@ def launch(
disable_annotations, disable_annotations,
annotations_file, annotations_file,
annotations_dir, annotations_dir,
genesets_file,
disable_genesets_save,
backed, backed,
disable_diffexp, disable_diffexp,
experimental_annotations_ontology, experimental_annotations_ontology,
@@ -389,6 +406,8 @@ def launch(
user_annotations__enable=not disable_annotations, user_annotations__enable=not disable_annotations,
user_annotations__local_file_csv__file=annotations_file, user_annotations__local_file_csv__file=annotations_file,
user_annotations__local_file_csv__directory=annotations_dir, user_annotations__local_file_csv__directory=annotations_dir,
user_annotations__local_file_csv__genesets_file=genesets_file,
user_annotations__genesets__readonly=disable_genesets_save,
user_annotations__ontology__enable=experimental_annotations_ontology, user_annotations__ontology__enable=experimental_annotations_ontology,
user_annotations__ontology__obo_location=experimental_annotations_ontology_obo, user_annotations__ontology__obo_location=experimental_annotations_ontology_obo,
presentation__max_categories=max_category_items, presentation__max_categories=max_category_items,
+11 -1
View File
@@ -8,7 +8,7 @@ from server.common.utils.type_conversion_utils import get_schema_type_hint_of_ar
class Annotations(metaclass=ABCMeta): class Annotations(metaclass=ABCMeta):
""" baseclass for annotations, including ontologies""" """ baseclass for annotations, including ontologies and gene sets"""
""" our default ontology is the PURL for the Cell Ontology. """ our default ontology is the PURL for the Cell Ontology.
See http://www.obofoundry.org/ontology/cl.html """ See http://www.obofoundry.org/ontology/cl.html """
@@ -64,6 +64,16 @@ class Annotations(metaclass=ABCMeta):
"""Write the labels (df) to a persistent storage such that it can later be read""" """Write the labels (df) to a persistent storage such that it can later be read"""
pass pass
@abstractmethod
def read_genesets(self, data_adaptor):
"""Return the genesets as a list of list, ie, [['gsname', ['gene1', 'gene2']], ...] """
pass
@abstractmethod
def write_genesets(self, gs, data_adaptor):
"""Write the genesets (gs) to a persistent storage such that it can later be read"""
pass
@abstractmethod @abstractmethod
def update_parameters(self, parameters, data_adaptor): def update_parameters(self, parameters, data_adaptor):
"""Update configuration parameters that describe information about the annotations feature""" """Update configuration parameters that describe information about the annotations feature"""
+83 -14
View File
@@ -4,6 +4,7 @@ import re
import threading import threading
from datetime import datetime from datetime import datetime
from hashlib import blake2b from hashlib import blake2b
import csv
import pandas as pd import pandas as pd
from flask import session, has_request_context, current_app from flask import session, has_request_context, current_app
@@ -16,14 +17,16 @@ from server.common.errors import AnnotationsError
class AnnotationsLocalFile(Annotations): class AnnotationsLocalFile(Annotations):
CXG_ANNO_COLLECTION = "cxg_anno_collection" CXG_ANNO_COLLECTION = "cxg_anno_collection"
def __init__(self, output_dir, output_file): def __init__(self, output_dir, label_output_file, genesets_output_file):
super().__init__() super().__init__()
self.output_dir = output_dir self.output_dir = output_dir
self.output_file = output_file self.label_output_file = label_output_file
# lock used to protect label file write ops self.genesets_output_file = genesets_output_file
# lock used to protect label and geneset file write ops
self.label_lock = threading.RLock() self.label_lock = threading.RLock()
self.genesets_lock = threading.RLock()
# cache the most recent annotations # cache the most recent annotations. We don't cache genesets as they are small
self.last_fname = None self.last_fname = None
self.last_labels = None self.last_labels = None
@@ -51,7 +54,7 @@ class AnnotationsLocalFile(Annotations):
if not current_app.auth.is_user_authenticated(): if not current_app.auth.is_user_authenticated():
return pd.DataFrame() return pd.DataFrame()
fname = self._get_filename(data_adaptor) fname = self._get_celllabels_filename(data_adaptor)
with self.label_lock: with self.label_lock:
if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0: if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
# returned the cached labels if possible, otherwise read them from the file # returned the cached labels if possible, otherwise read them from the file
@@ -81,7 +84,7 @@ class AnnotationsLocalFile(Annotations):
f"which was last modified on {lastmodstr}\n" f"which was last modified on {lastmodstr}\n"
) )
fname = self._get_filename(data_adaptor) fname = self._get_celllabels_filename(data_adaptor)
self._backup(fname) self._backup(fname)
if not df.empty: if not df.empty:
with open(fname, "w", newline="") as f: with open(fname, "w", newline="") as f:
@@ -95,6 +98,62 @@ class AnnotationsLocalFile(Annotations):
self.last_fname = fname self.last_fname = fname
self.last_labels = df self.last_labels = df
def read_genesets(self, data_adaptor):
if has_request_context():
if not current_app.auth.is_user_authenticated():
return {}
fname = self._get_genesets_filename(data_adaptor)
gs = []
with self.genesets_lock:
if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
with open(fname, newline="") as f:
sample = f.read(1024)
f.seek(0)
sniffer = csv.Sniffer()
dialect = sniffer.sniff(sample)
dialect.skipinitialspace = True
reader = csv.reader(f, dialect)
haveReadHeader = False
for row in reader:
if len(row) == 0:
continue
# if row starts with '#' it is a comment
if row[0].startswith("#"):
continue
# if this is the first non-comment row, assume it is a header
if not haveReadHeader:
haveReadHeader = True
continue
gs.append([row[0], row[1:]])
return gs
def write_genesets(self, genesets, data_adaptor):
with self.genesets_lock:
lastmod = data_adaptor.get_last_mod_time()
lastmodstr = "'unknown'" if lastmod is None else lastmod.isoformat(timespec="seconds")
header = (
f"# Geneset generated on {datetime.now().isoformat(timespec='seconds')} "
f"using cellxgene version {cellxgene_version}\n"
f"# Input data file was {data_adaptor.get_location()}, "
f"which was last modified on {lastmodstr}\n"
)
fname = self._get_genesets_filename(data_adaptor)
self._backup(fname)
if len(genesets) > 0:
gsRows = [[gs[0]] + gs[1] for gs in genesets]
with open(fname, "w", newline="") as f:
if header is not None:
f.write(header)
writer = csv.writer(f)
writer.writerow(["name", "genes..."]) # CSV column header row
writer.writerows(gsRows)
else:
open(fname, "w").close()
def _get_userdata_idhash(self, data_adaptor): def _get_userdata_idhash(self, data_adaptor):
""" """
Return a short hash that weakly identifies the user and dataset. Return a short hash that weakly identifies the user and dataset.
@@ -109,16 +168,26 @@ class AnnotationsLocalFile(Annotations):
if self.output_dir: if self.output_dir:
return self.output_dir return self.output_dir
if self.output_file: if self.label_output_file:
return os.path.dirname(self.path.abspath(self.output_dir)) return os.path.dirname(self.path.abspath(self.output_dir))
return os.getcwd() return os.getcwd()
def _get_filename(self, data_adaptor): def _get_celllabels_filename(self, data_adaptor):
""" return the current annotation file name """ """ return the current annotation file name """
if self.output_file: if self.label_output_file:
return self.output_file return self.label_output_file
return self._get_filename(data_adaptor, "celllabels")
def _get_genesets_filename(self, data_adaptor):
""" return the current annotation file name """
if self.genesets_output_file:
return self.genesets_output_file
return self._get_filename(data_adaptor, "genesets")
def _get_filename(self, data_adaptor, anno_name):
# we need to generate a file name, which we can only do if we have a UID and collection name # we need to generate a file name, which we can only do if we have a UID and collection name
if session is None: if session is None:
raise AnnotationsError("unable to determine file name for annotations") raise AnnotationsError("unable to determine file name for annotations")
@@ -131,7 +200,7 @@ class AnnotationsLocalFile(Annotations):
raise AnnotationsError("unable to determine file name for annotations") raise AnnotationsError("unable to determine file name for annotations")
idhash = self._get_userdata_idhash(data_adaptor) idhash = self._get_userdata_idhash(data_adaptor)
return os.path.join(self._get_output_dir(), f"{collection}-{idhash}.csv") return os.path.join(self._get_output_dir(), f"{collection}-{anno_name}-{idhash}.csv")
def _backup(self, fname, max_backups=9): def _backup(self, fname, max_backups=9):
""" """
@@ -179,9 +248,9 @@ class AnnotationsLocalFile(Annotations):
else: else:
params["annotations_cell_ontology_enabled"] = False params["annotations_cell_ontology_enabled"] = False
if self.output_file is not None: if self.label_output_file is not None:
# user has hard-wired the name of the annotation data collection # user has hard-wired the name of the annotation cell label data collection
fname = os.path.basename(self.output_file) fname = os.path.basename(self.label_output_file)
collection_fname = os.path.splitext(fname)[0] collection_fname = os.path.splitext(fname)[0]
params["annotations-data-collection-is-read-only"] = True params["annotations-data-collection-is-read-only"] = True
params["annotations-data-collection-name"] = collection_fname params["annotations-data-collection-name"] = collection_fname
+1
View File
@@ -44,6 +44,7 @@ def get_client_config(app_config, data_adaptor):
"annotations": False, "annotations": False,
"annotations_file": None, "annotations_file": None,
"annotations_dir": None, "annotations_dir": None,
"annotations_genesets_readonly": dataset_config.user_annotations__genesets__readonly,
"annotations_cell_ontology_enabled": False, "annotations_cell_ontology_enabled": False,
"annotations_cell_ontology_obopath": None, "annotations_cell_ontology_obopath": None,
"annotations_cell_ontology_terms": None, "annotations_cell_ontology_terms": None,
+35 -11
View File
@@ -42,6 +42,10 @@ class DatasetConfig(BaseConfig):
self.user_annotations__hosted_tiledb_array__hosted_file_directory = default_config["user_annotations"][ self.user_annotations__hosted_tiledb_array__hosted_file_directory = default_config["user_annotations"][
"hosted_tiledb_array" "hosted_tiledb_array"
]["hosted_file_directory"] ]["hosted_file_directory"]
self.user_annotations__genesets__readonly = default_config["user_annotations"]["genesets"]["readonly"]
self.user_annotations__local_file_csv__genesets_file = default_config["user_annotations"]["local_file_csv"][
"genesets_file"
]
self.embeddings__names = default_config["embeddings"]["names"] self.embeddings__names = default_config["embeddings"]["names"]
self.embeddings__enable_reembedding = default_config["embeddings"]["enable_reembedding"] self.embeddings__enable_reembedding = default_config["embeddings"]["enable_reembedding"]
@@ -98,6 +102,9 @@ class DatasetConfig(BaseConfig):
self.validate_correct_type_of_configuration_attribute( self.validate_correct_type_of_configuration_attribute(
"user_annotations__local_file_csv__file", (type(None), str) "user_annotations__local_file_csv__file", (type(None), str)
) )
self.validate_correct_type_of_configuration_attribute(
"user_annotations__local_file_csv__genesets_file", (type(None), str)
)
self.validate_correct_type_of_configuration_attribute("user_annotations__ontology__enable", bool) self.validate_correct_type_of_configuration_attribute("user_annotations__ontology__enable", bool)
self.validate_correct_type_of_configuration_attribute( self.validate_correct_type_of_configuration_attribute(
"user_annotations__ontology__obo_location", (type(None), str) "user_annotations__ontology__obo_location", (type(None), str)
@@ -108,6 +115,8 @@ class DatasetConfig(BaseConfig):
self.validate_correct_type_of_configuration_attribute( self.validate_correct_type_of_configuration_attribute(
"user_annotations__hosted_tiledb_array__hosted_file_directory", (type(None), str) "user_annotations__hosted_tiledb_array__hosted_file_directory", (type(None), str)
) )
self.validate_correct_type_of_configuration_attribute("user_annotations__genesets__readonly", bool)
if self.user_annotations__enable: if self.user_annotations__enable:
server_config = self.app_config.server_config server_config = self.app_config.server_config
if not self.app__authentication_enable: if not self.app__authentication_enable:
@@ -132,14 +141,22 @@ class DatasetConfig(BaseConfig):
def handle_local_file_csv_annotations(self): def handle_local_file_csv_annotations(self):
dirname = self.user_annotations__local_file_csv__directory dirname = self.user_annotations__local_file_csv__directory
filename = self.user_annotations__local_file_csv__file annotation_filename = self.user_annotations__local_file_csv__file
if filename is not None and dirname is not None: if annotation_filename is not None and dirname is not None:
raise ConfigurationError("'annotations-file' and 'annotations-dir' may not be used together.") raise ConfigurationError("'annotations-file' and 'annotations-dir' may not be used together.")
genesets_filename = self.user_annotations__local_file_csv__genesets_file
if genesets_filename is not None and dirname is not None:
raise ConfigurationError("'genesets-file' and 'annotations-dir' may not be used together.")
if filename is not None: if annotation_filename is not None:
lf_name, lf_ext = splitext(filename) lf_name, lf_ext = splitext(annotation_filename)
if lf_ext and lf_ext != ".csv": if lf_ext and lf_ext != ".csv":
raise ConfigurationError(f"annotation file type must be .csv: {filename}") raise ConfigurationError(f"annotation file type must be .csv: {annotation_filename}")
if genesets_filename is not None:
lf_name, lf_ext = splitext(genesets_filename)
if lf_ext and lf_ext != ".csv":
raise ConfigurationError(f"genesets file type must be .csv: {genesets_filename}")
if dirname is not None and not isdir(dirname): if dirname is not None and not isdir(dirname):
try: try:
@@ -147,16 +164,23 @@ class DatasetConfig(BaseConfig):
except OSError: except OSError:
raise ConfigurationError("Unable to create directory specified by --annotations-dir") raise ConfigurationError("Unable to create directory specified by --annotations-dir")
self.user_annotations = AnnotationsLocalFile(dirname, filename) self.user_annotations = AnnotationsLocalFile(dirname, annotation_filename, genesets_filename)
# if the user has specified a fixed label file, go ahead and validate it # if the user has specified a fixed label file, go ahead and validate it
# so that we can remove errors early in the process. # so that we can remove errors early in the process.
server_config = self.app_config.server_config server_config = self.app_config.server_config
if server_config.single_dataset__datapath and self.user_annotations__local_file_csv__file: if server_config.single_dataset__datapath:
with server_config.matrix_data_cache_manager.data_adaptor( if self.user_annotations__local_file_csv__file:
self.tag, server_config.single_dataset__datapath, self.app_config with server_config.matrix_data_cache_manager.data_adaptor(
) as data_adaptor: self.tag, server_config.single_dataset__datapath, self.app_config
data_adaptor.check_new_labels(self.user_annotations.read_labels(data_adaptor)) ) as data_adaptor:
data_adaptor.check_new_labels(self.user_annotations.read_labels(data_adaptor))
if self.user_annotations__local_file_csv__genesets_file:
with server_config.matrix_data_cache_manager.data_adaptor(
self.tag, server_config.single_dataset__datapath, self.app_config
) as data_adaptor:
data_adaptor.check_new_genesets(self.user_annotations.read_genesets(data_adaptor))
def handle_hosted_tiledb_annotations(self): def handle_hosted_tiledb_annotations(self):
self.validate_correct_type_of_configuration_attribute("user_annotations__hosted_tiledb_array__db_uri", str) self.validate_correct_type_of_configuration_attribute("user_annotations__hosted_tiledb_array__db_uri", str)
+40 -3
View File
@@ -196,9 +196,6 @@ def annotations_var_get(request, data_adaptor):
try: try:
labels = None labels = None
annotations = data_adaptor.dataset_config.user_annotations
if annotations is not None:
labels = annotations.read_labels(data_adaptor)
return make_response( return make_response(
data_adaptor.annotation_to_fbs_matrix(Axis.VAR, fields, labels), data_adaptor.annotation_to_fbs_matrix(Axis.VAR, fields, labels),
HTTPStatus.OK, HTTPStatus.OK,
@@ -328,3 +325,43 @@ def layout_obs_put(request, data_adaptor):
return abort_and_log(HTTPStatus.NOT_IMPLEMENTED, str(e)) return abort_and_log(HTTPStatus.NOT_IMPLEMENTED, str(e))
except (ValueError, DisabledFeatureError, FilterError) as e: except (ValueError, DisabledFeatureError, FilterError) as e:
return abort_and_log(HTTPStatus.BAD_REQUEST, str(e), include_exc_info=True) return abort_and_log(HTTPStatus.BAD_REQUEST, str(e), include_exc_info=True)
def genesets_get(request, data_adaptor):
preferred_mimetype = request.accept_mimetypes.best_match(["application/json"])
if preferred_mimetype != "application/json":
return abort(HTTPStatus.NOT_ACCEPTABLE)
annotations = data_adaptor.dataset_config.user_annotations
genesets = annotations.read_genesets(data_adaptor)
return make_response(jsonify({"genesets": genesets}), HTTPStatus.OK)
def genesets_put(request, data_adaptor):
annotations = data_adaptor.dataset_config.user_annotations
if annotations is None:
return abort(HTTPStatus.NOT_IMPLEMENTED)
if data_adaptor.dataset_config.user_annotations__genesets__readonly:
return abort(HTTPStatus.NOT_IMPLEMENTED)
anno_collection = request.args.get("annotation-collection-name", default=None)
if anno_collection is not None:
if not annotations.is_safe_collection_name(anno_collection):
return abort(HTTPStatus.BAD_REQUEST, "Bad annotation collection name")
annotations.set_collection(anno_collection)
args = request.get_json()
try:
genesets = args["genesets"]
if type(genesets) is list:
gs = data_adaptor.check_new_genesets(genesets)
annotations.write_genesets(gs, data_adaptor)
else:
pass
annotations.write_genesets([], data_adaptor)
res = json.dumps({"status": "OK"})
return make_response(res, HTTPStatus.OK, {"Content-Type": "application/json"})
except (ValueError, DisabledFeatureError, KeyError) as e:
return abort_and_log(HTTPStatus.BAD_REQUEST, str(e), include_exc_info=True)
+14 -1
View File
@@ -266,6 +266,19 @@ class DataAdaptor(metaclass=ABCMeta):
return labels_df return labels_df
def check_new_genesets(self, genesets):
"""Check gene sets, return if correct, else raise error"""
# server_config.single_dataset__var_names
print("<<<<<<<<>>>>>>>>>>> NEED TO IMPLEMENT check_new_genesets")
# XXX TODO -
# 1. check that all gs names are legal & non-duplicative.
# 2. check that all genes exist in var.index or var-index
# 3. de-dup genes
return genesets
def data_frame_to_fbs_matrix(self, filter, axis): def data_frame_to_fbs_matrix(self, filter, axis):
""" """
Retrieves data 'X' and returns in a flatbuffer Matrix. Retrieves data 'X' and returns in a flatbuffer Matrix.
@@ -338,7 +351,7 @@ class DataAdaptor(metaclass=ABCMeta):
@staticmethod @staticmethod
def normalize_embedding(embedding): def normalize_embedding(embedding):
"""Normalize embedding layout to meet client assumptions. """Normalize embedding layout to meet client assumptions.
Embedding is an ndarray, shape (n_obs, n)., where n is normally 2 Embedding is an ndarray, shape (n_obs, n)., where n is normally 2
""" """
# scale isotropically # scale isotropically
+4 -1
View File
@@ -191,10 +191,13 @@ dataset:
hosted_file_directory: null hosted_file_directory: null
local_file_csv: local_file_csv:
directory: null directory: null
file: null file: null # annotations file name
genesets_file: null # gene sets file name
ontology: ontology:
enable: false enable: false
obo_location: null obo_location: null
genesets:
readonly: true
embeddings: embeddings:
names : [] names : []