mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-03 09:28:11 +08:00
refactor config to support different config options for datasets in different dataroots. (#1596)
This will give us the ability to specify different config options for different dataroots. the key of the dataroot dictionary is no longer the same as the dataroot_url. Previously key==dataroot_url, and now those are separated. Added an "is_multi_dataset" function to simplify logic where it branched on single vs multi. Simplified the rest.py interface by no longer passing in the user annotations object, since that can be retrieved from the dataset.
This commit is contained in:
+511
-337
@@ -38,170 +38,110 @@ class AppFeature(object):
|
||||
|
||||
|
||||
class AppConfig(object):
|
||||
"""AppConfig stores all the configuration for cellxgene. The configuration is divided into two main parts:
|
||||
server attributes, and dataset attributes. The server_config contains attributes that refer to the server process
|
||||
as a whole. The default_dataset_config referes to attributes that are associated with the features and
|
||||
presentations of a dataset. The dataset config attributes can be overridden depending on the url by which the
|
||||
dataset was accessed. These are stored in dataroot_config.
|
||||
AppConfig has methods to initialize, modify, and access the configuration.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
|
||||
# the default configuration (see default_config.py)
|
||||
self.default_config = get_default_config()
|
||||
self.attr_checked = {k: False for k in self.__mapping(self.default_config).keys()}
|
||||
|
||||
dc = self.default_config
|
||||
try:
|
||||
self.server__verbose = dc["server"]["verbose"]
|
||||
self.server__debug = dc["server"]["debug"]
|
||||
self.server__host = dc["server"]["host"]
|
||||
self.server__port = dc["server"]["port"]
|
||||
self.server__scripts = dc["server"]["scripts"]
|
||||
self.server__inline_scripts = dc["server"]["inline_scripts"]
|
||||
self.server__open_browser = dc["server"]["open_browser"]
|
||||
self.server__about_legal_tos = dc["server"]["about_legal_tos"]
|
||||
self.server__about_legal_privacy = dc["server"]["about_legal_privacy"]
|
||||
self.server__force_https = dc["server"]["force_https"]
|
||||
self.server__flask_secret_key = dc["server"]["flask_secret_key"]
|
||||
self.server__generate_cache_control_headers = dc["server"]["generate_cache_control_headers"]
|
||||
self.server__server_timing_headers = dc["server"]["server_timing_headers"]
|
||||
self.server__csp_directives = dc["server"]["csp_directives"]
|
||||
|
||||
self.multi_dataset__dataroot = dc["multi_dataset"]["dataroot"]
|
||||
self.multi_dataset__index = dc["multi_dataset"]["index"]
|
||||
self.multi_dataset__allowed_matrix_types = dc["multi_dataset"]["allowed_matrix_types"]
|
||||
self.multi_dataset__matrix_cache__max_datasets = dc["multi_dataset"]["matrix_cache"]["max_datasets"]
|
||||
self.multi_dataset__matrix_cache__timelimit_s = dc["multi_dataset"]["matrix_cache"]["timelimit_s"]
|
||||
|
||||
self.single_dataset__datapath = dc["single_dataset"]["datapath"]
|
||||
self.single_dataset__obs_names = dc["single_dataset"]["obs_names"]
|
||||
self.single_dataset__var_names = dc["single_dataset"]["var_names"]
|
||||
self.single_dataset__about = dc["single_dataset"]["about"]
|
||||
self.single_dataset__title = dc["single_dataset"]["title"]
|
||||
|
||||
self.user_annotations__enable = dc["user_annotations"]["enable"]
|
||||
self.user_annotations__type = dc["user_annotations"]["type"]
|
||||
self.user_annotations__local_file_csv__directory = dc["user_annotations"]["local_file_csv"]["directory"]
|
||||
self.user_annotations__local_file_csv__file = dc["user_annotations"]["local_file_csv"]["file"]
|
||||
self.user_annotations__ontology__enable = dc["user_annotations"]["ontology"]["enable"]
|
||||
self.user_annotations__ontology__obo_location = dc["user_annotations"]["ontology"]["obo_location"]
|
||||
|
||||
self.presentation__max_categories = dc["presentation"]["max_categories"]
|
||||
self.presentation__custom_colors = dc["presentation"]["custom_colors"]
|
||||
|
||||
self.embeddings__names = dc["embeddings"]["names"]
|
||||
self.embeddings__enable_reembedding = dc["embeddings"]["enable_reembedding"]
|
||||
|
||||
self.diffexp__enable = dc["diffexp"]["enable"]
|
||||
self.diffexp__lfc_cutoff = dc["diffexp"]["lfc_cutoff"]
|
||||
self.diffexp__top_n = dc["diffexp"]["top_n"]
|
||||
self.diffexp__alg_cxg__max_workers = dc["diffexp"]["alg_cxg"]["max_workers"]
|
||||
self.diffexp__alg_cxg__cpu_multiplier = dc["diffexp"]["alg_cxg"]["cpu_multiplier"]
|
||||
self.diffexp__alg_cxg__target_workunit = dc["diffexp"]["alg_cxg"]["target_workunit"]
|
||||
|
||||
self.data_locator__s3__region_name = dc["data_locator"]["s3"]["region_name"]
|
||||
|
||||
self.adaptor__cxg_adaptor__tiledb_ctx = dc["adaptor"]["cxg_adaptor"]["tiledb_ctx"]
|
||||
self.adaptor__anndata_adaptor__backed = dc["adaptor"]["anndata_adaptor"]["backed"]
|
||||
|
||||
self.limits__diffexp_cellcount_max = dc["limits"]["diffexp_cellcount_max"]
|
||||
self.limits__column_request_max = dc["limits"]["column_request_max"]
|
||||
|
||||
except KeyError as e:
|
||||
raise ConfigurationError(f"Unexpected config: {str(e)}")
|
||||
|
||||
# The annotation object is created during complete_config and stored here.
|
||||
self.user_annotations = None
|
||||
|
||||
# The matrix data cache manager is created during the complete_config and stored here.
|
||||
self.matrix_data_cache_manager = None
|
||||
# the server configuration
|
||||
self.server_config = ServerConfig(self, self.default_config["server"])
|
||||
# the dataset config, unless overridden by an entry in dataroot_config
|
||||
self.default_dataset_config = DatasetConfig(None, self, self.default_config["dataset"])
|
||||
# a dictionary of keys to DatasetConfig objects. Each key must exist in the multi_dataset__dataroot
|
||||
# attribute of the server_config.
|
||||
self.dataroot_config = {}
|
||||
|
||||
# Set to true when config_completed is called
|
||||
self.is_completed = False
|
||||
|
||||
def get_dataset_config(self, dataroot_key):
|
||||
if self.server_config.single_dataset__datapath:
|
||||
return self.default_dataset_config
|
||||
else:
|
||||
return self.dataroot_config.get(dataroot_key, self.default_dataset_config)
|
||||
|
||||
def check_config(self):
|
||||
"""Verify all the attributes have been checked"""
|
||||
if not self.is_completed:
|
||||
raise ConfigurationError("The configuration has not been completed")
|
||||
mapping = self.__mapping(self.default_config)
|
||||
for key in mapping.keys():
|
||||
if not self.attr_checked[key]:
|
||||
raise ConfigurationError(f"The attr '{key}' has not been checked")
|
||||
self.server_config.check_config()
|
||||
self.default_dataset_config.check_config()
|
||||
for dataset_config in self.dataroot_config.values():
|
||||
dataset_config.check_config()
|
||||
|
||||
def __mapping(self, config):
|
||||
"""Create a mapping from attribute names to (location in the config tree, value)"""
|
||||
dc = copy.deepcopy(config)
|
||||
mapping = {}
|
||||
def update_server_config(self, **kw):
|
||||
self.server_config.update(**kw)
|
||||
self.is_complete = False
|
||||
|
||||
# special cases where the value could be a dict.
|
||||
# If its value is not None, the entry is added to the mapping, and not included
|
||||
# in the flattening below.
|
||||
dictval_cases = [
|
||||
("adaptor", "cxg_adaptor", "tiledb_ctx"),
|
||||
("server", "csp_directives"),
|
||||
("multi_dataset", "dataroot"),
|
||||
]
|
||||
for dictval_case in dictval_cases:
|
||||
cur = dc
|
||||
for part in dictval_case[:-1]:
|
||||
cur = cur.get(part, {})
|
||||
val = cur.get(dictval_case[-1])
|
||||
if val is not None:
|
||||
key = "__".join(dictval_case)
|
||||
mapping[key] = (dictval_case, val)
|
||||
del cur[dictval_case[-1]]
|
||||
|
||||
flat_config = flatten(dc)
|
||||
for key, value in flat_config.items():
|
||||
# name of the attribute
|
||||
attr = "__".join(key)
|
||||
mapping[attr] = (key, value)
|
||||
|
||||
return mapping
|
||||
def update_default_dataset_config(self, **kw):
|
||||
self.default_dataset_config.update(**kw)
|
||||
# update all the other dataset configs, if any
|
||||
for value in self.dataroot_config.values():
|
||||
value.update(**kw)
|
||||
self.is_complete = False
|
||||
|
||||
def update_from_config_file(self, config_file):
|
||||
with open(config_file) as fyaml:
|
||||
config = yaml.load(fyaml, Loader=yaml.FullLoader)
|
||||
|
||||
mapping = self.__mapping(config)
|
||||
for attr, (key, value) in mapping.items():
|
||||
if not hasattr(self, attr):
|
||||
raise ConfigurationError(f"Unknown key from config file: {key}")
|
||||
try:
|
||||
setattr(self, attr, value)
|
||||
except KeyError:
|
||||
raise ConfigurationError(f"Unable to set config attribute: {key}")
|
||||
self.server_config.update_from_config(config["server"], "server")
|
||||
self.default_dataset_config.update_from_config(config["dataset"], "dataset")
|
||||
|
||||
self.attr_checked[attr] = False
|
||||
per_dataset_config = config.get("per_dataset_config", {})
|
||||
for key, dataroot_config in per_dataset_config.items():
|
||||
self.add_dataroot_config(key, **dataroot_config)
|
||||
|
||||
self.is_completed = False
|
||||
self.is_complete = False
|
||||
|
||||
def write_config(self, config_file):
|
||||
"""output the config to a yaml file"""
|
||||
mapping = self.__mapping(self.default_config)
|
||||
for attrname in mapping.keys():
|
||||
mapping[attrname] = getattr(self, attrname)
|
||||
config = unflatten(mapping, splitter=lambda key: key.split("__"))
|
||||
server = self.server_config.create_mapping(self.server_config.default_config)
|
||||
dataset = self.default_dataset_config.create_mapping(self.default_dataset_config.default_config)
|
||||
config = dict(server={}, dataset={})
|
||||
for attrname in server.keys():
|
||||
config["server__" + attrname] = getattr(self.server_config, attrname)
|
||||
for attrname in dataset.keys():
|
||||
config["dataset__" + attrname] = getattr(self.default_dataset_config, attrname)
|
||||
if self.dataroot_config:
|
||||
config["per_dataset_config"] = {}
|
||||
for dataroot_tag, dataroot_config in self.dataroot_config.items():
|
||||
dataset = dataroot_config.create_mapping(dataroot_config.default_config)
|
||||
for attrname in dataset.keys():
|
||||
config[f"per_dataset_config__{dataroot_tag}__" + attrname] = getattr(dataroot_config, attrname)
|
||||
|
||||
config = unflatten(config, splitter=lambda key: key.split("__"))
|
||||
yaml.dump(config, open(config_file, "w"))
|
||||
|
||||
def update(self, **kw):
|
||||
for key, value in kw.items():
|
||||
if not hasattr(self, key):
|
||||
raise ConfigurationError(f"unknown config parameter {key}.")
|
||||
try:
|
||||
if type(value) == tuple:
|
||||
# convert tuple values to list values
|
||||
value = list(value)
|
||||
setattr(self, key, value)
|
||||
except KeyError:
|
||||
raise ConfigurationError(f"Unable to set config parameter {key}.")
|
||||
|
||||
self.attr_checked[key] = False
|
||||
|
||||
self.is_completed = False
|
||||
|
||||
def changes_from_default(self):
|
||||
"""Return all the attribute that are different from the default"""
|
||||
mapping = self.__mapping(self.default_config)
|
||||
diff = []
|
||||
for attrname, (key, defval) in mapping.items():
|
||||
curval = getattr(self, attrname)
|
||||
if curval != defval:
|
||||
diff.append((attrname, curval, defval))
|
||||
diff_server = self.server_config.changes_from_default()
|
||||
diff_dataset = self.default_dataset_config.changes_from_default()
|
||||
diff = dict(server=diff_server, dataset=diff_dataset)
|
||||
return diff
|
||||
|
||||
def add_dataroot_config(self, dataroot_tag, **kw):
|
||||
"""Create a new dataset config object based on the default dataset config, and kw parameters"""
|
||||
if dataroot_tag in self.dataroot_config:
|
||||
raise ConfigurationError(f"dataroot config already exists: {dataroot_tag}")
|
||||
if type(self.server_config.multi_dataset__dataroot) != dict:
|
||||
raise ConfigurationError("The server__multi_dataset__dataroot must be a dictionary")
|
||||
if dataroot_tag not in self.server_config.multi_dataset__dataroot:
|
||||
raise ConfigurationError(f"The dataroot_tag ({dataroot_tag}) not found in server__multi_dataset__dataroot")
|
||||
|
||||
self.is_completed = False
|
||||
self.dataroot_config[dataroot_tag] = DatasetConfig(dataroot_tag, self, self.default_config["dataset"])
|
||||
flat_config = self.default_dataset_config.create_mapping(self.default_dataset_config.default_config)
|
||||
config = {key: value[1] for key, value in flat_config.items()}
|
||||
self.dataroot_config[dataroot_tag].update(**config)
|
||||
self.dataroot_config[dataroot_tag].update_from_config(kw, dataroot_tag)
|
||||
|
||||
def complete_config(self, messagefn=None):
|
||||
"""The configure options are checked, and any additional setup based on the config
|
||||
parameters is done"""
|
||||
@@ -218,22 +158,149 @@ class AppConfig(object):
|
||||
# messages we can give correct context for attributes with bad value.
|
||||
context = dict(messagefn=messagefn)
|
||||
|
||||
self.handle_server(context)
|
||||
self.handle_adaptor(context)
|
||||
self.handle_data_locator(context)
|
||||
self.handle_adaptor(context) # may depend on data_locator
|
||||
self.handle_presentation(context)
|
||||
self.handle_single_dataset(context) # may depend on adaptor
|
||||
self.handle_multi_dataset(context) # may depend on adaptor
|
||||
self.handle_user_annotations(context)
|
||||
self.handle_embeddings(context)
|
||||
self.handle_diffexp(context)
|
||||
self.handle_limits(context)
|
||||
self.server_config.complete_config(context)
|
||||
self.default_dataset_config.complete_config(context)
|
||||
for dataroot_config in self.dataroot_config.values():
|
||||
dataroot_config.complete_config(context)
|
||||
|
||||
self.is_completed = True
|
||||
self.check_config()
|
||||
|
||||
def __check_attr(self, attrname, vtype):
|
||||
def get_matrix_data_cache_manager(self):
|
||||
return self.server_config.matrix_data_cache_manager
|
||||
|
||||
def is_multi_dataset(self):
|
||||
return self.server_config.multi_dataset__dataroot is not None
|
||||
|
||||
def get_title(self, data_adaptor):
|
||||
return (
|
||||
self.server_config.single_dataset__title
|
||||
if self.server_config.single_dataset__title
|
||||
else data_adaptor.get_title()
|
||||
)
|
||||
|
||||
def get_about(self, data_adaptor):
|
||||
return (
|
||||
self.server_config.single_dataset__about
|
||||
if self.server_config.single_dataset__about
|
||||
else data_adaptor.get_about()
|
||||
)
|
||||
|
||||
def get_client_config(self, data_adaptor):
|
||||
"""
|
||||
Return the configuration as required by the /config REST route
|
||||
"""
|
||||
|
||||
server_config = self.server_config
|
||||
dataset_config = data_adaptor.dataset_config
|
||||
annotation = dataset_config.user_annotations
|
||||
|
||||
# FIXME The current set of config is not consistently presented:
|
||||
# we have camalCase, hyphen-text, and underscore_text
|
||||
|
||||
# make sure the configuration has been checked.
|
||||
self.check_config()
|
||||
|
||||
# features
|
||||
features = [f.todict() for f in data_adaptor.get_features(annotation)]
|
||||
|
||||
# display_names
|
||||
title = self.get_title(data_adaptor)
|
||||
about = self.get_about(data_adaptor)
|
||||
|
||||
display_names = dict(engine=data_adaptor.get_name(), dataset=title)
|
||||
|
||||
# library_versions
|
||||
library_versions = {}
|
||||
library_versions.update(data_adaptor.get_library_versions())
|
||||
library_versions["cellxgene"] = cellxgene_version
|
||||
|
||||
# links
|
||||
links = {"about-dataset": about}
|
||||
|
||||
# parameters
|
||||
parameters = {
|
||||
"layout": dataset_config.embeddings__names,
|
||||
"max-category-items": dataset_config.presentation__max_categories,
|
||||
"obs_names": server_config.single_dataset__obs_names,
|
||||
"var_names": server_config.single_dataset__var_names,
|
||||
"diffexp_lfc_cutoff": dataset_config.diffexp__lfc_cutoff,
|
||||
"backed": server_config.adaptor__anndata_adaptor__backed,
|
||||
"disable-diffexp": not dataset_config.diffexp__enable,
|
||||
"enable-reembedding": dataset_config.embeddings__enable_reembedding,
|
||||
"annotations": False,
|
||||
"annotations_file": None,
|
||||
"annotations_dir": None,
|
||||
"annotations_cell_ontology_enabled": False,
|
||||
"annotations_cell_ontology_obopath": None,
|
||||
"annotations_cell_ontology_terms": None,
|
||||
"custom_colors": dataset_config.presentation__custom_colors,
|
||||
"diffexp-may-be-slow": False,
|
||||
"about_legal_tos": dataset_config.app__about_legal_tos,
|
||||
"about_legal_privacy": dataset_config.app__about_legal_privacy,
|
||||
}
|
||||
|
||||
data_adaptor.update_parameters(parameters)
|
||||
if annotation:
|
||||
annotation.update_parameters(parameters, data_adaptor)
|
||||
|
||||
# gather it all together
|
||||
c = {}
|
||||
config = c["config"] = {}
|
||||
config["features"] = features
|
||||
config["displayNames"] = display_names
|
||||
config["library_versions"] = library_versions
|
||||
config["links"] = links
|
||||
config["parameters"] = parameters
|
||||
config["limits"] = {
|
||||
"column_request_max": server_config.limits__column_request_max,
|
||||
"diffexp_cellcount_max": server_config.limits__diffexp_cellcount_max,
|
||||
}
|
||||
|
||||
return c
|
||||
|
||||
|
||||
class BaseConfig(object):
|
||||
"""This class handles the mechanics of updating and checking attributes.
|
||||
Derived classes are expected to store the actual attributes"""
|
||||
|
||||
def __init__(self, app_config, default_config, dictval_cases={}):
|
||||
# reference back to the app_config
|
||||
self.app_config = app_config
|
||||
# the complete set of attribute and their default values (unflattened)
|
||||
self.default_config = default_config
|
||||
# attributes where the value may be a dict (and therefore are not flattened)
|
||||
self.dictval_cases = dictval_cases
|
||||
# used to make sure every attribute value is checked
|
||||
self.attr_checked = {k: False for k in self.create_mapping(default_config).keys()}
|
||||
|
||||
def create_mapping(self, config):
|
||||
"""Create a mapping from attribute names to (location in the config tree, value)"""
|
||||
dc = copy.deepcopy(config)
|
||||
mapping = {}
|
||||
|
||||
# special cases where the value could be a dict.
|
||||
# If its value is not None, the entry is added to the mapping, and not included
|
||||
# in the flattening below.
|
||||
for dictval_case in self.dictval_cases:
|
||||
cur = dc
|
||||
for part in dictval_case[:-1]:
|
||||
cur = cur.get(part, {})
|
||||
val = cur.get(dictval_case[-1])
|
||||
if val is not None:
|
||||
key = "__".join(dictval_case)
|
||||
mapping[key] = (dictval_case, val)
|
||||
del cur[dictval_case[-1]]
|
||||
|
||||
flat_config = flatten(dc)
|
||||
for key, value in flat_config.items():
|
||||
# name of the attribute
|
||||
attr = "__".join(key)
|
||||
mapping[attr] = (key, value)
|
||||
|
||||
return mapping
|
||||
|
||||
def check_attr(self, attrname, vtype):
|
||||
val = getattr(self, attrname)
|
||||
if type(vtype) in (list, tuple):
|
||||
if type(val) not in vtype:
|
||||
@@ -250,48 +317,152 @@ class AppConfig(object):
|
||||
|
||||
self.attr_checked[attrname] = True
|
||||
|
||||
def handle_server(self, context):
|
||||
self.__check_attr("server__verbose", bool)
|
||||
self.__check_attr("server__debug", bool)
|
||||
self.__check_attr("server__host", str)
|
||||
self.__check_attr("server__port", (type(None), int))
|
||||
self.__check_attr("server__scripts", list)
|
||||
self.__check_attr("server__inline_scripts", list)
|
||||
self.__check_attr("server__open_browser", bool)
|
||||
self.__check_attr("server__force_https", bool)
|
||||
self.__check_attr("server__flask_secret_key", (type(None), str))
|
||||
self.__check_attr("server__generate_cache_control_headers", bool)
|
||||
self.__check_attr("server__about_legal_tos", (type(None), str))
|
||||
self.__check_attr("server__about_legal_privacy", (type(None), str))
|
||||
self.__check_attr("server__server_timing_headers", bool)
|
||||
self.__check_attr("server__csp_directives", (type(None), dict))
|
||||
def check_config(self):
|
||||
mapping = self.create_mapping(self.default_config)
|
||||
for key in mapping.keys():
|
||||
if not self.attr_checked[key]:
|
||||
raise ConfigurationError(f"The attr '{key}' has not been checked")
|
||||
|
||||
if self.server__port:
|
||||
if not is_port_available(self.server__host, self.server__port):
|
||||
def update(self, **kw):
|
||||
for key, value in kw.items():
|
||||
if not hasattr(self, key):
|
||||
raise ConfigurationError(f"unknown config parameter {key}.")
|
||||
try:
|
||||
if type(value) == tuple:
|
||||
# convert tuple values to list values
|
||||
value = list(value)
|
||||
setattr(self, key, value)
|
||||
except KeyError:
|
||||
raise ConfigurationError(f"Unable to set config parameter {key}.")
|
||||
|
||||
self.attr_checked[key] = False
|
||||
|
||||
def update_from_config(self, config, prefix):
|
||||
mapping = self.create_mapping(config)
|
||||
for attr, (key, value) in mapping.items():
|
||||
if not hasattr(self, attr):
|
||||
raise ConfigurationError(f"Unknown key from config file: {prefix}__{attr}")
|
||||
try:
|
||||
setattr(self, attr, value)
|
||||
except KeyError:
|
||||
raise ConfigurationError(f"Unable to set config attribute: {prefix}__{attr}")
|
||||
|
||||
self.attr_checked[attr] = False
|
||||
|
||||
def changes_from_default(self):
|
||||
"""Return all the attribute that are different from the default"""
|
||||
mapping = self.create_mapping(self.default_config)
|
||||
diff = []
|
||||
for attrname, (key, defval) in mapping.items():
|
||||
curval = getattr(self, attrname)
|
||||
if curval != defval:
|
||||
diff.append((attrname, curval, defval))
|
||||
return diff
|
||||
|
||||
|
||||
class ServerConfig(BaseConfig):
|
||||
"""Manages the config attribute associated with the server."""
|
||||
|
||||
def __init__(self, app_config, default_config):
|
||||
dictval_cases = [
|
||||
("app", "csp_directives"),
|
||||
("adaptor", "cxg_adaptor", "tiledb_ctx"),
|
||||
("multi_dataset", "dataroot"),
|
||||
]
|
||||
super().__init__(app_config, default_config, dictval_cases)
|
||||
|
||||
dc = default_config
|
||||
try:
|
||||
self.app__verbose = dc["app"]["verbose"]
|
||||
self.app__debug = dc["app"]["debug"]
|
||||
self.app__host = dc["app"]["host"]
|
||||
self.app__port = dc["app"]["port"]
|
||||
self.app__open_browser = dc["app"]["open_browser"]
|
||||
self.app__force_https = dc["app"]["force_https"]
|
||||
self.app__flask_secret_key = dc["app"]["flask_secret_key"]
|
||||
self.app__generate_cache_control_headers = dc["app"]["generate_cache_control_headers"]
|
||||
self.app__server_timing_headers = dc["app"]["server_timing_headers"]
|
||||
self.app__csp_directives = dc["app"]["csp_directives"]
|
||||
|
||||
self.multi_dataset__dataroot = dc["multi_dataset"]["dataroot"]
|
||||
self.multi_dataset__index = dc["multi_dataset"]["index"]
|
||||
self.multi_dataset__allowed_matrix_types = dc["multi_dataset"]["allowed_matrix_types"]
|
||||
self.multi_dataset__matrix_cache__max_datasets = dc["multi_dataset"]["matrix_cache"]["max_datasets"]
|
||||
self.multi_dataset__matrix_cache__timelimit_s = dc["multi_dataset"]["matrix_cache"]["timelimit_s"]
|
||||
|
||||
self.single_dataset__datapath = dc["single_dataset"]["datapath"]
|
||||
self.single_dataset__obs_names = dc["single_dataset"]["obs_names"]
|
||||
self.single_dataset__var_names = dc["single_dataset"]["var_names"]
|
||||
self.single_dataset__about = dc["single_dataset"]["about"]
|
||||
self.single_dataset__title = dc["single_dataset"]["title"]
|
||||
|
||||
self.diffexp__alg_cxg__max_workers = dc["diffexp"]["alg_cxg"]["max_workers"]
|
||||
self.diffexp__alg_cxg__cpu_multiplier = dc["diffexp"]["alg_cxg"]["cpu_multiplier"]
|
||||
self.diffexp__alg_cxg__target_workunit = dc["diffexp"]["alg_cxg"]["target_workunit"]
|
||||
|
||||
self.data_locator__s3__region_name = dc["data_locator"]["s3"]["region_name"]
|
||||
|
||||
self.adaptor__cxg_adaptor__tiledb_ctx = dc["adaptor"]["cxg_adaptor"]["tiledb_ctx"]
|
||||
self.adaptor__anndata_adaptor__backed = dc["adaptor"]["anndata_adaptor"]["backed"]
|
||||
|
||||
self.limits__diffexp_cellcount_max = dc["limits"]["diffexp_cellcount_max"]
|
||||
self.limits__column_request_max = dc["limits"]["column_request_max"]
|
||||
|
||||
except KeyError as e:
|
||||
raise ConfigurationError(f"Unexpected config: {str(e)}")
|
||||
|
||||
# The matrix data cache manager is created during the complete_config and stored here.
|
||||
self.matrix_data_cache_manager = None
|
||||
|
||||
def complete_config(self, context):
|
||||
self.handle_app(context)
|
||||
self.handle_data_locator(context)
|
||||
self.handle_adaptor(context) # may depend on data_locator
|
||||
self.handle_single_dataset(context) # may depend on adaptor
|
||||
self.handle_multi_dataset(context) # may depend on adaptor
|
||||
self.handle_diffexp(context)
|
||||
self.handle_limits(context)
|
||||
|
||||
self.check_config()
|
||||
|
||||
def handle_app(self, context):
|
||||
self.check_attr("app__verbose", bool)
|
||||
self.check_attr("app__debug", bool)
|
||||
self.check_attr("app__host", str)
|
||||
self.check_attr("app__port", (type(None), int))
|
||||
self.check_attr("app__open_browser", bool)
|
||||
self.check_attr("app__force_https", bool)
|
||||
self.check_attr("app__flask_secret_key", (type(None), str))
|
||||
self.check_attr("app__generate_cache_control_headers", bool)
|
||||
self.check_attr("app__server_timing_headers", bool)
|
||||
self.check_attr("app__csp_directives", (type(None), dict))
|
||||
|
||||
if self.app__port:
|
||||
if not is_port_available(self.app__host, self.app__port):
|
||||
raise ConfigurationError(
|
||||
f"The port selected {self.server__port} is in use, please configure an open port."
|
||||
f"The port selected {self.app__port} is in use, please configure an open port."
|
||||
)
|
||||
else:
|
||||
self.server__port = find_available_port(self.server__host, DEFAULT_SERVER_PORT)
|
||||
self.app__port = find_available_port(self.app__host, DEFAULT_SERVER_PORT)
|
||||
|
||||
if self.server__debug:
|
||||
if self.app__debug:
|
||||
context["messagefn"]("in debug mode, setting verbose=True and open_browser=False")
|
||||
self.server__verbose = True
|
||||
self.server__open_browser = False
|
||||
self.app__verbose = True
|
||||
self.app__open_browser = False
|
||||
else:
|
||||
warnings.formatwarning = custom_format_warning
|
||||
|
||||
if not self.server__verbose:
|
||||
if not self.app__verbose:
|
||||
sys.tracebacklimit = 0
|
||||
|
||||
# secret key:
|
||||
# first, from CXG_SECRET_KEY environment variable
|
||||
# second, from config file
|
||||
self.server__flask_secret_key = os.environ.get("CXG_SECRET_KEY", self.server__flask_secret_key)
|
||||
self.app__flask_secret_key = os.environ.get("CXG_SECRET_KEY", self.app__flask_secret_key)
|
||||
|
||||
# CSP Directives are a dict of string: list(string) or string: string
|
||||
if self.server__csp_directives is not None:
|
||||
for k, v in self.server__csp_directives.items():
|
||||
if self.app__csp_directives is not None:
|
||||
for k, v in self.app__csp_directives.items():
|
||||
if not isinstance(k, str):
|
||||
raise ConfigurationError("CSP directive names must be a string.")
|
||||
if isinstance(v, list):
|
||||
@@ -301,26 +472,15 @@ class AppConfig(object):
|
||||
elif not isinstance(v, str):
|
||||
raise ConfigurationError("CSP directive value must be a string or list of strings.")
|
||||
|
||||
# scripts can be string (filename) or dict (attributes). Convert string to dict.
|
||||
scripts = []
|
||||
for s in self.server__scripts:
|
||||
if isinstance(s, str):
|
||||
scripts.append({"src": s})
|
||||
elif isinstance(s, dict) and isinstance(s["src"], str):
|
||||
scripts.append(s)
|
||||
else:
|
||||
raise ConfigurationError("Scripts must be string or dict")
|
||||
self.server__scripts = scripts
|
||||
|
||||
def handle_data_locator(self, context):
|
||||
self.__check_attr("data_locator__s3__region_name", (type(None), bool, str))
|
||||
self.check_attr("data_locator__s3__region_name", (type(None), bool, str))
|
||||
if self.data_locator__s3__region_name is True:
|
||||
path = self.single_dataset__datapath or self.multi_dataset__dataroot
|
||||
if type(path) == dict:
|
||||
# if multi_dataset__dataroot is a dict, then use the first key
|
||||
# that is in s3. NOTE: it is not supported to have dataroots
|
||||
# in different regions.
|
||||
paths = path.values()
|
||||
paths = [val.get("dataroot") for val in path.values()]
|
||||
for path in paths:
|
||||
if path.startswith("s3://"):
|
||||
break
|
||||
@@ -332,16 +492,12 @@ class AppConfig(object):
|
||||
region_name = None
|
||||
self.data_locator__s3__region_name = region_name
|
||||
|
||||
def handle_presentation(self, context):
|
||||
self.__check_attr("presentation__max_categories", int)
|
||||
self.__check_attr("presentation__custom_colors", bool)
|
||||
|
||||
def handle_single_dataset(self, context):
|
||||
self.__check_attr("single_dataset__datapath", (str, type(None)))
|
||||
self.__check_attr("single_dataset__title", (str, type(None)))
|
||||
self.__check_attr("single_dataset__about", (str, type(None)))
|
||||
self.__check_attr("single_dataset__obs_names", (str, type(None)))
|
||||
self.__check_attr("single_dataset__var_names", (str, type(None)))
|
||||
self.check_attr("single_dataset__datapath", (str, type(None)))
|
||||
self.check_attr("single_dataset__title", (str, type(None)))
|
||||
self.check_attr("single_dataset__about", (str, type(None)))
|
||||
self.check_attr("single_dataset__obs_names", (str, type(None)))
|
||||
self.check_attr("single_dataset__var_names", (str, type(None)))
|
||||
|
||||
if self.single_dataset__datapath is None:
|
||||
if self.multi_dataset__dataroot is None:
|
||||
@@ -357,7 +513,7 @@ class AppConfig(object):
|
||||
self.matrix_data_cache_manager = MatrixDataCacheManager(max_cached=1, timelimit_s=None)
|
||||
|
||||
# preload this data set
|
||||
matrix_data_loader = MatrixDataLoader(self.single_dataset__datapath, app_config=self)
|
||||
matrix_data_loader = MatrixDataLoader(self.single_dataset__datapath, app_config=self.app_config)
|
||||
try:
|
||||
matrix_data_loader.pre_load_validation()
|
||||
except DatasetAccessError as e:
|
||||
@@ -388,26 +544,46 @@ class AppConfig(object):
|
||||
)
|
||||
|
||||
def handle_multi_dataset(self, context):
|
||||
self.__check_attr("multi_dataset__dataroot", (type(None), dict, str))
|
||||
self.__check_attr("multi_dataset__index", (type(None), bool, str))
|
||||
self.__check_attr("multi_dataset__allowed_matrix_types", list)
|
||||
self.__check_attr("multi_dataset__matrix_cache__max_datasets", int)
|
||||
self.__check_attr("multi_dataset__matrix_cache__timelimit_s", (type(None), int, float))
|
||||
self.check_attr("multi_dataset__dataroot", (type(None), dict, str))
|
||||
self.check_attr("multi_dataset__index", (type(None), bool, str))
|
||||
self.check_attr("multi_dataset__allowed_matrix_types", list)
|
||||
self.check_attr("multi_dataset__matrix_cache__max_datasets", int)
|
||||
self.check_attr("multi_dataset__matrix_cache__timelimit_s", (type(None), int, float))
|
||||
|
||||
if self.multi_dataset__dataroot is None:
|
||||
return
|
||||
|
||||
if type(self.multi_dataset__dataroot) == str:
|
||||
self.multi_dataset__dataroot = dict(d=self.multi_dataset__dataroot)
|
||||
default_dict = dict(base_url="d", dataroot=self.multi_dataset__dataroot)
|
||||
self.multi_dataset__dataroot = dict(d=default_dict)
|
||||
|
||||
for key in self.multi_dataset__dataroot.keys():
|
||||
# sanity check for well formed keys
|
||||
if type(key) != str:
|
||||
raise ConfigurationError(f"error in multi_dataset__dataroot {key}")
|
||||
if quote_plus(key) != key:
|
||||
raise ConfigurationError(f"error in multi_dataset__dataroot {key}")
|
||||
if os.path.split(os.path.normpath(key))[-1] != key:
|
||||
raise ConfigurationError(f"error in multi_dataset__dataroot {key}")
|
||||
for tag, dataroot_dict in self.multi_dataset__dataroot.items():
|
||||
if "base_url" not in dataroot_dict:
|
||||
raise ConfigurationError(f"error in multi_dataset__dataroot: missing base_url for tag {tag}")
|
||||
if "dataroot" not in dataroot_dict:
|
||||
raise ConfigurationError(f"error in multi_dataset__dataroot: missing dataroot, for tag {tag}")
|
||||
|
||||
base_url = dataroot_dict["base_url"]
|
||||
|
||||
# sanity check for well formed base urls
|
||||
bad = False
|
||||
if type(base_url) != str:
|
||||
bad = True
|
||||
elif os.path.normpath(base_url) != base_url:
|
||||
bad = True
|
||||
else:
|
||||
base_url_parts = base_url.split("/")
|
||||
if [quote_plus(part) for part in base_url_parts] != base_url_parts:
|
||||
bad = True
|
||||
if ".." in base_url_parts:
|
||||
bad = True
|
||||
if bad:
|
||||
raise ConfigurationError(f"error in multi_dataset__dataroot base_url {base_url} for tag {tag}")
|
||||
|
||||
# verify all the base_urls are unique
|
||||
base_urls = [d["base_url"] for d in self.multi_dataset__dataroot.values()]
|
||||
if len(base_urls) > len(set(base_urls)):
|
||||
raise ConfigurationError("error in multi_dataset__dataroot: base_urls must be unique")
|
||||
|
||||
# error checking
|
||||
for mtype in self.multi_dataset__allowed_matrix_types:
|
||||
@@ -423,13 +599,114 @@ class AppConfig(object):
|
||||
timelimit_s=self.multi_dataset__matrix_cache__timelimit_s,
|
||||
)
|
||||
|
||||
def handle_diffexp(self, context):
|
||||
self.check_attr("diffexp__alg_cxg__max_workers", (str, int))
|
||||
self.check_attr("diffexp__alg_cxg__cpu_multiplier", int)
|
||||
self.check_attr("diffexp__alg_cxg__target_workunit", int)
|
||||
|
||||
max_workers = self.diffexp__alg_cxg__max_workers
|
||||
cpu_multiplier = self.diffexp__alg_cxg__cpu_multiplier
|
||||
cpu_count = os.cpu_count()
|
||||
max_workers = min(max_workers, cpu_multiplier * cpu_count)
|
||||
diffexp_tiledb.set_config(max_workers, self.diffexp__alg_cxg__target_workunit)
|
||||
|
||||
def handle_adaptor(self, context):
|
||||
# cxg
|
||||
self.check_attr("adaptor__cxg_adaptor__tiledb_ctx", dict)
|
||||
regionkey = "vfs.s3.region"
|
||||
if regionkey not in self.adaptor__cxg_adaptor__tiledb_ctx:
|
||||
if type(self.data_locator__s3__region_name) == str:
|
||||
self.adaptor__cxg_adaptor__tiledb_ctx[regionkey] = self.data_locator__s3__region_name
|
||||
|
||||
from server.data_cxg.cxg_adaptor import CxgAdaptor
|
||||
|
||||
CxgAdaptor.set_tiledb_context(self.adaptor__cxg_adaptor__tiledb_ctx)
|
||||
|
||||
# anndata
|
||||
self.check_attr("adaptor__anndata_adaptor__backed", bool)
|
||||
|
||||
def handle_limits(self, context):
|
||||
self.check_attr("limits__diffexp_cellcount_max", (type(None), int))
|
||||
self.check_attr("limits__column_request_max", (type(None), int))
|
||||
|
||||
def exceeds_limit(self, limit_name, value):
|
||||
limit_value = getattr(self, "limits__" + limit_name, None)
|
||||
if limit_value is None: # disabled
|
||||
return False
|
||||
return value > limit_value
|
||||
|
||||
|
||||
class DatasetConfig(BaseConfig):
|
||||
"""Manages the config attribute associated with a dataset."""
|
||||
|
||||
def __init__(self, tag, app_config, default_config):
|
||||
super().__init__(app_config, default_config)
|
||||
self.tag = tag
|
||||
dc = default_config
|
||||
try:
|
||||
self.app__scripts = dc["app"]["scripts"]
|
||||
self.app__inline_scripts = dc["app"]["inline_scripts"]
|
||||
self.app__about_legal_tos = dc["app"]["about_legal_tos"]
|
||||
self.app__about_legal_privacy = dc["app"]["about_legal_privacy"]
|
||||
|
||||
self.presentation__max_categories = dc["presentation"]["max_categories"]
|
||||
self.presentation__custom_colors = dc["presentation"]["custom_colors"]
|
||||
|
||||
self.user_annotations__enable = dc["user_annotations"]["enable"]
|
||||
self.user_annotations__type = dc["user_annotations"]["type"]
|
||||
self.user_annotations__local_file_csv__directory = dc["user_annotations"]["local_file_csv"]["directory"]
|
||||
self.user_annotations__local_file_csv__file = dc["user_annotations"]["local_file_csv"]["file"]
|
||||
self.user_annotations__ontology__enable = dc["user_annotations"]["ontology"]["enable"]
|
||||
self.user_annotations__ontology__obo_location = dc["user_annotations"]["ontology"]["obo_location"]
|
||||
|
||||
self.embeddings__names = dc["embeddings"]["names"]
|
||||
self.embeddings__enable_reembedding = dc["embeddings"]["enable_reembedding"]
|
||||
|
||||
self.diffexp__enable = dc["diffexp"]["enable"]
|
||||
self.diffexp__lfc_cutoff = dc["diffexp"]["lfc_cutoff"]
|
||||
self.diffexp__top_n = dc["diffexp"]["top_n"]
|
||||
|
||||
except KeyError as e:
|
||||
raise ConfigurationError(f"Unexpected config: {str(e)}")
|
||||
|
||||
# The annotation object is created during complete_config and stored here.
|
||||
self.user_annotations = None
|
||||
|
||||
def complete_config(self, context):
|
||||
self.handle_app(context)
|
||||
self.handle_presentation(context)
|
||||
self.handle_user_annotations(context)
|
||||
self.handle_embeddings(context)
|
||||
self.handle_diffexp(context)
|
||||
|
||||
def handle_app(self, context):
|
||||
self.check_attr("app__scripts", list)
|
||||
self.check_attr("app__inline_scripts", list)
|
||||
self.check_attr("app__about_legal_tos", (type(None), str))
|
||||
self.check_attr("app__about_legal_privacy", (type(None), str))
|
||||
|
||||
# scripts can be string (filename) or dict (attributes). Convert string to dict.
|
||||
scripts = []
|
||||
for s in self.app__scripts:
|
||||
if isinstance(s, str):
|
||||
scripts.append({"src": s})
|
||||
elif isinstance(s, dict) and isinstance(s["src"], str):
|
||||
scripts.append(s)
|
||||
else:
|
||||
raise ConfigurationError("Scripts must be string or dict")
|
||||
self.app__scripts = scripts
|
||||
|
||||
def handle_presentation(self, context):
|
||||
self.check_attr("presentation__max_categories", int)
|
||||
self.check_attr("presentation__custom_colors", bool)
|
||||
|
||||
def handle_user_annotations(self, context):
|
||||
self.__check_attr("user_annotations__enable", bool)
|
||||
self.__check_attr("user_annotations__type", str)
|
||||
self.__check_attr("user_annotations__local_file_csv__directory", (type(None), str))
|
||||
self.__check_attr("user_annotations__local_file_csv__file", (type(None), str))
|
||||
self.__check_attr("user_annotations__ontology__enable", bool)
|
||||
self.__check_attr("user_annotations__ontology__obo_location", (type(None), str))
|
||||
self.check_attr("user_annotations__enable", bool)
|
||||
self.check_attr("user_annotations__type", str)
|
||||
self.check_attr("user_annotations__local_file_csv__directory", (type(None), str))
|
||||
self.check_attr("user_annotations__local_file_csv__file", (type(None), str))
|
||||
self.check_attr("user_annotations__ontology__enable", bool)
|
||||
self.check_attr("user_annotations__ontology__obo_location", (type(None), str))
|
||||
|
||||
if self.user_annotations__enable:
|
||||
# TODO, replace this with a factory pattern once we have more than one way
|
||||
@@ -458,8 +735,11 @@ class AppConfig(object):
|
||||
|
||||
# if the user has specified a fixed label file, go ahead and validate it
|
||||
# so that we can remove errors early in the process.
|
||||
if self.single_dataset__datapath and self.user_annotations__local_file_csv__file:
|
||||
with self.matrix_data_cache_manager.data_adaptor(self.single_dataset__datapath, self) as data_adaptor:
|
||||
server_config = self.app_config.server_config
|
||||
if server_config.single_dataset__datapath and self.user_annotations__local_file_csv__file:
|
||||
with server_config.matrix_data_cache_manager.data_adaptor(
|
||||
self.tag, server_config.single_dataset__datapath, self.app_config
|
||||
) as data_adaptor:
|
||||
data_adaptor.check_new_labels(self.user_annotations.read_labels(data_adaptor))
|
||||
|
||||
if self.user_annotations__ontology__enable or self.user_annotations__ontology__obo_location:
|
||||
@@ -487,135 +767,29 @@ class AppConfig(object):
|
||||
)
|
||||
|
||||
def handle_embeddings(self, context):
|
||||
self.__check_attr("embeddings__names", list)
|
||||
self.__check_attr("embeddings__enable_reembedding", bool)
|
||||
self.check_attr("embeddings__names", list)
|
||||
self.check_attr("embeddings__enable_reembedding", bool)
|
||||
|
||||
if self.single_dataset__datapath:
|
||||
if self.app_config.server_config.single_dataset__datapath:
|
||||
if self.embeddings__enable_reembedding:
|
||||
matrix_data_loader = MatrixDataLoader(self.single_dataset__datapath, app_config=self)
|
||||
matrix_data_loader = MatrixDataLoader(self.single_dataset__datapath, app_config=self.app_config)
|
||||
if matrix_data_loader.matrix_data_type() != MatrixDataType.H5AD:
|
||||
raise ConfigurationError("'enable-reembedding is only supported with H5AD files.")
|
||||
if self.adaptor__anndata_adaptor__backed:
|
||||
raise ConfigurationError("enable-reembedding is not supported when run in --backed mode.")
|
||||
|
||||
def handle_diffexp(self, context):
|
||||
self.__check_attr("diffexp__enable", bool)
|
||||
self.__check_attr("diffexp__lfc_cutoff", float)
|
||||
self.__check_attr("diffexp__top_n", int)
|
||||
self.__check_attr("diffexp__alg_cxg__max_workers", (str, int))
|
||||
self.__check_attr("diffexp__alg_cxg__cpu_multiplier", int)
|
||||
self.__check_attr("diffexp__alg_cxg__target_workunit", int)
|
||||
self.check_attr("diffexp__enable", bool)
|
||||
self.check_attr("diffexp__lfc_cutoff", float)
|
||||
self.check_attr("diffexp__top_n", int)
|
||||
|
||||
if self.single_dataset__datapath:
|
||||
with self.matrix_data_cache_manager.data_adaptor(self.single_dataset__datapath, self) as data_adaptor:
|
||||
server_config = self.app_config.server_config
|
||||
if server_config.single_dataset__datapath:
|
||||
with server_config.matrix_data_cache_manager.data_adaptor(
|
||||
self.tag, server_config.single_dataset__datapath, self.app_config
|
||||
) as data_adaptor:
|
||||
if self.diffexp__enable and data_adaptor.parameters.get("diffexp_may_be_slow", False):
|
||||
context["messagefn"](
|
||||
"CAUTION: due to the size of your dataset, "
|
||||
"running differential expression may take longer or fail."
|
||||
)
|
||||
|
||||
max_workers = self.diffexp__alg_cxg__max_workers
|
||||
cpu_multiplier = self.diffexp__alg_cxg__cpu_multiplier
|
||||
cpu_count = os.cpu_count()
|
||||
max_workers = min(max_workers, cpu_multiplier * cpu_count)
|
||||
diffexp_tiledb.set_config(max_workers, self.diffexp__alg_cxg__target_workunit)
|
||||
|
||||
def handle_adaptor(self, context):
|
||||
# cxg
|
||||
self.__check_attr("adaptor__cxg_adaptor__tiledb_ctx", dict)
|
||||
regionkey = "vfs.s3.region"
|
||||
if regionkey not in self.adaptor__cxg_adaptor__tiledb_ctx:
|
||||
if type(self.data_locator__s3__region_name) == str:
|
||||
self.adaptor__cxg_adaptor__tiledb_ctx[regionkey] = self.data_locator__s3__region_name
|
||||
|
||||
from server.data_cxg.cxg_adaptor import CxgAdaptor
|
||||
|
||||
CxgAdaptor.set_tiledb_context(self.adaptor__cxg_adaptor__tiledb_ctx)
|
||||
|
||||
# anndata
|
||||
self.__check_attr("adaptor__anndata_adaptor__backed", bool)
|
||||
|
||||
def handle_limits(self, context):
|
||||
self.__check_attr("limits__diffexp_cellcount_max", (type(None), int))
|
||||
self.__check_attr("limits__column_request_max", (type(None), int))
|
||||
|
||||
def get_title(self, data_adaptor):
|
||||
return self.single_dataset__title if self.single_dataset__title else data_adaptor.get_title()
|
||||
|
||||
def get_about(self, data_adaptor):
|
||||
return self.single_dataset__about if self.single_dataset__about else data_adaptor.get_about()
|
||||
|
||||
def get_client_config(self, data_adaptor, annotation=None):
|
||||
"""
|
||||
Return the configuration as required by the /config REST route
|
||||
"""
|
||||
|
||||
# FIXME The current set of config is not consistently presented:
|
||||
# we have camalCase, hyphen-text, and underscore_text
|
||||
|
||||
# make sure the configuration has been checked.
|
||||
self.check_config()
|
||||
|
||||
# features
|
||||
features = [f.todict() for f in data_adaptor.get_features(annotation)]
|
||||
|
||||
# display_names
|
||||
title = self.get_title(data_adaptor)
|
||||
about = self.get_about(data_adaptor)
|
||||
|
||||
display_names = dict(engine=data_adaptor.get_name(), dataset=title)
|
||||
|
||||
# library_versions
|
||||
library_versions = {}
|
||||
library_versions.update(data_adaptor.get_library_versions())
|
||||
library_versions["cellxgene"] = cellxgene_version
|
||||
|
||||
# links
|
||||
links = {"about-dataset": about}
|
||||
|
||||
# parameters
|
||||
parameters = {
|
||||
"layout": self.embeddings__names,
|
||||
"max-category-items": self.presentation__max_categories,
|
||||
"obs_names": self.single_dataset__obs_names,
|
||||
"var_names": self.single_dataset__var_names,
|
||||
"diffexp_lfc_cutoff": self.diffexp__lfc_cutoff,
|
||||
"backed": self.adaptor__anndata_adaptor__backed,
|
||||
"disable-diffexp": not self.diffexp__enable,
|
||||
"enable-reembedding": self.embeddings__enable_reembedding,
|
||||
"annotations": False,
|
||||
"annotations_file": None,
|
||||
"annotations_dir": None,
|
||||
"annotations_cell_ontology_enabled": False,
|
||||
"annotations_cell_ontology_obopath": None,
|
||||
"annotations_cell_ontology_terms": None,
|
||||
"custom_colors": self.presentation__custom_colors,
|
||||
"diffexp-may-be-slow": False,
|
||||
"about_legal_tos": self.server__about_legal_tos,
|
||||
"about_legal_privacy": self.server__about_legal_privacy,
|
||||
}
|
||||
|
||||
data_adaptor.update_parameters(parameters)
|
||||
if annotation:
|
||||
annotation.update_parameters(parameters, data_adaptor)
|
||||
|
||||
# gather it all together
|
||||
c = {}
|
||||
config = c["config"] = {}
|
||||
config["features"] = features
|
||||
config["displayNames"] = display_names
|
||||
config["library_versions"] = library_versions
|
||||
config["links"] = links
|
||||
config["parameters"] = parameters
|
||||
config["limits"] = {
|
||||
"column_request_max": self.limits__column_request_max,
|
||||
"diffexp_cellcount_max": self.limits__diffexp_cellcount_max,
|
||||
}
|
||||
|
||||
return c
|
||||
|
||||
def exceeds_limit(self, limit_name, value):
|
||||
limit_value = getattr(self, "limits__" + limit_name, None)
|
||||
if limit_value is None: # disabled
|
||||
return False
|
||||
return value > limit_value
|
||||
|
||||
+138
-108
@@ -1,129 +1,159 @@
|
||||
import yaml
|
||||
|
||||
default_config = """
|
||||
# cellxgene configuration
|
||||
|
||||
server:
|
||||
verbose: false
|
||||
debug: false
|
||||
host: "127.0.0.1"
|
||||
port : null
|
||||
app:
|
||||
verbose: false
|
||||
debug: false
|
||||
host: "127.0.0.1"
|
||||
port : null
|
||||
open_browser: false
|
||||
force_https: false
|
||||
flask_secret_key: null
|
||||
generate_cache_control_headers: false
|
||||
server_timing_headers: false
|
||||
csp_directives: null
|
||||
|
||||
# Scripts can be a list of either file names (string) or dicts containing keys src, integrity and crossorigin.
|
||||
# these will be injected into the index template as script tags with these attributes set.
|
||||
scripts: []
|
||||
# Inline scripts are a list of file names, where the contents of the file will be injected into the index.
|
||||
inline_scripts: []
|
||||
multi_dataset:
|
||||
# If dataroot is set, then cellxgene may serve multiple datasets. This parameter is not
|
||||
# compatible with single_dataset/datapath.
|
||||
# dataroot may be a string, representing the path to a directory or S3 prefix. In this
|
||||
# case the datasets in that location are accessed from <server>/d/<datasetname>.
|
||||
# example:
|
||||
# dataroot: /path/to/datasets/
|
||||
# or
|
||||
# dataroot: s3://bucket/prefix/
|
||||
#
|
||||
# As an alternative, dataroot can be a dictionary, where a dataset key is associated with a base_url
|
||||
# and a dataroot.
|
||||
# example:
|
||||
# dataroot:
|
||||
# d1:
|
||||
# base_url: set1
|
||||
# dataroot: /path/to/set1_datasets/
|
||||
# d2:
|
||||
# base_url: set2/subdir
|
||||
# dataroot: /path/to/set2_datasets/
|
||||
#
|
||||
# In this case, datasets can be accessed from <server>/set1/<datasetname> or
|
||||
# <server>/set2/subdir/<datasetname>. It is possible to have different dataset configurations
|
||||
# for datasets accessed through different dataroots. For example, in one dataroot, the
|
||||
# user annotations could be enabled, and in another dataroot they could be disabled.
|
||||
# To specify dataroot configurations, add a new top level dictionary to the config named
|
||||
# per_dataset_config. Within per_dataset_config create a dictionary for each dataroot to specialize
|
||||
# ("d1" or "d2" from the example). Each of these dictionaries has the exact same form as the "dataset"
|
||||
# dictionary (see below).
|
||||
# When this approach is used, the values for each configuration option are checked in
|
||||
# this order: per_dataset_config/<key>, dataset, then the default values.
|
||||
#
|
||||
# example:
|
||||
#
|
||||
# per_dataset_config:
|
||||
# d1:
|
||||
# user_annotations:
|
||||
# enable: false
|
||||
# d2:
|
||||
# user_annotations:
|
||||
# enable: true
|
||||
|
||||
open_browser: false
|
||||
about_legal_tos: null
|
||||
about_legal_privacy: null
|
||||
force_https: false
|
||||
flask_secret_key: null
|
||||
generate_cache_control_headers: false
|
||||
server_timing_headers: false
|
||||
csp_directives: null
|
||||
dataroot: null
|
||||
|
||||
presentation:
|
||||
max_categories: 1000
|
||||
custom_colors: true
|
||||
# The index page when in multi-dataset mode:
|
||||
# false or null: this returns a 404 code
|
||||
# true: loads a test index page, which links to the datasets that are available in the dataroot
|
||||
# string/URL: redirect to this URL: flask.redirect(config.multi_dataset__index)
|
||||
index: false
|
||||
|
||||
multi_dataset:
|
||||
# If dataroot is set, then cellxgene may serve multiple datasets. This parameter is not
|
||||
# compatable with single_dataset/datapath.
|
||||
# dataroot may be a string, representing the path to a directory or S3 prefix. In this
|
||||
# case the datasets in that location are accessed from <server>/d/<datasetname>.
|
||||
# example:
|
||||
# dataroot: /path/to/datasets/
|
||||
# or
|
||||
# dataroot: s3://bucket/prefix/
|
||||
#
|
||||
# As an alternative, dataroot can be a dictionary, mapping url prefixes to dataroot paths.
|
||||
# example:
|
||||
# dataroot:
|
||||
# set1 : /path/to/set1_datasets/
|
||||
# set2 : /path/to/set2_datasets/
|
||||
# In this case, datasets can be accessed from <server>/set1/<datasetname> or
|
||||
# <server>/set2/<datasetname>.
|
||||
# A list of allowed matrix types. If an empty list, then all matrix types are allowed
|
||||
allowed_matrix_types: []
|
||||
|
||||
dataroot: null
|
||||
matrix_cache:
|
||||
# The maximum number of datasets that may be opened at one time. The least recently used dataset
|
||||
# is evicted from the cache first.
|
||||
max_datasets: 5
|
||||
|
||||
# The index page when in multi-dataset mode:
|
||||
# false or null: this returns a 404 code
|
||||
# true: loads a test index page, which links to the datasets that are available in the dataroot
|
||||
# string/URL: redirect to this URL: flask.redirect(config.multi_dataset__index)
|
||||
index: false
|
||||
# A matrix is automatically removed from the cache after timelimit_s number of seconds.
|
||||
# If timelimit_s is set to None, then there is no time limit.
|
||||
timelimit_s: 30
|
||||
|
||||
# A list of allowed matrix types. If an empty list, then all matrix types are allowed
|
||||
allowed_matrix_types: []
|
||||
single_dataset:
|
||||
# If datapath is set, then cellxgene with serve a single dataset located at datapath. This parameter is not
|
||||
# compatible with multi_dataset/dataroot.
|
||||
datapath: null
|
||||
obs_names: null
|
||||
var_names: null
|
||||
about: null
|
||||
title: null
|
||||
|
||||
matrix_cache:
|
||||
# The maximum number of datasets that may be opened at one time. The least recently used dataset
|
||||
# is evicted from the cache first.
|
||||
max_datasets: 5
|
||||
diffexp:
|
||||
alg_cxg:
|
||||
# The number of threads to use is computed from: min(max_workers, cpu_multipler * cpu_count).
|
||||
# Where cpu_count is determined at runtime.
|
||||
max_workers: 64
|
||||
cpu_multiplier: 4
|
||||
|
||||
# A matrix is automatically removed from the cache after timelimit_s number of seconds.
|
||||
# If timelimit_s is set to None, then there is no time limit.
|
||||
timelimit_s: 30
|
||||
# The target number of matrix elements that are evaluated
|
||||
# together in one thread.
|
||||
target_workunit: 16_000_000
|
||||
|
||||
single_dataset:
|
||||
datapath: null
|
||||
obs_names: null
|
||||
var_names: null
|
||||
about: null
|
||||
title: null
|
||||
data_locator:
|
||||
s3:
|
||||
# s3 region name.
|
||||
# if true, then the s3 location is automatically determined from the datapath or dataroot.
|
||||
# if false/null, then do not set.
|
||||
# if a string, then use that value (e.g. us-east-1).
|
||||
region_name: true
|
||||
|
||||
user_annotations:
|
||||
enable: true
|
||||
type: local_file_csv
|
||||
local_file_csv:
|
||||
directory: null
|
||||
file: null
|
||||
ontology:
|
||||
enable: false
|
||||
obo_location: null
|
||||
adaptor:
|
||||
cxg_adaptor:
|
||||
# The key/values under tiledb_ctx will be used to initialize the tiledb Context.
|
||||
# If 'vfs.s3.region' is not set, then it will automatically use the setting from
|
||||
# data_locator / s3 / region_name.
|
||||
tiledb_ctx:
|
||||
sm.tile_cache_size: 8589934592
|
||||
sm.num_reader_threads: 32
|
||||
|
||||
embeddings:
|
||||
names : []
|
||||
enable_reembedding: false
|
||||
|
||||
diffexp:
|
||||
enable: true
|
||||
lfc_cutoff: 0.01
|
||||
top_n: 10
|
||||
alg_cxg:
|
||||
# The number of threads to use is computed from: min(max_workers, cpu_multipler * cpu_count).
|
||||
# Where cpu_count is determined at runtime.
|
||||
max_workers: 64
|
||||
cpu_multiplier: 4
|
||||
|
||||
# The target number of matrix elements that are evaluated
|
||||
# together in one thread.
|
||||
target_workunit: 16_000_000
|
||||
|
||||
data_locator:
|
||||
s3:
|
||||
# s3 region name.
|
||||
# if true, then the s3 location is automatically determined from the datapath or dataroot.
|
||||
# if false/null, then do not set.
|
||||
# if a string, then use that value (e.g. us-east-1).
|
||||
region_name: true
|
||||
|
||||
adaptor:
|
||||
cxg_adaptor:
|
||||
# The key/values under tiledb_ctx will be used to initialize the tiledb Context.
|
||||
# If 'vfs.s3.region' is not set, then it will automatically use the setting from
|
||||
# data_locator / s3 / region_name.
|
||||
tiledb_ctx:
|
||||
sm.tile_cache_size: 8589934592
|
||||
sm.num_reader_threads: 32
|
||||
|
||||
anndata_adaptor:
|
||||
anndata_adaptor:
|
||||
backed: false
|
||||
|
||||
limits:
|
||||
column_request_max: 32
|
||||
diffexp_cellcount_max: null
|
||||
limits:
|
||||
column_request_max: 32
|
||||
diffexp_cellcount_max: null
|
||||
|
||||
|
||||
dataset:
|
||||
app:
|
||||
# Scripts can be a list of either file names (string) or dicts containing keys src, integrity and crossorigin.
|
||||
# these will be injected into the index template as script tags with these attributes set.
|
||||
scripts: []
|
||||
# Inline scripts are a list of file names, where the contents of the file will be injected into the index.
|
||||
inline_scripts: []
|
||||
|
||||
about_legal_tos: null
|
||||
about_legal_privacy: null
|
||||
|
||||
presentation:
|
||||
max_categories: 1000
|
||||
custom_colors: true
|
||||
|
||||
user_annotations:
|
||||
enable: true
|
||||
type: local_file_csv
|
||||
local_file_csv:
|
||||
directory: null
|
||||
file: null
|
||||
ontology:
|
||||
enable: false
|
||||
obo_location: null
|
||||
|
||||
embeddings:
|
||||
names : []
|
||||
enable_reembedding: false
|
||||
|
||||
diffexp:
|
||||
enable: true
|
||||
lfc_cutoff: 0.01
|
||||
top_n: 10
|
||||
|
||||
"""
|
||||
|
||||
|
||||
@@ -24,10 +24,12 @@ def health_check(config):
|
||||
health = {"status": None, "version": "1", "releaseID": cellxgene_version}
|
||||
|
||||
checks = False
|
||||
if config.single_dataset__datapath is not None:
|
||||
checks = _is_accessible(config.single_dataset__datapath, config)
|
||||
elif config.multi_dataset__dataroot is not None:
|
||||
checks = all([_is_accessible(datapath, config) for datapath in config.multi_dataset__dataroot.values()])
|
||||
server_config = config.server_config
|
||||
if config.is_multi_dataset():
|
||||
dataroots = [datapath_dict["dataroot"] for datapath_dict in server_config.multi_dataset__dataroot.values()]
|
||||
checks = all([_is_accessible(dataroot, server_config) for dataroot in dataroots])
|
||||
else:
|
||||
checks = _is_accessible(server_config.single_dataset__datapath, server_config)
|
||||
|
||||
health["status"] = "pass" if checks else "fail"
|
||||
code = HTTPStatus.OK if health["status"] == "pass" else HTTPStatus.BAD_REQUEST
|
||||
|
||||
+21
-16
@@ -97,12 +97,13 @@ def _query_parameter_to_filter(args):
|
||||
return result
|
||||
|
||||
|
||||
def schema_get_helper(data_adaptor, annotations):
|
||||
def schema_get_helper(data_adaptor):
|
||||
"""helper function to gather the schema from the data source and annotations"""
|
||||
schema = data_adaptor.get_schema()
|
||||
schema = copy.deepcopy(schema)
|
||||
|
||||
# add label obs annotations as needed
|
||||
annotations = data_adaptor.dataset_config.user_annotations
|
||||
if annotations is not None:
|
||||
label_schema = annotations.get_schema(data_adaptor)
|
||||
schema["annotations"]["obs"]["columns"].extend(label_schema)
|
||||
@@ -110,20 +111,20 @@ def schema_get_helper(data_adaptor, annotations):
|
||||
return schema
|
||||
|
||||
|
||||
def schema_get(data_adaptor, annotations):
|
||||
schema = schema_get_helper(data_adaptor, annotations)
|
||||
def schema_get(data_adaptor):
|
||||
schema = schema_get_helper(data_adaptor)
|
||||
return make_response(jsonify({"schema": schema}), HTTPStatus.OK)
|
||||
|
||||
|
||||
def config_get(app_config, data_adaptor, annotations):
|
||||
config = app_config.get_client_config(data_adaptor, annotations)
|
||||
def config_get(app_config, data_adaptor):
|
||||
config = app_config.get_client_config(data_adaptor)
|
||||
return make_response(jsonify(config), HTTPStatus.OK)
|
||||
|
||||
|
||||
def annotations_obs_get(request, data_adaptor, annotations):
|
||||
def annotations_obs_get(request, data_adaptor):
|
||||
fields = request.args.getlist("annotation-name", None)
|
||||
num_columns_requested = len(data_adaptor.get_obs_keys()) if len(fields) == 0 else len(fields)
|
||||
if data_adaptor.config.exceeds_limit("column_request_max", num_columns_requested):
|
||||
if data_adaptor.server_config.exceeds_limit("column_request_max", num_columns_requested):
|
||||
return abort(HTTPStatus.BAD_REQUEST)
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
if preferred_mimetype != "application/octet-stream":
|
||||
@@ -131,6 +132,7 @@ def annotations_obs_get(request, data_adaptor, annotations):
|
||||
|
||||
try:
|
||||
labels = None
|
||||
annotations = data_adaptor.dataset_config.user_annotations
|
||||
if annotations:
|
||||
labels = annotations.read_labels(data_adaptor)
|
||||
fbs = data_adaptor.annotation_to_fbs_matrix(Axis.OBS, fields, labels)
|
||||
@@ -139,8 +141,9 @@ def annotations_obs_get(request, data_adaptor, annotations):
|
||||
return abort_and_log(HTTPStatus.BAD_REQUEST, str(e), include_exc_info=True)
|
||||
|
||||
|
||||
def annotations_put_fbs_helper(data_adaptor, annotations, fbs):
|
||||
def annotations_put_fbs_helper(data_adaptor, fbs):
|
||||
"""helper function to write annotations from fbs"""
|
||||
annotations = data_adaptor.dataset_config.user_annotations
|
||||
if annotations is None:
|
||||
raise DisabledFeatureError("Writable annotations are not enabled")
|
||||
|
||||
@@ -150,7 +153,8 @@ def annotations_put_fbs_helper(data_adaptor, annotations, fbs):
|
||||
annotations.write_labels(new_label_df, data_adaptor)
|
||||
|
||||
|
||||
def annotations_obs_put(request, data_adaptor, annotations):
|
||||
def annotations_obs_put(request, data_adaptor):
|
||||
annotations = data_adaptor.dataset_config.user_annotations
|
||||
if annotations is None:
|
||||
return abort(HTTPStatus.NOT_IMPLEMENTED)
|
||||
|
||||
@@ -163,17 +167,17 @@ def annotations_obs_put(request, data_adaptor, annotations):
|
||||
annotations.set_collection(anno_collection)
|
||||
|
||||
try:
|
||||
annotations_put_fbs_helper(data_adaptor, annotations, fbs)
|
||||
annotations_put_fbs_helper(data_adaptor, fbs)
|
||||
res = json.dumps({"status": "OK"})
|
||||
return make_response(res, HTTPStatus.OK, {"Content-Type": "application/json"})
|
||||
except (ValueError, DisabledFeatureError, KeyError) as e:
|
||||
return abort_and_log(HTTPStatus.BAD_REQUEST, str(e), include_exc_info=True)
|
||||
|
||||
|
||||
def annotations_var_get(request, data_adaptor, annotations):
|
||||
def annotations_var_get(request, data_adaptor):
|
||||
fields = request.args.getlist("annotation-name", None)
|
||||
num_columns_requested = len(data_adaptor.get_var_keys()) if len(fields) == 0 else len(fields)
|
||||
if data_adaptor.config.exceeds_limit("column_request_max", num_columns_requested):
|
||||
if data_adaptor.server_config.exceeds_limit("column_request_max", num_columns_requested):
|
||||
return abort(HTTPStatus.BAD_REQUEST)
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
if preferred_mimetype != "application/octet-stream":
|
||||
@@ -181,6 +185,7 @@ def annotations_var_get(request, data_adaptor, annotations):
|
||||
|
||||
try:
|
||||
labels = None
|
||||
annotations = data_adaptor.dataset_config.user_annotations
|
||||
if annotations is not None:
|
||||
labels = annotations.read_labels(data_adaptor)
|
||||
return make_response(
|
||||
@@ -226,7 +231,7 @@ def data_var_get(request, data_adaptor):
|
||||
|
||||
|
||||
def colors_get(data_adaptor):
|
||||
if not data_adaptor.config.presentation__custom_colors:
|
||||
if not data_adaptor.dataset_config.presentation__custom_colors:
|
||||
return make_response(jsonify({}), HTTPStatus.OK)
|
||||
try:
|
||||
return make_response(jsonify(data_adaptor.get_colors()), HTTPStatus.OK)
|
||||
@@ -235,7 +240,7 @@ def colors_get(data_adaptor):
|
||||
|
||||
|
||||
def diffexp_obs_post(request, data_adaptor):
|
||||
if not data_adaptor.config.diffexp__enable:
|
||||
if not data_adaptor.dataset_config.diffexp__enable:
|
||||
return abort(HTTPStatus.NOT_IMPLEMENTED)
|
||||
|
||||
args = request.get_json()
|
||||
@@ -273,7 +278,7 @@ def diffexp_obs_post(request, data_adaptor):
|
||||
def layout_obs_get(request, data_adaptor):
|
||||
fields = request.args.getlist("layout-name", None)
|
||||
num_columns_requested = len(data_adaptor.get_embedding_names()) if len(fields) == 0 else len(fields)
|
||||
if data_adaptor.config.exceeds_limit("column_request_max", num_columns_requested):
|
||||
if data_adaptor.server_config.exceeds_limit("column_request_max", num_columns_requested):
|
||||
return abort(HTTPStatus.BAD_REQUEST)
|
||||
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
@@ -296,7 +301,7 @@ def layout_obs_get(request, data_adaptor):
|
||||
|
||||
|
||||
def layout_obs_put(request, data_adaptor):
|
||||
if not data_adaptor.config.embedding__enable_reembedding:
|
||||
if not data_adaptor.dataset_config.embedding__enable_reembedding:
|
||||
return abort(HTTPStatus.NOT_IMPLEMENTED)
|
||||
|
||||
preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
|
||||
|
||||
Reference in New Issue
Block a user