add config option to handle multiple dataroots (#1531)

#1513
This commit is contained in:
bmccandless
2020-06-04 19:29:37 -07:00
committed by GitHub
parent df6b42f5d6
commit 99d004d1f0
6 changed files with 240 additions and 60 deletions
+48 -14
View File
@@ -1,9 +1,9 @@
from server import __version__ as cellxgene_version
from flatten_dict import flatten
from flatten_dict import flatten, unflatten
import os
from os.path import splitext, basename, isdir
import sys
from urllib.parse import urlparse
from urllib.parse import urlparse, quote_plus
import yaml
import copy
@@ -125,17 +125,23 @@ class AppConfig(object):
dc = copy.deepcopy(config)
mapping = {}
# special case for tiledb_ctx whose value is a dict.
val = config.get("adaptor", {}).get("cxg_adaptor", {}).get("tiledb_ctx")
if val is not None:
mapping["adaptor__cxg_adaptor__tiledb_ctx"] = (("adaptor", "cxg_adaptor", "tiledb_ctx"), val)
del dc["adaptor"]["cxg_adaptor"]["tiledb_ctx"]
# special case for csp_directives whose value is a dict.
val = config.get("server", {}).get("csp_directives")
if val is not None:
mapping["server__csp_directives"] = (("server", "csp_directives"), val)
del dc["server"]["csp_directives"]
# special cases where the value could be a dict.
# If its value is not None, the entry is added to the mapping, and not included
# in the flattening below.
dictval_cases = [
("adaptor", "cxg_adaptor", "tiledb_ctx"),
("server", "csp_directives"),
("multi_dataset", "dataroot"),
]
for dictval_case in dictval_cases:
cur = dc
for part in dictval_case[:-1]:
cur = cur.get(part, {})
val = cur.get(dictval_case[-1])
if val is not None:
key = "__".join(dictval_case)
mapping[key] = (dictval_case, val)
del cur[dictval_case[-1]]
flat_config = flatten(dc)
for key, value in flat_config.items():
@@ -162,6 +168,14 @@ class AppConfig(object):
self.is_completed = False
def write_config(self, config_file):
"""output the config to a yaml file"""
mapping = self.__mapping(self.default_config)
for attrname in mapping.keys():
mapping[attrname] = getattr(self, attrname)
config = unflatten(mapping, splitter=lambda key: key.split("__"))
yaml.dump(config, open(config_file, "w"))
def update(self, **kw):
for key, value in kw.items():
if not hasattr(self, key):
@@ -302,6 +316,14 @@ class AppConfig(object):
self.__check_attr("data_locator__s3__region_name", (type(None), bool, str))
if self.data_locator__s3__region_name is True:
path = self.single_dataset__datapath or self.multi_dataset__dataroot
if type(path) == dict:
# if multi_dataset__dataroot is a dict, then use the first key
# that is in s3. NOTE: it is not supported to have dataroots
# in different regions.
paths = path.values()
for path in paths:
if path.startswith("s3://"):
break
if path.startswith("s3://"):
region_name = discover_s3_region_name(path)
if region_name is None:
@@ -366,7 +388,7 @@ class AppConfig(object):
)
def handle_multi_dataset(self, context):
self.__check_attr("multi_dataset__dataroot", (type(None), str))
self.__check_attr("multi_dataset__dataroot", (type(None), dict, str))
self.__check_attr("multi_dataset__index", (type(None), bool, str))
self.__check_attr("multi_dataset__allowed_matrix_types", list)
self.__check_attr("multi_dataset__matrix_cache__max_datasets", int)
@@ -375,6 +397,18 @@ class AppConfig(object):
if self.multi_dataset__dataroot is None:
return
if type(self.multi_dataset__dataroot) == str:
self.multi_dataset__dataroot = dict(d=self.multi_dataset__dataroot)
for key in self.multi_dataset__dataroot.keys():
# sanity check for well formed keys
if type(key) != str:
raise ConfigurationError(f"error in multi_dataset__dataroot {key}")
if quote_plus(key) != key:
raise ConfigurationError(f"error in multi_dataset__dataroot {key}")
if os.path.split(os.path.normpath(key))[-1] != key:
raise ConfigurationError(f"error in multi_dataset__dataroot {key}")
# error checking
for mtype in self.multi_dataset__allowed_matrix_types:
try:
+17
View File
@@ -29,6 +29,23 @@ presentation:
custom_colors: true
multi_dataset:
# If dataroot is set, then cellxgene may serve multiple datasets. This parameter is not
# compatable with single_dataset/datapath.
# dataroot may be a string, representing the path to a directory or S3 prefix. In this
# case the datasets in that location are accessed from <server>/d/<datasetname>.
# example:
# dataroot: /path/to/datasets/
# or
# dataroot: s3://bucket/prefix/
#
# As an alternative, dataroot can be a dictionary, mapping url prefixes to dataroot paths.
# example:
# dataroot:
# set1 : /path/to/set1_datasets/
# set2 : /path/to/set2_datasets/
# In this case, datasets can be accessed from <server>/set1/<datasetname> or
# <server>/set2/<datasetname>.
dataroot: null
# The index page when in multi-dataset mode:
+7 -6
View File
@@ -23,12 +23,13 @@ def health_check(config):
"""
health = {"status": None, "version": "1", "releaseID": cellxgene_version}
checks = [
(config.single_dataset__datapath is not None or config.multi_dataset__dataroot is not None),
_is_accessible(config.single_dataset__datapath, config),
_is_accessible(config.multi_dataset__dataroot, config),
]
health["status"] = "pass" if all(checks) else "fail"
checks = False
if config.single_dataset__datapath is not None:
checks = _is_accessible(config.single_dataset__datapath, config)
elif config.multi_dataset__dataroot is not None:
checks = all([_is_accessible(datapath, config) for datapath in config.multi_dataset__dataroot.values()])
health["status"] = "pass" if checks else "fail"
code = HTTPStatus.OK if health["status"] == "pass" else HTTPStatus.BAD_REQUEST
response = make_response(jsonify(health), code)
response.headers["Content-Type"] = "application/health+json"