refactor config to support different config options for datasets in different dataroots. (#1596)

This will give us the ability to specify different config options for
different dataroots.

the key of the dataroot dictionary is no longer the same as the dataroot_url.
Previously key==dataroot_url, and now those are separated.

Added an "is_multi_dataset" function to simplify logic where it branched on single vs multi.

Simplified the rest.py interface by no longer passing in the user annotations object, since
that can be retrieved from the dataset.
This commit is contained in:
bmccandless
2020-07-10 16:21:40 -07:00
committed by GitHub
parent 13246cb6d1
commit f69d141336
21 changed files with 1021 additions and 710 deletions
+19 -17
View File
@@ -14,12 +14,17 @@ from server.common.app_config import AppFeature, AppConfig
class DataAdaptor(metaclass=ABCMeta):
"""Base class for loading and accessing matrix data"""
def __init__(self, config):
if type(config) != AppConfig:
def __init__(self, data_locator, app_config, dataset_config=None):
if type(app_config) != AppConfig:
raise TypeError("config expected to be of type AppConfig")
# location to the dataset
self.data_locator = data_locator
# config is the application configuration
self.config = config
self.app_config = app_config
self.server_config = self.app_config.server_config
self.dataset_config = dataset_config or app_config.default_dataset_config
# parameters set by this data adaptor based on the data.
self.parameters = {}
@@ -31,7 +36,7 @@ class DataAdaptor(metaclass=ABCMeta):
@staticmethod
@abstractmethod
def open(data_locator, config):
def open(data_locator, app_config, dataset_config):
pass
@staticmethod
@@ -109,13 +114,11 @@ class DataAdaptor(metaclass=ABCMeta):
def cleanup(self):
pass
@abstractmethod
def get_location(self):
pass
@abstractmethod
def get_data_locator(self):
pass
return self.data_locator
def get_location(self):
return self.data_locator.uri_or_path
def get_about(self):
return None
@@ -149,8 +152,8 @@ class DataAdaptor(metaclass=ABCMeta):
features = [
AppFeature("/cluster/", method="POST", available=False),
AppFeature("/layout/obs", method="GET", available=self.get_embedding_names() is not None),
AppFeature("/layout/obs", method="PUT", available=self.config.embeddings__enable_reembedding),
AppFeature("/diffexp/", method="POST", available=self.config.diffexp__enable),
AppFeature("/layout/obs", method="PUT", available=self.dataset_config.embeddings__enable_reembedding),
AppFeature("/diffexp/", method="POST", available=self.dataset_config.diffexp__enable),
AppFeature("/annotations/obs", method="PUT", available=annotations is not None),
]
return features
@@ -260,7 +263,6 @@ class DataAdaptor(metaclass=ABCMeta):
* currently only supports access on VAR axis
* currently only supports filtering on VAR axis
"""
if axis != Axis.VAR:
raise ValueError("Only VAR dimension access is supported")
@@ -273,7 +275,7 @@ class DataAdaptor(metaclass=ABCMeta):
raise FilterError("filtering on obs unsupported")
num_columns = self.get_shape()[1] if var_selector is None else np.count_nonzero(var_selector)
if self.config.exceeds_limit("column_request_max", num_columns):
if self.server_config.exceeds_limit("column_request_max", num_columns):
raise ExceedsLimitError("Requested dataframe columns exceed column request limit")
X = self.get_X_array(obs_selector, var_selector)
@@ -301,14 +303,14 @@ class DataAdaptor(metaclass=ABCMeta):
except (KeyError, IndexError):
raise FilterError("Error parsing filter")
if top_n is None:
top_n = self.config.diffexp__top_n
top_n = self.dataset_config.diffexp__top_n
if self.config.exceeds_limit(
if self.server_config.exceeds_limit(
"diffexp_cellcount_max", np.count_nonzero(obs_mask_A) + np.count_nonzero(obs_mask_B)
):
raise ExceedsLimitError("Diffexp request exceeds max cell count limit")
result = self.compute_diffexp_ttest(obs_mask_A, obs_mask_B, top_n, self.config.diffexp__lfc_cutoff)
result = self.compute_diffexp_ttest(obs_mask_A, obs_mask_B, top_n, self.dataset_config.diffexp__lfc_cutoff)
try:
return jsonify_numpy(result)
+17 -15
View File
@@ -29,7 +29,7 @@ class MatrixDataCacheItem(object):
self.data_lock.r_release()
return None
def acquire_and_open(self, app_config):
def acquire_and_open(self, app_config, dataset_config=None):
"""returns the data_adaptor if cached. opens the data_adaptor if not.
In either case, the a reader lock is taken. Must call release when
the data_adaptor is no longer needed"""
@@ -43,7 +43,7 @@ class MatrixDataCacheItem(object):
if not self.data_adaptor:
try:
self.loader.pre_load_validation()
self.data_adaptor = self.loader.open(app_config)
self.data_adaptor = self.loader.open(app_config, dataset_config)
except Exception as e:
# necessary to hold the reader lock after an exception, since
# the release will occur when the context exits.
@@ -115,7 +115,7 @@ class MatrixDataCacheManager(object):
# will automatically be refreshed.
def __init__(self, max_cached, timelimit_s=None):
# key is location, value is a MatrixDataCacheInfo
# key is tuple(url_dataroot, location), value is a MatrixDataCacheInfo
self.datasets = {}
# lock to protect the datasets
@@ -131,20 +131,21 @@ class MatrixDataCacheManager(object):
self.timelimit_s = timelimit_s
@contextmanager
def data_adaptor(self, location, app_config):
def data_adaptor(self, url_dataroot, location, app_config):
# create a loader for to this location if it does not already exist
delete_adaptor = None
data_adaptor = None
cache_item = None
key = (url_dataroot, location)
with self.lock:
self.evict_old_datasets()
info = self.datasets.get(location)
info = self.datasets.get(key)
if info is not None:
info.last_access = time.time()
info.num_access += 1
self.datasets[location] = info
self.datasets[key] = info
data_adaptor = info.cache_item.acquire_existing()
cache_item = info.cache_item
@@ -165,19 +166,20 @@ class MatrixDataCacheManager(object):
loader = MatrixDataLoader(location, app_config=app_config)
cache_item = MatrixDataCacheItem(loader)
item = MatrixDataCacheInfo(cache_item, time.time())
self.datasets[location] = item
self.datasets[key] = item
try:
assert cache_item
if delete_adaptor:
delete_adaptor.delete()
if data_adaptor is None:
data_adaptor = cache_item.acquire_and_open(app_config)
dataset_config = app_config.get_dataset_config(url_dataroot)
data_adaptor = cache_item.acquire_and_open(app_config, dataset_config)
yield data_adaptor
except DatasetAccessError:
cache_item.release()
with self.lock:
del self.datasets[location]
del self.datasets[key]
cache_item.delete()
cache_item = None
raise
@@ -215,7 +217,7 @@ class MatrixDataType(Enum):
class MatrixDataLoader(object):
def __init__(self, location, matrix_data_type=None, app_config=None):
""" location can be a string or DataLocator """
region_name = None if app_config is None else app_config.data_locator__s3__region_name
region_name = None if app_config is None else app_config.server_config.data_locator__s3__region_name
self.location = DataLocator(location, region_name=region_name)
if not self.location.exists():
raise DatasetAccessError("Dataset does not exist.", HTTPStatus.NOT_FOUND)
@@ -254,12 +256,12 @@ class MatrixDataLoader(object):
if not app_config:
return True
if not app_config.multi_dataset__dataroot:
if not app_config.is_multi_dataset():
return True
if len(app_config.multi_dataset__allowed_matrix_types) == 0:
if len(app_config.server_config.multi_dataset__allowed_matrix_types) == 0:
return True
for val in app_config.multi_dataset__allowed_matrix_types:
for val in app_config.server_config.multi_dataset__allowed_matrix_types:
try:
if self.matrix_data_type == MatrixDataType(val):
return True
@@ -279,6 +281,6 @@ class MatrixDataLoader(object):
def file_size(self):
return self.matrix_type.file_size(self.location)
def open(self, app_config):
def open(self, app_config, dataset_config=None):
# create and return a DataAdaptor object
return self.matrix_type.open(self.location, app_config)
return self.matrix_type.open(self.location, app_config, dataset_config)