refactor config to support different config options for datasets in different dataroots. (#1596)

This will give us the ability to specify different config options for
different dataroots.

the key of the dataroot dictionary is no longer the same as the dataroot_url.
Previously key==dataroot_url, and now those are separated.

Added an "is_multi_dataset" function to simplify logic where it branched on single vs multi.

Simplified the rest.py interface by no longer passing in the user annotations object, since
that can be retrieved from the dataset.
This commit is contained in:
bmccandless
2020-07-10 16:21:40 -07:00
committed by GitHub
parent 13246cb6d1
commit f69d141336
21 changed files with 1021 additions and 710 deletions
+138 -108
View File
@@ -1,129 +1,159 @@
import yaml
default_config = """
# cellxgene configuration
server:
verbose: false
debug: false
host: "127.0.0.1"
port : null
app:
verbose: false
debug: false
host: "127.0.0.1"
port : null
open_browser: false
force_https: false
flask_secret_key: null
generate_cache_control_headers: false
server_timing_headers: false
csp_directives: null
# Scripts can be a list of either file names (string) or dicts containing keys src, integrity and crossorigin.
# these will be injected into the index template as script tags with these attributes set.
scripts: []
# Inline scripts are a list of file names, where the contents of the file will be injected into the index.
inline_scripts: []
multi_dataset:
# If dataroot is set, then cellxgene may serve multiple datasets. This parameter is not
# compatible with single_dataset/datapath.
# dataroot may be a string, representing the path to a directory or S3 prefix. In this
# case the datasets in that location are accessed from <server>/d/<datasetname>.
# example:
# dataroot: /path/to/datasets/
# or
# dataroot: s3://bucket/prefix/
#
# As an alternative, dataroot can be a dictionary, where a dataset key is associated with a base_url
# and a dataroot.
# example:
# dataroot:
# d1:
# base_url: set1
# dataroot: /path/to/set1_datasets/
# d2:
# base_url: set2/subdir
# dataroot: /path/to/set2_datasets/
#
# In this case, datasets can be accessed from <server>/set1/<datasetname> or
# <server>/set2/subdir/<datasetname>. It is possible to have different dataset configurations
# for datasets accessed through different dataroots. For example, in one dataroot, the
# user annotations could be enabled, and in another dataroot they could be disabled.
# To specify dataroot configurations, add a new top level dictionary to the config named
# per_dataset_config. Within per_dataset_config create a dictionary for each dataroot to specialize
# ("d1" or "d2" from the example). Each of these dictionaries has the exact same form as the "dataset"
# dictionary (see below).
# When this approach is used, the values for each configuration option are checked in
# this order: per_dataset_config/<key>, dataset, then the default values.
#
# example:
#
# per_dataset_config:
# d1:
# user_annotations:
# enable: false
# d2:
# user_annotations:
# enable: true
open_browser: false
about_legal_tos: null
about_legal_privacy: null
force_https: false
flask_secret_key: null
generate_cache_control_headers: false
server_timing_headers: false
csp_directives: null
dataroot: null
presentation:
max_categories: 1000
custom_colors: true
# The index page when in multi-dataset mode:
# false or null: this returns a 404 code
# true: loads a test index page, which links to the datasets that are available in the dataroot
# string/URL: redirect to this URL: flask.redirect(config.multi_dataset__index)
index: false
multi_dataset:
# If dataroot is set, then cellxgene may serve multiple datasets. This parameter is not
# compatable with single_dataset/datapath.
# dataroot may be a string, representing the path to a directory or S3 prefix. In this
# case the datasets in that location are accessed from <server>/d/<datasetname>.
# example:
# dataroot: /path/to/datasets/
# or
# dataroot: s3://bucket/prefix/
#
# As an alternative, dataroot can be a dictionary, mapping url prefixes to dataroot paths.
# example:
# dataroot:
# set1 : /path/to/set1_datasets/
# set2 : /path/to/set2_datasets/
# In this case, datasets can be accessed from <server>/set1/<datasetname> or
# <server>/set2/<datasetname>.
# A list of allowed matrix types. If an empty list, then all matrix types are allowed
allowed_matrix_types: []
dataroot: null
matrix_cache:
# The maximum number of datasets that may be opened at one time. The least recently used dataset
# is evicted from the cache first.
max_datasets: 5
# The index page when in multi-dataset mode:
# false or null: this returns a 404 code
# true: loads a test index page, which links to the datasets that are available in the dataroot
# string/URL: redirect to this URL: flask.redirect(config.multi_dataset__index)
index: false
# A matrix is automatically removed from the cache after timelimit_s number of seconds.
# If timelimit_s is set to None, then there is no time limit.
timelimit_s: 30
# A list of allowed matrix types. If an empty list, then all matrix types are allowed
allowed_matrix_types: []
single_dataset:
# If datapath is set, then cellxgene with serve a single dataset located at datapath. This parameter is not
# compatible with multi_dataset/dataroot.
datapath: null
obs_names: null
var_names: null
about: null
title: null
matrix_cache:
# The maximum number of datasets that may be opened at one time. The least recently used dataset
# is evicted from the cache first.
max_datasets: 5
diffexp:
alg_cxg:
# The number of threads to use is computed from: min(max_workers, cpu_multipler * cpu_count).
# Where cpu_count is determined at runtime.
max_workers: 64
cpu_multiplier: 4
# A matrix is automatically removed from the cache after timelimit_s number of seconds.
# If timelimit_s is set to None, then there is no time limit.
timelimit_s: 30
# The target number of matrix elements that are evaluated
# together in one thread.
target_workunit: 16_000_000
single_dataset:
datapath: null
obs_names: null
var_names: null
about: null
title: null
data_locator:
s3:
# s3 region name.
# if true, then the s3 location is automatically determined from the datapath or dataroot.
# if false/null, then do not set.
# if a string, then use that value (e.g. us-east-1).
region_name: true
user_annotations:
enable: true
type: local_file_csv
local_file_csv:
directory: null
file: null
ontology:
enable: false
obo_location: null
adaptor:
cxg_adaptor:
# The key/values under tiledb_ctx will be used to initialize the tiledb Context.
# If 'vfs.s3.region' is not set, then it will automatically use the setting from
# data_locator / s3 / region_name.
tiledb_ctx:
sm.tile_cache_size: 8589934592
sm.num_reader_threads: 32
embeddings:
names : []
enable_reembedding: false
diffexp:
enable: true
lfc_cutoff: 0.01
top_n: 10
alg_cxg:
# The number of threads to use is computed from: min(max_workers, cpu_multipler * cpu_count).
# Where cpu_count is determined at runtime.
max_workers: 64
cpu_multiplier: 4
# The target number of matrix elements that are evaluated
# together in one thread.
target_workunit: 16_000_000
data_locator:
s3:
# s3 region name.
# if true, then the s3 location is automatically determined from the datapath or dataroot.
# if false/null, then do not set.
# if a string, then use that value (e.g. us-east-1).
region_name: true
adaptor:
cxg_adaptor:
# The key/values under tiledb_ctx will be used to initialize the tiledb Context.
# If 'vfs.s3.region' is not set, then it will automatically use the setting from
# data_locator / s3 / region_name.
tiledb_ctx:
sm.tile_cache_size: 8589934592
sm.num_reader_threads: 32
anndata_adaptor:
anndata_adaptor:
backed: false
limits:
column_request_max: 32
diffexp_cellcount_max: null
limits:
column_request_max: 32
diffexp_cellcount_max: null
dataset:
app:
# Scripts can be a list of either file names (string) or dicts containing keys src, integrity and crossorigin.
# these will be injected into the index template as script tags with these attributes set.
scripts: []
# Inline scripts are a list of file names, where the contents of the file will be injected into the index.
inline_scripts: []
about_legal_tos: null
about_legal_privacy: null
presentation:
max_categories: 1000
custom_colors: true
user_annotations:
enable: true
type: local_file_csv
local_file_csv:
directory: null
file: null
ontology:
enable: false
obo_location: null
embeddings:
names : []
enable_reembedding: false
diffexp:
enable: true
lfc_cutoff: 0.01
top_n: 10
"""