Files
cellxgene/server/common/default_config.py
bmccandless 5c0b8c6296 Improve diffexp for tiledb (#1388)
* Improve diffexp for tiledb

- The rows from the A and B sets are gathered and processed at the same time.  In this
  way the matrix is only accessed once instead of twice for each tile.
- There is now a single thread queue that gets shared between all callers of the diffexp.
  This will slow down work if diffexp gets too busy.
- There is a target_workunit amount of work given to each thread.  Previously the
  workunit was (rows selected * width of tile), which could be small.  Now multiple
  column tiles can be combined into one workunit.  If the target is too small then
  thread and other overheads may reduce performance.  If target_workunit is too large
  then the size of the gathered sub matrix may take up too much memory.
- add configuration parameters (max_workers, cpu_multiplier, and  target_workunit)
2020-04-13 18:53:13 -07:00

108 lines
2.6 KiB
Python

import yaml
default_config = """
# cellxgene configuration
server:
verbose: false
debug: false
host: "127.0.0.1"
port : null
scripts : []
open_browser: false
about_legal_tos: null
about_legal_privacy: null
force_https: false
flask_secret_key: null
generate_cache_control_headers: false
server_timing_headers: false
presentation:
max_categories: 1000
multi_dataset:
dataroot: null
# The index page when in multi-dataset mode:
# false or null: this returns a 404 code
# true: loads a test index page, which links to the datasets that are available in the dataroot
# string/URL: redirect to this URL: flask.redirect(config.multi_dataset__index)
index: false
# A list of allowed matrix types. If an empty list, then all matrix types are allowed
allowed_matrix_types: []
matrix_cache:
# The maximum number of datasets that may be opened at one time. The least recently used dataset
# is evicted from the cache first.
max_datasets: 5
# A matrix is automatically removed from the cache after timelimit_s number of seconds.
# If timelimit_s is set to None, then there is no time limit.
timelimit_s: 30
single_dataset:
datapath: null
obs_names: null
var_names: null
about: null
title: null
user_annotations:
enable: true
type: local_file_csv
local_file_csv:
directory: null
file: null
ontology:
enable: false
obo_location: null
embeddings:
names : []
enable_reembedding: false
diffexp:
enable: true
lfc_cutoff: 0.01
top_n: 10
alg_cxg:
# The number of threads to use is computed from: min(max_workers, cpu_multipler * cpu_count).
# Where cpu_count is determined at runtime.
max_workers: 64
cpu_multiplier: 4
# The target number of matrix elements that are evaluated
# together in one thread.
target_workunit: 16_000_000
data_locator:
s3:
# s3 region name.
# if true, then the s3 location is automatically determined from the datapath or dataroot.
# if false/null, then do not set.
# if a string, then use that value (e.g. us-east-1).
region_name: true
adaptor:
cxg_adaptor:
# The key/values under tiledb_ctx will be used to initialize the tiledb Context.
# If 'vfs.s3.region' is not set, then it will automatically use the setting from
# data_locator / s3 / region_name.
tiledb_ctx:
sm.tile_cache_size: 8589934592
sm.num_reader_threads: 32
anndata_adaptor:
backed: false
limits:
column_request_max: 32
diffexp_cellcount_max: null
"""
def get_default_config():
return yaml.load(default_config, Loader=yaml.Loader)