mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-16 21:37:59 +08:00
190 lines
9.2 KiB
Python
190 lines
9.2 KiB
Python
import os
|
|
from os.path import splitext, isdir
|
|
|
|
from server.common.annotations.local_file_csv import AnnotationsLocalFile
|
|
from server.common.config.base_config import BaseConfig
|
|
from server.common.errors import ConfigurationError, AnnotationsError
|
|
from server.common.utils.data_locator import DataLocator
|
|
from server.data_common.matrix_loader import MatrixDataLoader
|
|
|
|
|
|
class DatasetConfig(BaseConfig):
|
|
"""Manages the config attribute associated with a dataset."""
|
|
|
|
def __init__(self, tag, app_config, default_config):
|
|
super().__init__(app_config, default_config)
|
|
self.tag = tag
|
|
try:
|
|
self.app__scripts = default_config["app"]["scripts"]
|
|
self.app__inline_scripts = default_config["app"]["inline_scripts"]
|
|
|
|
self.presentation__max_categories = default_config["presentation"]["max_categories"]
|
|
self.presentation__custom_colors = default_config["presentation"]["custom_colors"]
|
|
|
|
self.user_annotations__enable = default_config["user_annotations"]["enable"]
|
|
self.user_annotations__type = default_config["user_annotations"]["type"]
|
|
self.user_annotations__local_file_csv__directory = default_config["user_annotations"]["local_file_csv"][
|
|
"directory"
|
|
]
|
|
self.user_annotations__local_file_csv__file = default_config["user_annotations"]["local_file_csv"]["file"]
|
|
self.user_annotations__gene_sets__readonly = default_config["user_annotations"]["gene_sets"]["readonly"]
|
|
self.user_annotations__local_file_csv__gene_sets_file = default_config["user_annotations"][
|
|
"local_file_csv"
|
|
]["gene_sets_file"]
|
|
|
|
self.embeddings__names = default_config["embeddings"]["names"]
|
|
|
|
self.diffexp__enable = default_config["diffexp"]["enable"]
|
|
self.diffexp__lfc_cutoff = default_config["diffexp"]["lfc_cutoff"]
|
|
self.diffexp__top_n = default_config["diffexp"]["top_n"]
|
|
|
|
self.X_approximate_distribution = default_config["X_approximate_distribution"]
|
|
|
|
except KeyError as e:
|
|
raise ConfigurationError(f"Unexpected config: {str(e)}")
|
|
|
|
# The annotation object is created during complete_config and stored here.
|
|
self.user_annotations = None
|
|
|
|
def complete_config(self, context):
|
|
self.handle_app()
|
|
self.handle_presentation()
|
|
self.handle_user_annotations(context)
|
|
self.handle_embeddings()
|
|
self.handle_diffexp(context)
|
|
self.handle_X_approximate_distribution()
|
|
|
|
def get_data_adaptor(self):
|
|
server_config = self.app_config.server_config
|
|
if not server_config.data_adaptor:
|
|
matrix_data_loader = MatrixDataLoader(server_config.single_dataset__datapath, app_config=self.app_config)
|
|
server_config.data_adaptor = matrix_data_loader.open(self.app_config)
|
|
|
|
return server_config.data_adaptor
|
|
|
|
def handle_app(self):
|
|
self.validate_correct_type_of_configuration_attribute("app__scripts", list)
|
|
self.validate_correct_type_of_configuration_attribute("app__inline_scripts", list)
|
|
|
|
# scripts can be string (filename) or dict (attributes). Convert string to dict.
|
|
scripts = []
|
|
for script in self.app__scripts:
|
|
try:
|
|
if isinstance(script, str):
|
|
scripts.append({"src": script})
|
|
elif isinstance(script, dict) and isinstance(script["src"], str):
|
|
scripts.append(script)
|
|
else:
|
|
raise Exception
|
|
except Exception as e:
|
|
raise ConfigurationError(f"Scripts must be string or a dict containing an src key: {e}")
|
|
|
|
self.app__scripts = scripts
|
|
|
|
def handle_presentation(self):
|
|
self.validate_correct_type_of_configuration_attribute("presentation__max_categories", int)
|
|
self.validate_correct_type_of_configuration_attribute("presentation__custom_colors", bool)
|
|
|
|
def handle_user_annotations(self, context):
|
|
self.validate_correct_type_of_configuration_attribute("user_annotations__enable", bool)
|
|
self.validate_correct_type_of_configuration_attribute("user_annotations__type", str)
|
|
self.validate_correct_type_of_configuration_attribute(
|
|
"user_annotations__local_file_csv__directory", (type(None), str)
|
|
)
|
|
self.validate_correct_type_of_configuration_attribute(
|
|
"user_annotations__local_file_csv__file", (type(None), str)
|
|
)
|
|
self.validate_correct_type_of_configuration_attribute(
|
|
"user_annotations__local_file_csv__gene_sets_file", (type(None), str)
|
|
)
|
|
self.validate_correct_type_of_configuration_attribute("user_annotations__gene_sets__readonly", bool)
|
|
|
|
# Must always have an annotations instance to support genesets. User annotation (cell labels) are optional
|
|
# as are writable gene sets
|
|
if self.user_annotations__type == "local_file_csv":
|
|
self.handle_local_file_csv_annotations(context)
|
|
else:
|
|
raise ConfigurationError('The only annotation type support is "local_file_csv"')
|
|
|
|
self.check_annotation_config_vars_not_set(context)
|
|
|
|
def handle_local_file_csv_annotations(self, context):
|
|
dirname = self.user_annotations__local_file_csv__directory
|
|
filename = self.user_annotations__local_file_csv__file
|
|
genesets_filename = self.user_annotations__local_file_csv__gene_sets_file
|
|
|
|
if dirname is not None and (filename is not None or genesets_filename is not None):
|
|
raise ConfigurationError(
|
|
"'user-generated-data-dir' may not be used with 'annotations-file' or 'gene-sets-file'."
|
|
)
|
|
|
|
if filename is not None:
|
|
lf_name, lf_ext = splitext(filename)
|
|
if lf_ext and lf_ext != ".csv":
|
|
raise ConfigurationError(f"annotation file type must be .csv: {filename}")
|
|
|
|
if genesets_filename is not None:
|
|
lf_name, lf_ext = splitext(genesets_filename)
|
|
if lf_ext and lf_ext != ".csv":
|
|
raise ConfigurationError(f"genesets file type must be .csv: {genesets_filename}")
|
|
|
|
if dirname is not None:
|
|
if not DataLocator(dirname).islocal():
|
|
# remote object stores only support objects but not directories, do nothing
|
|
pass
|
|
elif not isdir(dirname):
|
|
try:
|
|
os.mkdir(dirname)
|
|
except OSError:
|
|
raise ConfigurationError("Unable to create directory specified by --user-generated-data-dir")
|
|
|
|
anno_config = {
|
|
"user-annotations": self.user_annotations__enable,
|
|
"genesets-save": not self.user_annotations__gene_sets__readonly,
|
|
}
|
|
self.user_annotations = AnnotationsLocalFile(anno_config, dirname, filename, genesets_filename)
|
|
|
|
# if the user has specified a fixed label file, go ahead and validate it
|
|
# so that we can remove errors early in the process.
|
|
server_config = self.app_config.server_config
|
|
if server_config.single_dataset__datapath:
|
|
data_adaptor = self.get_data_adaptor()
|
|
if self.user_annotations__local_file_csv__file:
|
|
self.user_annotations.read_labels(data_adaptor)
|
|
if self.user_annotations__local_file_csv__gene_sets_file:
|
|
try:
|
|
self.user_annotations.read_gene_sets(data_adaptor, context)
|
|
except (ValueError, AnnotationsError, KeyError) as e:
|
|
raise ConfigurationError(f"Unable to read genesets CSV file: {str(e)}") from e
|
|
|
|
def check_annotation_config_vars_not_set(self, context):
|
|
if self.user_annotations__type is not None:
|
|
dirname = self.user_annotations__local_file_csv__directory
|
|
filename = self.user_annotations__local_file_csv__file
|
|
if not self.user_annotations__enable:
|
|
if filename is not None:
|
|
context["messagefn"]("Warning: --annotations-file ignored as annotations are disabled.")
|
|
if dirname is not None:
|
|
context["messagefn"]("Warning: --user-generated-data-dir ignored as annotations are disabled.")
|
|
|
|
def handle_embeddings(self):
|
|
self.validate_correct_type_of_configuration_attribute("embeddings__names", list)
|
|
|
|
def handle_diffexp(self, context):
|
|
self.validate_correct_type_of_configuration_attribute("diffexp__enable", bool)
|
|
self.validate_correct_type_of_configuration_attribute("diffexp__lfc_cutoff", float)
|
|
self.validate_correct_type_of_configuration_attribute("diffexp__top_n", int)
|
|
|
|
data_adaptor = self.get_data_adaptor()
|
|
if self.diffexp__enable and data_adaptor.parameters.get("diffexp-may-be-slow", False):
|
|
context["messagefn"](
|
|
"CAUTION: due to the size of your dataset, " "running differential expression may take longer or fail."
|
|
)
|
|
|
|
def handle_X_approximate_distribution(self):
|
|
self.validate_correct_type_of_configuration_attribute("X_approximate_distribution", str)
|
|
if self.X_approximate_distribution not in ["auto", "normal", "count"]:
|
|
raise ConfigurationError(
|
|
"X_approximate_distribution has unknown value -- must be 'auto', 'normal' or 'count'."
|
|
)
|