mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-08 01:48:11 +08:00
prepare - work around anndata bug (#1260)
* work around anndata bug 344 * fix accidental cut and paste error * Use modified make_index_unique function Temporarily copy code from https://github.com/theislab/anndata/pull/345 until the issue is resolved and released. * Add notes and test for make_index_unique * Lint fix * Format python Co-authored-by: Matt Weiden <538456+mweiden@users.noreply.github.com>
This commit is contained in:
co-authored by
Matt Weiden
parent
db7a485796
commit
d99b84ba09
+5
-10
@@ -138,7 +138,7 @@ def dataset_args(func):
|
|||||||
"-t",
|
"-t",
|
||||||
default=DEFAULT_CONFIG.single_dataset__title,
|
default=DEFAULT_CONFIG.single_dataset__title,
|
||||||
metavar="<text>",
|
metavar="<text>",
|
||||||
help="Title to display. If omitted will use file name."
|
help="Title to display. If omitted will use file name.",
|
||||||
)
|
)
|
||||||
@click.option(
|
@click.option(
|
||||||
"--about",
|
"--about",
|
||||||
@@ -300,7 +300,7 @@ def launch(
|
|||||||
experimental_annotations_ontology_obo,
|
experimental_annotations_ontology_obo,
|
||||||
experimental_enable_reembedding,
|
experimental_enable_reembedding,
|
||||||
config_file,
|
config_file,
|
||||||
dump_default_config
|
dump_default_config,
|
||||||
):
|
):
|
||||||
"""Launch the cellxgene data viewer.
|
"""Launch the cellxgene data viewer.
|
||||||
This web app lets you explore single-cell expression data.
|
This web app lets you explore single-cell expression data.
|
||||||
@@ -342,29 +342,22 @@ def launch(
|
|||||||
server__port=port,
|
server__port=port,
|
||||||
server__scripts=scripts,
|
server__scripts=scripts,
|
||||||
server__open_browser=open_browser,
|
server__open_browser=open_browser,
|
||||||
|
|
||||||
single_dataset__datapath=datapath,
|
single_dataset__datapath=datapath,
|
||||||
single_dataset__title=title,
|
single_dataset__title=title,
|
||||||
single_dataset__about=about,
|
single_dataset__about=about,
|
||||||
single_dataset__obs_names=obs_names,
|
single_dataset__obs_names=obs_names,
|
||||||
single_dataset__var_names=var_names,
|
single_dataset__var_names=var_names,
|
||||||
|
|
||||||
multi_dataset__dataroot=dataroot,
|
multi_dataset__dataroot=dataroot,
|
||||||
|
|
||||||
user_annotations__enable=not disable_annotations,
|
user_annotations__enable=not disable_annotations,
|
||||||
user_annotations__local_file_csv__file=annotations_file,
|
user_annotations__local_file_csv__file=annotations_file,
|
||||||
user_annotations__local_file_csv__directory=annotations_dir,
|
user_annotations__local_file_csv__directory=annotations_dir,
|
||||||
user_annotations__ontology__enable=experimental_annotations_ontology,
|
user_annotations__ontology__enable=experimental_annotations_ontology,
|
||||||
user_annotations__ontology__obo_location=experimental_annotations_ontology_obo,
|
user_annotations__ontology__obo_location=experimental_annotations_ontology_obo,
|
||||||
|
|
||||||
presentation__max_categories=max_category_items,
|
presentation__max_categories=max_category_items,
|
||||||
|
|
||||||
embeddings__names=embedding,
|
embeddings__names=embedding,
|
||||||
embeddings__enable_reembedding=experimental_enable_reembedding,
|
embeddings__enable_reembedding=experimental_enable_reembedding,
|
||||||
|
|
||||||
diffexp__enable=not disable_diffexp,
|
diffexp__enable=not disable_diffexp,
|
||||||
diffexp__lfc_cutoff=diffexp_lfc_cutoff,
|
diffexp__lfc_cutoff=diffexp_lfc_cutoff,
|
||||||
|
|
||||||
adaptor__anndata_adaptor__backed=backed,
|
adaptor__anndata_adaptor__backed=backed,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -386,6 +379,7 @@ def launch(
|
|||||||
|
|
||||||
# create the server
|
# create the server
|
||||||
from server.app.app import Server
|
from server.app.app import Server
|
||||||
|
|
||||||
server = Server(matrix_data_cache_manager, user_annotations, app_config)
|
server = Server(matrix_data_cache_manager, user_annotations, app_config)
|
||||||
|
|
||||||
if not app_config.server__verbose:
|
if not app_config.server__verbose:
|
||||||
@@ -411,7 +405,8 @@ def launch(
|
|||||||
debug=app_config.server__debug,
|
debug=app_config.server__debug,
|
||||||
port=app_config.server__port,
|
port=app_config.server__port,
|
||||||
threaded=not app_config.server__debug,
|
threaded=not app_config.server__debug,
|
||||||
use_debugger=False)
|
use_debugger=False,
|
||||||
|
)
|
||||||
except OSError as e:
|
except OSError as e:
|
||||||
if e.errno == errno.EADDRINUSE:
|
if e.errno == errno.EADDRINUSE:
|
||||||
raise click.ClickException("Port is in use, please specify an open port using the --port flag.") from e
|
raise click.ClickException("Port is in use, please specify an open port using the --port flag.") from e
|
||||||
|
|||||||
+60
-10
@@ -1,6 +1,7 @@
|
|||||||
from os.path import expanduser, isdir, isfile, sep, splitext
|
from os.path import expanduser, isdir, isfile, sep, splitext
|
||||||
|
|
||||||
import click
|
import click
|
||||||
|
import pandas as pd
|
||||||
from numpy import ndarray, unique
|
from numpy import ndarray, unique
|
||||||
from scipy.sparse.csc import csc_matrix
|
from scipy.sparse.csc import csc_matrix
|
||||||
|
|
||||||
@@ -23,12 +24,7 @@ from server.common.utils import sort_options
|
|||||||
show_default=True,
|
show_default=True,
|
||||||
)
|
)
|
||||||
@click.option(
|
@click.option(
|
||||||
"--recipe",
|
"--recipe", "-r", default="none", type=click.Choice(["none", "seurat", "zheng17"]), show_default=True,
|
||||||
"-r",
|
|
||||||
default="none",
|
|
||||||
type=click.Choice(["none", "seurat", "zheng17"]),
|
|
||||||
help="Preprocessing to run.",
|
|
||||||
show_default=True,
|
|
||||||
)
|
)
|
||||||
@click.option("--output", "-o", default="", help="Save a new file to filename.", metavar="<filename>")
|
@click.option("--output", "-o", default="", help="Save a new file to filename.", metavar="<filename>")
|
||||||
@click.option("--plotting", "-p", default=False, is_flag=True, help="Generate plots.", show_default=True)
|
@click.option("--plotting", "-p", default=False, is_flag=True, help="Generate plots.", show_default=True)
|
||||||
@@ -44,10 +40,16 @@ from server.common.utils import sort_options
|
|||||||
"(saved to adata.obs and adata.var; see scanpy.pp.calculate_qc_metrics for details).",
|
"(saved to adata.obs and adata.var; see scanpy.pp.calculate_qc_metrics for details).",
|
||||||
)
|
)
|
||||||
@click.option(
|
@click.option(
|
||||||
"--make-obs-names-unique", default=True, is_flag=True, help="Ensure obs index is unique.", show_default=True
|
"--make-obs-names-unique/--no-make-obs-names-unique",
|
||||||
|
default=True,
|
||||||
|
help="Ensure obs index is unique.",
|
||||||
|
show_default=True,
|
||||||
)
|
)
|
||||||
@click.option(
|
@click.option(
|
||||||
"--make-var-names-unique", default=True, is_flag=True, help="Ensure var index is unique.", show_default=True
|
"--make-var-names-unique/--no-make-var-names-unique",
|
||||||
|
default=True,
|
||||||
|
help="Ensure var index is unique.",
|
||||||
|
show_default=True,
|
||||||
)
|
)
|
||||||
@click.help_option("--help", "-h", help="Show this message and exit.")
|
@click.help_option("--help", "-h", help="Show this message and exit.")
|
||||||
def prepare(
|
def prepare(
|
||||||
@@ -129,9 +131,9 @@ def prepare(
|
|||||||
raise click.UsageError(f"var {set_var_names} not found, options are: {adata.var_keys()}")
|
raise click.UsageError(f"var {set_var_names} not found, options are: {adata.var_keys()}")
|
||||||
adata.var_names = adata.var[set_var_names]
|
adata.var_names = adata.var[set_var_names]
|
||||||
if make_obs_names_unique:
|
if make_obs_names_unique:
|
||||||
adata.obs_names_make_unique()
|
adata.obs.index = make_index_unique(adata.obs.index)
|
||||||
if make_var_names_unique:
|
if make_var_names_unique:
|
||||||
adata.var_names_make_unique()
|
adata.var.index = make_index_unique(adata.var.index)
|
||||||
if not adata._obs.index.is_unique:
|
if not adata._obs.index.is_unique:
|
||||||
click.echo("Warning: obs index is not unique")
|
click.echo("Warning: obs index is not unique")
|
||||||
if not adata._var.index.is_unique:
|
if not adata._var.index.is_unique:
|
||||||
@@ -221,3 +223,51 @@ def prepare(
|
|||||||
adata.write(output)
|
adata.write(output)
|
||||||
|
|
||||||
click.echo("[cellxgene] Success!")
|
click.echo("[cellxgene] Success!")
|
||||||
|
|
||||||
|
|
||||||
|
# TODO (mweiden): remove this once this issue is resolved https://github.com/theislab/anndata/issues/344
|
||||||
|
# Note: tentative solution here https://github.com/theislab/anndata/pull/345
|
||||||
|
def make_index_unique(index: pd.Index, join: str = "-"):
|
||||||
|
"""
|
||||||
|
Makes the index unique by appending a number string to each duplicate index element: '1', '2', etc.
|
||||||
|
|
||||||
|
If a tentative name created by the algorithm already exists in the index, it tries the next integer in the sequence.
|
||||||
|
|
||||||
|
The first occurrence of a non-unique value is ignored.
|
||||||
|
Parameters
|
||||||
|
----------
|
||||||
|
join
|
||||||
|
The connecting string between name and integer.
|
||||||
|
Examples
|
||||||
|
--------
|
||||||
|
>>> from anndata import AnnData
|
||||||
|
>>> adata1 = AnnData(np.ones((3, 2)), dict(obs_names=['a', 'b', 'c']))
|
||||||
|
>>> adata2 = AnnData(np.zeros((3, 2)), dict(obs_names=['d', 'b', 'b']))
|
||||||
|
>>> adata = adata1.concatenate(adata2)
|
||||||
|
>>> adata.obs_names
|
||||||
|
Index(['a', 'b', 'c', 'd', 'b', 'b'], dtype='object')
|
||||||
|
>>> adata.obs_names_make_unique()
|
||||||
|
>>> adata.obs_names
|
||||||
|
Index(['a', 'b', 'c', 'd', 'b-1', 'b-2'], dtype='object')
|
||||||
|
"""
|
||||||
|
if index.is_unique:
|
||||||
|
return index
|
||||||
|
from collections import defaultdict
|
||||||
|
|
||||||
|
values = index.values
|
||||||
|
values_set = set(values)
|
||||||
|
indices_dup = index.duplicated(keep="first")
|
||||||
|
values_dup = values[indices_dup]
|
||||||
|
counter = defaultdict(lambda: 0)
|
||||||
|
for i, v in enumerate(values_dup):
|
||||||
|
while True:
|
||||||
|
counter[v] += 1
|
||||||
|
tentative_new_name = v + join + str(counter[v])
|
||||||
|
if tentative_new_name not in values_set:
|
||||||
|
values_set.add(tentative_new_name)
|
||||||
|
values_dup[i] = tentative_new_name
|
||||||
|
break
|
||||||
|
|
||||||
|
values[indices_dup] = values_dup
|
||||||
|
index = pd.Index(values)
|
||||||
|
return index
|
||||||
|
|||||||
@@ -344,6 +344,7 @@ class AppConfig(object):
|
|||||||
# cxg
|
# cxg
|
||||||
self.__check_attr("adaptor__cxg_adaptor__tiledb_ctx", dict)
|
self.__check_attr("adaptor__cxg_adaptor__tiledb_ctx", dict)
|
||||||
from server.data_cxg.cxg_adaptor import CxgAdaptor
|
from server.data_cxg.cxg_adaptor import CxgAdaptor
|
||||||
|
|
||||||
CxgAdaptor.set_tiledb_context(self.adaptor__cxg_adaptor__tiledb_ctx)
|
CxgAdaptor.set_tiledb_context(self.adaptor__cxg_adaptor__tiledb_ctx)
|
||||||
|
|
||||||
# anndata
|
# anndata
|
||||||
|
|||||||
@@ -17,11 +17,9 @@ import threading
|
|||||||
class CxgAdaptor(DataAdaptor):
|
class CxgAdaptor(DataAdaptor):
|
||||||
|
|
||||||
# TODO: The tiledb context parameters should be a configuration option
|
# TODO: The tiledb context parameters should be a configuration option
|
||||||
tiledb_ctx = tiledb.Ctx({
|
tiledb_ctx = tiledb.Ctx(
|
||||||
"sm.tile_cache_size": 8 * 1024 * 1024 * 1024,
|
{"sm.tile_cache_size": 8 * 1024 * 1024 * 1024, "sm.num_reader_threads": 32, "vfs.s3.region": "us-east-1"}
|
||||||
"sm.num_reader_threads": 32,
|
)
|
||||||
"vfs.s3.region": "us-east-1"
|
|
||||||
})
|
|
||||||
|
|
||||||
def __init__(self, data_locator, config=None):
|
def __init__(self, data_locator, config=None):
|
||||||
super().__init__(config)
|
super().__init__(config)
|
||||||
|
|||||||
+1
-3
@@ -38,9 +38,7 @@ try:
|
|||||||
|
|
||||||
if dataroot:
|
if dataroot:
|
||||||
logging.info(f"Configuration from CXG_DATAROOT")
|
logging.info(f"Configuration from CXG_DATAROOT")
|
||||||
app_config.update(
|
app_config.update(multi_dataset__dataroot=dataroot,)
|
||||||
multi_dataset__dataroot=dataroot,
|
|
||||||
)
|
|
||||||
|
|
||||||
# features are unsupported in the current hosted server
|
# features are unsupported in the current hosted server
|
||||||
app_config.update(
|
app_config.update(
|
||||||
|
|||||||
@@ -40,7 +40,7 @@ class AdaptorTest(unittest.TestCase):
|
|||||||
"single_dataset__var_names": None,
|
"single_dataset__var_names": None,
|
||||||
"diffexp__lfc_cutoff": 0.01,
|
"diffexp__lfc_cutoff": 0.01,
|
||||||
"adaptor__anndata_adaptor__backed": self.backed,
|
"adaptor__anndata_adaptor__backed": self.backed,
|
||||||
"single_dataset__datapath" : self.data_locator
|
"single_dataset__datapath": self.data_locator,
|
||||||
}
|
}
|
||||||
config = AppConfig()
|
config = AppConfig()
|
||||||
config.update(**args)
|
config.update(**args)
|
||||||
|
|||||||
@@ -0,0 +1,15 @@
|
|||||||
|
import unittest
|
||||||
|
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
from server.cli.prepare import make_index_unique
|
||||||
|
|
||||||
|
|
||||||
|
class CLIPrepareTests(unittest.TestCase):
|
||||||
|
""" Test cases for CLI prepare logic """
|
||||||
|
|
||||||
|
def test_make_index_unique(self):
|
||||||
|
index = pd.Index(["SNORD113", "SNORD113", "SNORD113-1"])
|
||||||
|
result = make_index_unique(index)
|
||||||
|
expected = pd.Index(["SNORD113", "SNORD113-2", "SNORD113-1"])
|
||||||
|
self.assertTrue(all(left == right for left, right in zip(result.values, expected.values)))
|
||||||
Reference in New Issue
Block a user