prepare - work around anndata bug (#1260)

* work around anndata bug 344

* fix accidental cut and paste error

* Use modified make_index_unique function

Temporarily copy code from https://github.com/theislab/anndata/pull/345
until the issue is resolved and released.

* Add notes and test for make_index_unique

* Lint fix

* Format python

Co-authored-by: Matt Weiden <538456+mweiden@users.noreply.github.com>
This commit is contained in:
Bruce Martin
2020-03-22 12:27:59 -07:00
committed by GitHub
co-authored by Matt Weiden
parent db7a485796
commit d99b84ba09
7 changed files with 86 additions and 29 deletions
+5 -10
View File
@@ -138,7 +138,7 @@ def dataset_args(func):
"-t", "-t",
default=DEFAULT_CONFIG.single_dataset__title, default=DEFAULT_CONFIG.single_dataset__title,
metavar="<text>", metavar="<text>",
help="Title to display. If omitted will use file name." help="Title to display. If omitted will use file name.",
) )
@click.option( @click.option(
"--about", "--about",
@@ -300,7 +300,7 @@ def launch(
experimental_annotations_ontology_obo, experimental_annotations_ontology_obo,
experimental_enable_reembedding, experimental_enable_reembedding,
config_file, config_file,
dump_default_config dump_default_config,
): ):
"""Launch the cellxgene data viewer. """Launch the cellxgene data viewer.
This web app lets you explore single-cell expression data. This web app lets you explore single-cell expression data.
@@ -342,29 +342,22 @@ def launch(
server__port=port, server__port=port,
server__scripts=scripts, server__scripts=scripts,
server__open_browser=open_browser, server__open_browser=open_browser,
single_dataset__datapath=datapath, single_dataset__datapath=datapath,
single_dataset__title=title, single_dataset__title=title,
single_dataset__about=about, single_dataset__about=about,
single_dataset__obs_names=obs_names, single_dataset__obs_names=obs_names,
single_dataset__var_names=var_names, single_dataset__var_names=var_names,
multi_dataset__dataroot=dataroot, multi_dataset__dataroot=dataroot,
user_annotations__enable=not disable_annotations, user_annotations__enable=not disable_annotations,
user_annotations__local_file_csv__file=annotations_file, user_annotations__local_file_csv__file=annotations_file,
user_annotations__local_file_csv__directory=annotations_dir, user_annotations__local_file_csv__directory=annotations_dir,
user_annotations__ontology__enable=experimental_annotations_ontology, user_annotations__ontology__enable=experimental_annotations_ontology,
user_annotations__ontology__obo_location=experimental_annotations_ontology_obo, user_annotations__ontology__obo_location=experimental_annotations_ontology_obo,
presentation__max_categories=max_category_items, presentation__max_categories=max_category_items,
embeddings__names=embedding, embeddings__names=embedding,
embeddings__enable_reembedding=experimental_enable_reembedding, embeddings__enable_reembedding=experimental_enable_reembedding,
diffexp__enable=not disable_diffexp, diffexp__enable=not disable_diffexp,
diffexp__lfc_cutoff=diffexp_lfc_cutoff, diffexp__lfc_cutoff=diffexp_lfc_cutoff,
adaptor__anndata_adaptor__backed=backed, adaptor__anndata_adaptor__backed=backed,
) )
@@ -386,6 +379,7 @@ def launch(
# create the server # create the server
from server.app.app import Server from server.app.app import Server
server = Server(matrix_data_cache_manager, user_annotations, app_config) server = Server(matrix_data_cache_manager, user_annotations, app_config)
if not app_config.server__verbose: if not app_config.server__verbose:
@@ -411,7 +405,8 @@ def launch(
debug=app_config.server__debug, debug=app_config.server__debug,
port=app_config.server__port, port=app_config.server__port,
threaded=not app_config.server__debug, threaded=not app_config.server__debug,
use_debugger=False) use_debugger=False,
)
except OSError as e: except OSError as e:
if e.errno == errno.EADDRINUSE: if e.errno == errno.EADDRINUSE:
raise click.ClickException("Port is in use, please specify an open port using the --port flag.") from e raise click.ClickException("Port is in use, please specify an open port using the --port flag.") from e
+60 -10
View File
@@ -1,6 +1,7 @@
from os.path import expanduser, isdir, isfile, sep, splitext from os.path import expanduser, isdir, isfile, sep, splitext
import click import click
import pandas as pd
from numpy import ndarray, unique from numpy import ndarray, unique
from scipy.sparse.csc import csc_matrix from scipy.sparse.csc import csc_matrix
@@ -23,12 +24,7 @@ from server.common.utils import sort_options
show_default=True, show_default=True,
) )
@click.option( @click.option(
"--recipe", "--recipe", "-r", default="none", type=click.Choice(["none", "seurat", "zheng17"]), show_default=True,
"-r",
default="none",
type=click.Choice(["none", "seurat", "zheng17"]),
help="Preprocessing to run.",
show_default=True,
) )
@click.option("--output", "-o", default="", help="Save a new file to filename.", metavar="<filename>") @click.option("--output", "-o", default="", help="Save a new file to filename.", metavar="<filename>")
@click.option("--plotting", "-p", default=False, is_flag=True, help="Generate plots.", show_default=True) @click.option("--plotting", "-p", default=False, is_flag=True, help="Generate plots.", show_default=True)
@@ -44,10 +40,16 @@ from server.common.utils import sort_options
"(saved to adata.obs and adata.var; see scanpy.pp.calculate_qc_metrics for details).", "(saved to adata.obs and adata.var; see scanpy.pp.calculate_qc_metrics for details).",
) )
@click.option( @click.option(
"--make-obs-names-unique", default=True, is_flag=True, help="Ensure obs index is unique.", show_default=True "--make-obs-names-unique/--no-make-obs-names-unique",
default=True,
help="Ensure obs index is unique.",
show_default=True,
) )
@click.option( @click.option(
"--make-var-names-unique", default=True, is_flag=True, help="Ensure var index is unique.", show_default=True "--make-var-names-unique/--no-make-var-names-unique",
default=True,
help="Ensure var index is unique.",
show_default=True,
) )
@click.help_option("--help", "-h", help="Show this message and exit.") @click.help_option("--help", "-h", help="Show this message and exit.")
def prepare( def prepare(
@@ -129,9 +131,9 @@ def prepare(
raise click.UsageError(f"var {set_var_names} not found, options are: {adata.var_keys()}") raise click.UsageError(f"var {set_var_names} not found, options are: {adata.var_keys()}")
adata.var_names = adata.var[set_var_names] adata.var_names = adata.var[set_var_names]
if make_obs_names_unique: if make_obs_names_unique:
adata.obs_names_make_unique() adata.obs.index = make_index_unique(adata.obs.index)
if make_var_names_unique: if make_var_names_unique:
adata.var_names_make_unique() adata.var.index = make_index_unique(adata.var.index)
if not adata._obs.index.is_unique: if not adata._obs.index.is_unique:
click.echo("Warning: obs index is not unique") click.echo("Warning: obs index is not unique")
if not adata._var.index.is_unique: if not adata._var.index.is_unique:
@@ -221,3 +223,51 @@ def prepare(
adata.write(output) adata.write(output)
click.echo("[cellxgene] Success!") click.echo("[cellxgene] Success!")
# TODO (mweiden): remove this once this issue is resolved https://github.com/theislab/anndata/issues/344
# Note: tentative solution here https://github.com/theislab/anndata/pull/345
def make_index_unique(index: pd.Index, join: str = "-"):
"""
Makes the index unique by appending a number string to each duplicate index element: '1', '2', etc.
If a tentative name created by the algorithm already exists in the index, it tries the next integer in the sequence.
The first occurrence of a non-unique value is ignored.
Parameters
----------
join
The connecting string between name and integer.
Examples
--------
>>> from anndata import AnnData
>>> adata1 = AnnData(np.ones((3, 2)), dict(obs_names=['a', 'b', 'c']))
>>> adata2 = AnnData(np.zeros((3, 2)), dict(obs_names=['d', 'b', 'b']))
>>> adata = adata1.concatenate(adata2)
>>> adata.obs_names
Index(['a', 'b', 'c', 'd', 'b', 'b'], dtype='object')
>>> adata.obs_names_make_unique()
>>> adata.obs_names
Index(['a', 'b', 'c', 'd', 'b-1', 'b-2'], dtype='object')
"""
if index.is_unique:
return index
from collections import defaultdict
values = index.values
values_set = set(values)
indices_dup = index.duplicated(keep="first")
values_dup = values[indices_dup]
counter = defaultdict(lambda: 0)
for i, v in enumerate(values_dup):
while True:
counter[v] += 1
tentative_new_name = v + join + str(counter[v])
if tentative_new_name not in values_set:
values_set.add(tentative_new_name)
values_dup[i] = tentative_new_name
break
values[indices_dup] = values_dup
index = pd.Index(values)
return index
+1
View File
@@ -344,6 +344,7 @@ class AppConfig(object):
# cxg # cxg
self.__check_attr("adaptor__cxg_adaptor__tiledb_ctx", dict) self.__check_attr("adaptor__cxg_adaptor__tiledb_ctx", dict)
from server.data_cxg.cxg_adaptor import CxgAdaptor from server.data_cxg.cxg_adaptor import CxgAdaptor
CxgAdaptor.set_tiledb_context(self.adaptor__cxg_adaptor__tiledb_ctx) CxgAdaptor.set_tiledb_context(self.adaptor__cxg_adaptor__tiledb_ctx)
# anndata # anndata
+3 -5
View File
@@ -17,11 +17,9 @@ import threading
class CxgAdaptor(DataAdaptor): class CxgAdaptor(DataAdaptor):
# TODO: The tiledb context parameters should be a configuration option # TODO: The tiledb context parameters should be a configuration option
tiledb_ctx = tiledb.Ctx({ tiledb_ctx = tiledb.Ctx(
"sm.tile_cache_size": 8 * 1024 * 1024 * 1024, {"sm.tile_cache_size": 8 * 1024 * 1024 * 1024, "sm.num_reader_threads": 32, "vfs.s3.region": "us-east-1"}
"sm.num_reader_threads": 32, )
"vfs.s3.region": "us-east-1"
})
def __init__(self, data_locator, config=None): def __init__(self, data_locator, config=None):
super().__init__(config) super().__init__(config)
+1 -3
View File
@@ -38,9 +38,7 @@ try:
if dataroot: if dataroot:
logging.info(f"Configuration from CXG_DATAROOT") logging.info(f"Configuration from CXG_DATAROOT")
app_config.update( app_config.update(multi_dataset__dataroot=dataroot,)
multi_dataset__dataroot=dataroot,
)
# features are unsupported in the current hosted server # features are unsupported in the current hosted server
app_config.update( app_config.update(
+1 -1
View File
@@ -40,7 +40,7 @@ class AdaptorTest(unittest.TestCase):
"single_dataset__var_names": None, "single_dataset__var_names": None,
"diffexp__lfc_cutoff": 0.01, "diffexp__lfc_cutoff": 0.01,
"adaptor__anndata_adaptor__backed": self.backed, "adaptor__anndata_adaptor__backed": self.backed,
"single_dataset__datapath" : self.data_locator "single_dataset__datapath": self.data_locator,
} }
config = AppConfig() config = AppConfig()
config.update(**args) config.update(**args)
+15
View File
@@ -0,0 +1,15 @@
import unittest
import pandas as pd
from server.cli.prepare import make_index_unique
class CLIPrepareTests(unittest.TestCase):
""" Test cases for CLI prepare logic """
def test_make_index_unique(self):
index = pd.Index(["SNORD113", "SNORD113", "SNORD113-1"])
result = make_index_unique(index)
expected = pd.Index(["SNORD113", "SNORD113-2", "SNORD113-1"])
self.assertTrue(all(left == right for left, right in zip(result.values, expected.values)))