fix for incorrect stats computation in diff exp t-test (#2318)

* 2211 fixes

* lint

* lint

* add missing test and bug found by test

* change terminology for count distribution

* update scanpy requirement

* update scanpy requirement
This commit is contained in:
Bruce Martin
2021-07-23 11:36:26 -07:00
committed by GitHub
parent 1ebde2213d
commit 1ea2b7fe80
28 changed files with 336 additions and 90 deletions
@@ -22,19 +22,27 @@ Test the anndata adaptor using the pbmc3k data set.
@parameterized_class(
("data_locator", "backed"),
("data_locator", "backed", "X_approx_distribution"),
[
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", False),
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", False),
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", False),
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", True),
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", True),
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", True),
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", False, "auto"),
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", False, "auto"),
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", False, "auto"),
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", True, "auto"),
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", True, "auto"),
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", True, "auto"),
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", False, "normal"),
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", False, "normal"),
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", False, "normal"),
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", True, "normal"),
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", True, "normal"),
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", True, "normal"),
],
)
class AdaptorTest(unittest.TestCase):
def setUp(self):
config = app_config(self.data_locator, self.backed)
config = app_config(
self.data_locator, self.backed, extra_dataset_config=dict(X_approx_distribution=self.X_approx_distribution)
)
self.data = AnndataAdaptor(DataLocator(self.data_locator), config)
def test_init(self):
@@ -90,7 +98,8 @@ class AdaptorTest(unittest.TestCase):
def test_schema_produces_error(self):
self.data.data.obs["time"] = pd.Series(
list([time.time() for i in range(self.data.cell_count)]), dtype="datetime64[ns]",
list([time.time() for i in range(self.data.cell_count)]),
dtype="datetime64[ns]",
)
with pytest.raises(TypeError):
self.data._create_schema()
@@ -107,7 +116,7 @@ class AdaptorTest(unittest.TestCase):
self.assertTrue((Y >= 0).all() and (Y <= 1).all())
def test_layout_fields(self):
""" X_pca, X_tsne, X_umap are available """
"""X_pca, X_tsne, X_umap are available"""
fbs = self.data.layout_to_fbs_matrix(["pca"])
layout = decode_fbs.decode_matrix_FBS(fbs)
self.assertEqual(layout["n_cols"], 2)
@@ -127,7 +136,8 @@ class AdaptorTest(unittest.TestCase):
self.assertEqual(annotations["n_cols"], 5)
obs_index_col_name = self.data.get_schema()["annotations"]["obs"]["index"]
self.assertEqual(
annotations["col_idx"], [obs_index_col_name, "n_genes", "percent_mito", "n_counts", "louvain"],
annotations["col_idx"],
[obs_index_col_name, "n_genes", "percent_mito", "n_counts", "louvain"],
)
fbs = self.data.annotation_to_fbs_matrix("var")
@@ -153,12 +163,12 @@ class AdaptorTest(unittest.TestCase):
f1 = {"filter": {"obs": {"index": [[0, 500]]}}}
f2 = {"filter": {"obs": {"index": [[500, 1000]]}}}
result = json.loads(self.data.diffexp_topN(f1["filter"], f2["filter"]))
self.assertEqual(len(result['positive']), 10)
self.assertEqual(len(result['negative']), 10)
self.assertEqual(len(result["positive"]), 10)
self.assertEqual(len(result["negative"]), 10)
result = json.loads(self.data.diffexp_topN(f1["filter"], f2["filter"], 20))
self.assertEqual(len(result['positive']), 20)
self.assertEqual(len(result['negative']), 20)
self.assertEqual(len(result["positive"]), 20)
self.assertEqual(len(result["negative"]), 20)
def test_data_frame(self):
f1 = {"var": {"index": [[0, 10]]}}