mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-29 09:48:12 +08:00
fix for incorrect stats computation in diff exp t-test (#2318)
* 2211 fixes * lint * lint * add missing test and bug found by test * change terminology for count distribution * update scanpy requirement * update scanpy requirement
This commit is contained in:
@@ -103,6 +103,7 @@ class ConfigTests(unittest.TestCase):
|
||||
environment=None,
|
||||
aws_secrets_manager_region=None,
|
||||
aws_secrets_manager_secrets=[],
|
||||
X_approx_distribution="auto",
|
||||
config_file_name="app_config.yml",
|
||||
):
|
||||
random_num = random.randrange(999999)
|
||||
@@ -150,6 +151,7 @@ class ConfigTests(unittest.TestCase):
|
||||
enable_difexp=enable_difexp,
|
||||
lfc_cutoff=lfc_cutoff,
|
||||
top_n=top_n,
|
||||
X_approx_distribution=X_approx_distribution,
|
||||
config_file_name=f"temp_dataset_config_{random_num}.yml",
|
||||
)
|
||||
external_config = self.custom_external_config(
|
||||
@@ -185,6 +187,7 @@ class ConfigTests(unittest.TestCase):
|
||||
enable_difexp="true",
|
||||
lfc_cutoff=0.01,
|
||||
top_n=10,
|
||||
X_approx_distribution="auto",
|
||||
config_file_name="dataset_config.yml",
|
||||
):
|
||||
configfile = os.path.join(self.tmp_fixtures_directory, config_file_name)
|
||||
|
||||
@@ -7,7 +7,7 @@ from unittest.mock import patch
|
||||
from backend.server.common.annotations.local_file_csv import AnnotationsLocalFile
|
||||
from backend.server.common.config.app_config import AppConfig
|
||||
from backend.server.common.config.base_config import BaseConfig
|
||||
from backend.test import FIXTURES_ROOT, H5AD_FIXTURE
|
||||
from backend.test import H5AD_FIXTURE
|
||||
|
||||
from backend.common.errors import ConfigurationError
|
||||
from backend.test.test_server.unit.common.config import ConfigTests
|
||||
@@ -46,7 +46,7 @@ class TestDatasetConfig(ConfigTests):
|
||||
mock_check_attrs.side_effect = BaseConfig.validate_correct_type_of_configuration_attribute()
|
||||
self.dataset_config.complete_config(self.context)
|
||||
self.assertIsNotNone(self.config.server_config.data_adaptor)
|
||||
self.assertEqual(mock_check_attrs.call_count, 16)
|
||||
self.assertEqual(mock_check_attrs.call_count, 17)
|
||||
|
||||
def test_app_sets_script_vars(self):
|
||||
config = self.get_config(scripts=["path/to/script"])
|
||||
|
||||
@@ -38,28 +38,28 @@ class DiffExpTest(unittest.TestCase):
|
||||
"""Checks the results for a specific set of rows selections"""
|
||||
|
||||
positive_expects = [
|
||||
[1712, -0.5525154, 0.0051788902660723345, 1.0],
|
||||
[1575, 1.0317602, 0.007830310753043345, 1.0],
|
||||
[693, 0.4703904, 0.008715846769131548, 1.0],
|
||||
[916, 0.9567287, 0.009080596532247588, 1.0],
|
||||
[77, 0.02665649, 0.010070392939027756, 1.0],
|
||||
[782, -1.0981874, 0.010161745218916036, 1.0],
|
||||
[913, 0.5683986, 0.010782030711612685, 1.0],
|
||||
[910, 0.83164597, 0.014596411069229197, 1.0],
|
||||
[1727, 0.4127781, 0.015168372104237176, 1.0],
|
||||
[1443, -0.8241895, 0.015337080567465522, 1.0]
|
||||
[1712, 0.24104056, 0.0051788902660723345, 1.0],
|
||||
[1575, 0.2615018, 0.007830310753043345, 1.0],
|
||||
[693, 0.23106655, 0.008715846769131548, 1.0],
|
||||
[916, 0.2395215, 0.009080596532247588, 1.0],
|
||||
[77, 0.22927025, 0.010070392939027756, 1.0],
|
||||
[782, 0.20581803, 0.010161745218916036, 1.0],
|
||||
[913, 0.23841085, 0.010782030711612685, 1.0],
|
||||
[910, 0.21493295, 0.014596411069229197, 1.0],
|
||||
[1727, 0.21911663, 0.015168372104237176, 1.0],
|
||||
[1443, 0.19814226, 0.015337080567465522, 1.0],
|
||||
]
|
||||
negative_expects = [
|
||||
[956, 0.016060986, 0.0008649321884808977, 1.0],
|
||||
[1124, 0.96602094, 0.0011717216548271284, 1.0],
|
||||
[1809, 1.1110606, 0.0019304405196777848, 1.0],
|
||||
[1754, 0.5201581, 0.005691734062127954, 1.0],
|
||||
[948, 1.6390722, 0.006622111055981219, 1.0],
|
||||
[1810, 0.78618884, 0.007055917428377063, 1.0],
|
||||
[779, 1.5241305, 0.007202934422407284, 1.0],
|
||||
[576, 0.97873515, 0.008272092578813124, 1.0],
|
||||
[538, 0.89114505, 0.01062259019889307, 1.0],
|
||||
[436, 0.3119122, 0.01127515110543434, 1.0]
|
||||
[956, -0.29662406, 0.0008649321884808977, 1.0],
|
||||
[1124, -0.2607333, 0.0011717216548271284, 1.0],
|
||||
[1809, -0.24854594, 0.0019304405196777848, 1.0],
|
||||
[1754, -0.24683577, 0.005691734062127954, 1.0],
|
||||
[948, -0.18708363, 0.006622111055981219, 1.0],
|
||||
[1810, -0.2172082, 0.007055917428377063, 1.0],
|
||||
[779, -0.21150622, 0.007202934422407284, 1.0],
|
||||
[576, -0.19008157, 0.008272092578813124, 1.0],
|
||||
[538, -0.21803819, 0.01062259019889307, 1.0],
|
||||
[436, -0.2100364, 0.01127515110543434, 1.0],
|
||||
]
|
||||
|
||||
self.compare_diffexp_results(results["positive"], positive_expects)
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
import unittest
|
||||
import numpy as np
|
||||
from scipy import sparse
|
||||
from backend.common.compute.estimate_distribution import estimate_approximate_distribution
|
||||
from backend.common.constants import XApproxDistribution
|
||||
from backend.server.data_common.matrix_loader import MatrixDataLoader
|
||||
from backend.test.test_server.unit import app_config
|
||||
from backend.test import PROJECT_ROOT
|
||||
|
||||
|
||||
class EstDistTest(unittest.TestCase):
|
||||
"""Tests the diffexp returns the expected results for one test case, using the h5ad
|
||||
adaptor types and different algorithms."""
|
||||
|
||||
def load_dataset(self, path, extra_server_config={}, extra_dataset_config={}):
|
||||
config = app_config(path, extra_server_config=extra_server_config, extra_dataset_config=extra_dataset_config)
|
||||
loader = MatrixDataLoader(path)
|
||||
adaptor = loader.open(config)
|
||||
return adaptor
|
||||
|
||||
def test_adaptestimate_approximate_distribution(self):
|
||||
adaptor = self.load_dataset(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad")
|
||||
self.assertEqual(adaptor.get_X_approx_distribution(), XApproxDistribution.NORMAL)
|
||||
|
||||
def test_estimate_approximate_distribution(self):
|
||||
raw = np.random.exponential(scale=1000, size=(100, 40))
|
||||
|
||||
# ndarray
|
||||
self.assertEqual(estimate_approximate_distribution(raw), XApproxDistribution.COUNT)
|
||||
self.assertEqual(estimate_approximate_distribution(np.log1p(raw)), XApproxDistribution.NORMAL)
|
||||
|
||||
# csr_matrix
|
||||
self.assertEqual(estimate_approximate_distribution(sparse.csr_matrix(raw)), XApproxDistribution.COUNT)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(sparse.csr_matrix(np.log1p(raw))), XApproxDistribution.NORMAL
|
||||
)
|
||||
|
||||
# BIG (ie, trigger MT)
|
||||
big = np.random.exponential(scale=100, size=(1_000_000, 100))
|
||||
self.assertEqual(estimate_approximate_distribution(big), XApproxDistribution.COUNT)
|
||||
self.assertEqual(estimate_approximate_distribution(np.log1p(big)), XApproxDistribution.NORMAL)
|
||||
@@ -22,19 +22,27 @@ Test the anndata adaptor using the pbmc3k data set.
|
||||
|
||||
|
||||
@parameterized_class(
|
||||
("data_locator", "backed"),
|
||||
("data_locator", "backed", "X_approx_distribution"),
|
||||
[
|
||||
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", False),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", False),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", False),
|
||||
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", True),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", True),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", True),
|
||||
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", False, "auto"),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", False, "auto"),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", False, "auto"),
|
||||
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", True, "auto"),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", True, "auto"),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", True, "auto"),
|
||||
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", False, "normal"),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", False, "normal"),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", False, "normal"),
|
||||
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", True, "normal"),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", True, "normal"),
|
||||
(f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", True, "normal"),
|
||||
],
|
||||
)
|
||||
class AdaptorTest(unittest.TestCase):
|
||||
def setUp(self):
|
||||
config = app_config(self.data_locator, self.backed)
|
||||
config = app_config(
|
||||
self.data_locator, self.backed, extra_dataset_config=dict(X_approx_distribution=self.X_approx_distribution)
|
||||
)
|
||||
self.data = AnndataAdaptor(DataLocator(self.data_locator), config)
|
||||
|
||||
def test_init(self):
|
||||
@@ -90,7 +98,8 @@ class AdaptorTest(unittest.TestCase):
|
||||
|
||||
def test_schema_produces_error(self):
|
||||
self.data.data.obs["time"] = pd.Series(
|
||||
list([time.time() for i in range(self.data.cell_count)]), dtype="datetime64[ns]",
|
||||
list([time.time() for i in range(self.data.cell_count)]),
|
||||
dtype="datetime64[ns]",
|
||||
)
|
||||
with pytest.raises(TypeError):
|
||||
self.data._create_schema()
|
||||
@@ -107,7 +116,7 @@ class AdaptorTest(unittest.TestCase):
|
||||
self.assertTrue((Y >= 0).all() and (Y <= 1).all())
|
||||
|
||||
def test_layout_fields(self):
|
||||
""" X_pca, X_tsne, X_umap are available """
|
||||
"""X_pca, X_tsne, X_umap are available"""
|
||||
fbs = self.data.layout_to_fbs_matrix(["pca"])
|
||||
layout = decode_fbs.decode_matrix_FBS(fbs)
|
||||
self.assertEqual(layout["n_cols"], 2)
|
||||
@@ -127,7 +136,8 @@ class AdaptorTest(unittest.TestCase):
|
||||
self.assertEqual(annotations["n_cols"], 5)
|
||||
obs_index_col_name = self.data.get_schema()["annotations"]["obs"]["index"]
|
||||
self.assertEqual(
|
||||
annotations["col_idx"], [obs_index_col_name, "n_genes", "percent_mito", "n_counts", "louvain"],
|
||||
annotations["col_idx"],
|
||||
[obs_index_col_name, "n_genes", "percent_mito", "n_counts", "louvain"],
|
||||
)
|
||||
|
||||
fbs = self.data.annotation_to_fbs_matrix("var")
|
||||
@@ -153,12 +163,12 @@ class AdaptorTest(unittest.TestCase):
|
||||
f1 = {"filter": {"obs": {"index": [[0, 500]]}}}
|
||||
f2 = {"filter": {"obs": {"index": [[500, 1000]]}}}
|
||||
result = json.loads(self.data.diffexp_topN(f1["filter"], f2["filter"]))
|
||||
self.assertEqual(len(result['positive']), 10)
|
||||
self.assertEqual(len(result['negative']), 10)
|
||||
self.assertEqual(len(result["positive"]), 10)
|
||||
self.assertEqual(len(result["negative"]), 10)
|
||||
|
||||
result = json.loads(self.data.diffexp_topN(f1["filter"], f2["filter"], 20))
|
||||
self.assertEqual(len(result['positive']), 20)
|
||||
self.assertEqual(len(result['negative']), 20)
|
||||
self.assertEqual(len(result["positive"]), 20)
|
||||
self.assertEqual(len(result["negative"]), 20)
|
||||
|
||||
def test_data_frame(self):
|
||||
f1 = {"var": {"index": [[0, 10]]}}
|
||||
|
||||
Reference in New Issue
Block a user