mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-29 18:48:11 +08:00
move common code into server, update tests and makefile (#2425)
* move common code into server, update tests and makefile remove backend directory, refactor update smoke tests
This commit is contained in:
@@ -0,0 +1,105 @@
|
||||
import unittest
|
||||
|
||||
import numpy as np
|
||||
|
||||
from server.common.compute import diffexp_generic
|
||||
from server.data_common.matrix_loader import MatrixDataLoader
|
||||
from test.unit import app_config
|
||||
from test import PROJECT_ROOT
|
||||
|
||||
|
||||
class DiffExpTest(unittest.TestCase):
|
||||
"""Tests the diffexp returns the expected results for one test case, using the h5ad
|
||||
adaptor types and different algorithms."""
|
||||
|
||||
def load_dataset(self, path, extra_server_config={}, extra_dataset_config={}):
|
||||
config = app_config(path, extra_server_config=extra_server_config, extra_dataset_config=extra_dataset_config)
|
||||
loader = MatrixDataLoader(path)
|
||||
adaptor = loader.open(config)
|
||||
return adaptor
|
||||
|
||||
def get_mask(self, adaptor, start, stride):
|
||||
"""Simple function to return a mask or rows"""
|
||||
rows = adaptor.get_shape()[0]
|
||||
sel = list(range(start, rows, stride))
|
||||
mask = np.zeros(rows, dtype=bool)
|
||||
mask[sel] = True
|
||||
return mask
|
||||
|
||||
def compare_diffexp_results(self, results, expects):
|
||||
self.assertEqual(len(results), len(expects))
|
||||
for result, expect in zip(results, expects):
|
||||
self.assertEqual(result[0], expect[0])
|
||||
self.assertTrue(np.isclose(result[1], expect[1], 1e-6, 1e-4))
|
||||
self.assertTrue(np.isclose(result[2], expect[2], 1e-6, 1e-4))
|
||||
self.assertTrue(np.isclose(result[3], expect[3], 1e-6, 1e-4))
|
||||
|
||||
def check_1_10_2_10(self, results):
|
||||
"""Checks the results for a specific set of rows selections"""
|
||||
|
||||
positive_expects = [
|
||||
[1712, 0.24104056, 0.0051788902660723345, 1.0],
|
||||
[1575, 0.2615018, 0.007830310753043345, 1.0],
|
||||
[693, 0.23106655, 0.008715846769131548, 1.0],
|
||||
[916, 0.2395215, 0.009080596532247588, 1.0],
|
||||
[77, 0.22927025, 0.010070392939027756, 1.0],
|
||||
[782, 0.20581803, 0.010161745218916036, 1.0],
|
||||
[913, 0.23841085, 0.010782030711612685, 1.0],
|
||||
[910, 0.21493295, 0.014596411069229197, 1.0],
|
||||
[1727, 0.21911663, 0.015168372104237176, 1.0],
|
||||
[1443, 0.19814226, 0.015337080567465522, 1.0],
|
||||
]
|
||||
negative_expects = [
|
||||
[956, -0.29662406, 0.0008649321884808977, 1.0],
|
||||
[1124, -0.2607333, 0.0011717216548271284, 1.0],
|
||||
[1809, -0.24854594, 0.0019304405196777848, 1.0],
|
||||
[1754, -0.24683577, 0.005691734062127954, 1.0],
|
||||
[948, -0.18708363, 0.006622111055981219, 1.0],
|
||||
[1810, -0.2172082, 0.007055917428377063, 1.0],
|
||||
[779, -0.21150622, 0.007202934422407284, 1.0],
|
||||
[576, -0.19008157, 0.008272092578813124, 1.0],
|
||||
[538, -0.21803819, 0.01062259019889307, 1.0],
|
||||
[436, -0.2100364, 0.01127515110543434, 1.0],
|
||||
]
|
||||
|
||||
self.compare_diffexp_results(results["positive"], positive_expects)
|
||||
self.compare_diffexp_results(results["negative"], negative_expects)
|
||||
|
||||
def get_X_col(self, adaptor, cols):
|
||||
varmask = np.zeros(adaptor.get_shape()[1], dtype=bool)
|
||||
varmask[cols] = True
|
||||
return adaptor.get_X_array(None, varmask)
|
||||
|
||||
def test_anndata_default(self):
|
||||
"""Test an anndata adaptor with its default diffexp algorithm (diffexp_generic)"""
|
||||
adaptor = self.load_dataset(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad")
|
||||
maskA = self.get_mask(adaptor, 1, 10)
|
||||
maskB = self.get_mask(adaptor, 2, 10)
|
||||
results = adaptor.compute_diffexp_ttest(maskA, maskB, 10)
|
||||
self.check_1_10_2_10(results)
|
||||
|
||||
|
||||
def test_h5ad_default(self):
|
||||
"""Test a h5ad adaptor with its default diffexp algorithm (diffexp_cxg)"""
|
||||
adaptor = self.load_dataset(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad")
|
||||
maskA = self.get_mask(adaptor, 1, 10)
|
||||
maskB = self.get_mask(adaptor, 2, 10)
|
||||
|
||||
# run it through the adaptor
|
||||
results = adaptor.compute_diffexp_ttest(maskA, maskB, 10)
|
||||
self.check_1_10_2_10(results)
|
||||
|
||||
# run it directly
|
||||
|
||||
results = diffexp_generic.diffexp_ttest(adaptor, maskA, maskB, 10)
|
||||
self.check_1_10_2_10(results)
|
||||
|
||||
|
||||
def test_h5ad_generic(self):
|
||||
"""Test a h5ad adaptor with the generic adaptor"""
|
||||
adaptor = self.load_dataset(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad")
|
||||
maskA = self.get_mask(adaptor, 1, 10)
|
||||
maskB = self.get_mask(adaptor, 2, 10)
|
||||
# run it directly
|
||||
results = diffexp_generic.diffexp_ttest(adaptor, maskA, maskB, 10)
|
||||
self.check_1_10_2_10(results)
|
||||
@@ -0,0 +1,120 @@
|
||||
import unittest
|
||||
import numpy as np
|
||||
from scipy import sparse
|
||||
from server.common.compute.estimate_distribution import estimate_approximate_distribution
|
||||
from server.common.constants import XApproximateDistribution
|
||||
from server.data_common.matrix_loader import MatrixDataLoader
|
||||
from test.unit import app_config
|
||||
from test import PROJECT_ROOT
|
||||
|
||||
|
||||
class EstDistTest(unittest.TestCase):
|
||||
"""Tests the diffexp returns the expected results for one test case, using the h5ad
|
||||
adaptor types and different algorithms."""
|
||||
|
||||
def load_dataset(self, path, extra_server_config={}, extra_dataset_config={}):
|
||||
config = app_config(path, extra_server_config=extra_server_config, extra_dataset_config=extra_dataset_config)
|
||||
loader = MatrixDataLoader(path)
|
||||
adaptor = loader.open(config)
|
||||
return adaptor
|
||||
|
||||
def test_adaptestimate_approximate_distribution(self):
|
||||
adaptor = self.load_dataset(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad")
|
||||
self.assertEqual(adaptor.get_X_approximate_distribution(), XApproximateDistribution.NORMAL)
|
||||
|
||||
def test_estimate_approximate_distribution(self):
|
||||
raw = np.random.exponential(scale=1000, size=(100, 40))
|
||||
|
||||
# empty
|
||||
self.assertEqual(estimate_approximate_distribution(np.zeros((0,))), XApproximateDistribution.NORMAL)
|
||||
|
||||
# ndarray
|
||||
self.assertEqual(estimate_approximate_distribution(raw), XApproximateDistribution.COUNT)
|
||||
self.assertEqual(estimate_approximate_distribution(np.log1p(raw)), XApproximateDistribution.NORMAL)
|
||||
|
||||
# csr_matrix
|
||||
self.assertEqual(estimate_approximate_distribution(sparse.csr_matrix(raw)), XApproximateDistribution.COUNT)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(sparse.csr_matrix(np.log1p(raw))), XApproximateDistribution.NORMAL
|
||||
)
|
||||
|
||||
# csc_matrix
|
||||
self.assertEqual(estimate_approximate_distribution(sparse.csc_matrix(raw)), XApproximateDistribution.COUNT)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(sparse.csc_matrix(np.log1p(raw))), XApproximateDistribution.NORMAL
|
||||
)
|
||||
|
||||
# BIG (ie, trigger MT)
|
||||
big = np.random.exponential(scale=100, size=(1_000_000, 100))
|
||||
self.assertEqual(estimate_approximate_distribution(big), XApproximateDistribution.COUNT)
|
||||
self.assertEqual(estimate_approximate_distribution(np.log1p(big)), XApproximateDistribution.NORMAL)
|
||||
|
||||
def test_unsupported_throws(self):
|
||||
# dtypes and matrix formats we do not support
|
||||
with self.assertRaises(TypeError):
|
||||
estimate_approximate_distribution(np.array(["a", "b"]))
|
||||
with self.assertRaises(TypeError):
|
||||
estimate_approximate_distribution(sparse.coo_matrix(np.array([[0, 1, 2], [3, 0, 2]])))
|
||||
|
||||
def test_nonfinites(self):
|
||||
def put(arr, ind, vals):
|
||||
# like np.put, but creates and returns a modified copy of original array
|
||||
a = arr.copy()
|
||||
np.put(a, ind, vals)
|
||||
return a
|
||||
|
||||
# non-finites
|
||||
self.assertEqual(estimate_approximate_distribution(np.array([np.nan])), XApproximateDistribution.NORMAL)
|
||||
self.assertEqual(estimate_approximate_distribution(np.array([np.PINF])), XApproximateDistribution.NORMAL)
|
||||
self.assertEqual(estimate_approximate_distribution(np.array([np.NINF])), XApproximateDistribution.NORMAL)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(np.array([np.PINF, np.NINF, 0])), XApproximateDistribution.NORMAL
|
||||
)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(np.array([np.nan, np.PINF, np.NINF])), XApproximateDistribution.NORMAL
|
||||
)
|
||||
|
||||
raw = np.random.exponential(scale=1000, size=(50, 3))
|
||||
logged = np.log1p(raw)
|
||||
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(raw, [1], [np.nan])),
|
||||
XApproximateDistribution.COUNT,
|
||||
)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(raw, [1], [np.PINF])),
|
||||
XApproximateDistribution.COUNT,
|
||||
)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(raw, [1], [np.NINF])),
|
||||
XApproximateDistribution.COUNT,
|
||||
)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(raw, [1, 3, 88], [np.nan, np.PINF, np.NINF])),
|
||||
XApproximateDistribution.COUNT,
|
||||
)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(raw, [0, 1], [np.nan, np.nan])),
|
||||
XApproximateDistribution.COUNT,
|
||||
)
|
||||
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(logged, [1], [np.nan])),
|
||||
XApproximateDistribution.NORMAL,
|
||||
)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(logged, [1], [np.PINF])),
|
||||
XApproximateDistribution.NORMAL,
|
||||
)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(logged, [1], [np.NINF])),
|
||||
XApproximateDistribution.NORMAL,
|
||||
)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(logged, [1, 3, 88], [np.nan, np.PINF, np.NINF])),
|
||||
XApproximateDistribution.NORMAL,
|
||||
)
|
||||
self.assertEqual(
|
||||
estimate_approximate_distribution(put(logged, [0, 1], [np.nan, np.nan])),
|
||||
XApproximateDistribution.NORMAL,
|
||||
)
|
||||
Reference in New Issue
Block a user