mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-03 20:28:12 +08:00
Merge branch 'master' into colinmegill/geneset-prototype
This commit is contained in:
+60
-2
@@ -2,13 +2,19 @@ import random
|
||||
import shutil
|
||||
import string
|
||||
import tempfile
|
||||
import requests
|
||||
import time
|
||||
import os
|
||||
from subprocess import Popen
|
||||
from os import path, popen
|
||||
from contextlib import contextmanager
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from server.common.annotations import AnnotationsLocalFile
|
||||
from server.common.data_locator import DataLocator
|
||||
from server.common.app_config import AppConfig
|
||||
from server.common.app_config import AppConfig, DEFAULT_SERVER_PORT
|
||||
from server.common.utils import find_available_port
|
||||
from server.data_common.fbs.matrix import encode_matrix_fbs
|
||||
from server.data_common.matrix_loader import MatrixDataLoader, MatrixDataType
|
||||
|
||||
@@ -60,7 +66,7 @@ def skip_if(condition, reason: str):
|
||||
return decorator
|
||||
|
||||
|
||||
def app_config(data_locator, backed=False):
|
||||
def app_config(data_locator, backed=False, extra={}):
|
||||
args = {
|
||||
"embeddings__names": ["umap", "tsne", "pca"],
|
||||
"presentation__max_categories": 100,
|
||||
@@ -74,9 +80,61 @@ def app_config(data_locator, backed=False):
|
||||
}
|
||||
config = AppConfig()
|
||||
config.update(**args)
|
||||
config.update(**extra)
|
||||
config.complete_config()
|
||||
return config
|
||||
|
||||
|
||||
def random_string(n):
|
||||
return "".join(random.choice(string.ascii_letters) for _ in range(n))
|
||||
|
||||
|
||||
@contextmanager
|
||||
def test_server(command_line_args=[], app_config=None):
|
||||
"""A context to run the cellxgene server.
|
||||
Command line arguments can be passed in, as well as an app_config.
|
||||
This function is meant to be used like this, for example:
|
||||
|
||||
with test_server(...) as server:
|
||||
r = requests.get(f"{server}/...")
|
||||
// check r
|
||||
|
||||
where the server can be accessed within the context, and is terminated when
|
||||
the context is exited.
|
||||
The port is automatically set using find_available_port.
|
||||
The verbose flag is automatically set to True.
|
||||
If an app_config is provided, then this function writes a temporary
|
||||
yaml config file, which this server will read and parse.
|
||||
"""
|
||||
|
||||
port = DEFAULT_SERVER_PORT
|
||||
port = find_available_port("localhost", port)
|
||||
command = ["cellxgene", "--no-upgrade-check", "launch", "--verbose", "--port=%d" % port] + command_line_args
|
||||
|
||||
tempdir = None
|
||||
if app_config:
|
||||
tempdir = tempfile.TemporaryDirectory()
|
||||
config_file = os.path.join(tempdir.name, "config.yaml")
|
||||
app_config.write_config(config_file)
|
||||
command.extend(["-c", config_file])
|
||||
|
||||
server = f"http://localhost:{port}"
|
||||
ps = Popen(command)
|
||||
|
||||
for _ in range(10):
|
||||
try:
|
||||
requests.get(f"{server}/health")
|
||||
break
|
||||
except requests.exceptions.ConnectionError:
|
||||
time.sleep(1)
|
||||
|
||||
if tempdir:
|
||||
tempdir.cleanup()
|
||||
|
||||
try:
|
||||
yield server
|
||||
finally:
|
||||
try:
|
||||
ps.terminate()
|
||||
except ProcessLookupError:
|
||||
pass
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
import anndata
|
||||
import argparse
|
||||
import random
|
||||
import scipy
|
||||
import numpy as np
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser("A command to generate test h5ad files")
|
||||
parser.add_argument("output", help="Name of the output file")
|
||||
parser.add_argument("nobs", type=int, help="Number of observations (rows)")
|
||||
parser.add_argument("nvar", type=int, help="Number of variables (columns)")
|
||||
parser.add_argument("-n", "--nnz-percent", type=float, default=100, help="percent of non-zeros")
|
||||
parser.add_argument("-c", "--col-shift", action="store_true", help="add a random value to each column")
|
||||
parser.add_argument("--seed", type=int, default=None, help="add a random value to each column")
|
||||
|
||||
args = parser.parse_args()
|
||||
create_test_h5ad(args.output, args.nobs, args.nvar, args.nnz_percent, args.col_shift, args.seed)
|
||||
|
||||
|
||||
def create_test_h5ad(outfile, nobs, nvar, nnz_percent=100, apply_col_shift=False, seed=None):
|
||||
random.seed(seed)
|
||||
np.random.seed(seed)
|
||||
x = create_X_array(nobs, nvar, nnz_percent, apply_col_shift)
|
||||
obsm = {"X_random": np.random.rand(nobs, 2).astype(np.float32)}
|
||||
adata = anndata.AnnData(x, obsm=obsm)
|
||||
adata.write(outfile)
|
||||
|
||||
|
||||
def create_X_array(nobs, nvar, nnz_percent, apply_col_shift):
|
||||
if nnz_percent < 100:
|
||||
array = scipy.sparse.random(nobs, nvar, nnz_percent * 0.01, dtype=np.float32, format="csc")
|
||||
else:
|
||||
array = np.random.rand(nobs, nvar).astype(np.float32)
|
||||
|
||||
if apply_col_shift:
|
||||
col_shift = np.random.rand((nvar))
|
||||
array += col_shift
|
||||
|
||||
return array
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+43
-12
@@ -15,8 +15,10 @@ from server.data_cxg.cxg_adaptor import CxgAdaptor
|
||||
def main():
|
||||
parser = argparse.ArgumentParser("A command to test diffexp")
|
||||
parser.add_argument("dataset", help="name of a dataset to load")
|
||||
parser.add_argument("-na", "--numA", type=int, required=True, help="number of rows in group A")
|
||||
parser.add_argument("-nb", "--numB", type=int, required=True, help="number of rows in group B")
|
||||
parser.add_argument("-na", "--numA", type=int, help="number of rows in group A")
|
||||
parser.add_argument("-nb", "--numB", type=int, help="number of rows in group B")
|
||||
parser.add_argument("-va", "--varA", help="obs variable:value to use for group A")
|
||||
parser.add_argument("-vb", "--varB", help="obs variable:value to use for group B")
|
||||
parser.add_argument("-t", "--trials", default=1, type=int, help="number of trials")
|
||||
parser.add_argument(
|
||||
"-a", "--alg", choices=("default", "generic", "cxg"), default="default", help="algorithm to use"
|
||||
@@ -41,22 +43,34 @@ def main():
|
||||
if isinstance(adaptor, CxgAdaptor):
|
||||
adaptor.open_array("X").schema.dump()
|
||||
|
||||
numA = args.numA
|
||||
numB = args.numB
|
||||
random.seed(args.seed)
|
||||
np.random.seed(args.seed)
|
||||
rows = adaptor.get_shape()[0]
|
||||
|
||||
random.seed(args.seed)
|
||||
if args.numA:
|
||||
filterA = random.sample(range(rows), args.numA)
|
||||
elif args.varA:
|
||||
vname, vval = args.varA.split(":")
|
||||
filterA = get_filter_from_obs(adaptor, vname, vval)
|
||||
else:
|
||||
print("must supply numA or varA")
|
||||
sys.exit(1)
|
||||
|
||||
if not args.new_selection:
|
||||
samples = random.sample(range(rows), numA + numB)
|
||||
filterA = samples[:numA]
|
||||
filterB = samples[numA:]
|
||||
if args.numB:
|
||||
filterB = random.sample(range(rows), args.numB)
|
||||
elif args.varB:
|
||||
vname, vval = args.varB.split(":")
|
||||
filterB = get_filter_from_obs(adaptor, vname, vval)
|
||||
else:
|
||||
print("must supply numB or varB")
|
||||
sys.exit(1)
|
||||
|
||||
for i in range(args.trials):
|
||||
if args.new_selection:
|
||||
samples = random.sample(range(rows), numA + numB)
|
||||
filterA = samples[:numA]
|
||||
filterB = samples[numA:]
|
||||
if args.numA:
|
||||
filterA = random.sample(range(rows), args.numA)
|
||||
if args.numB:
|
||||
filterB = random.sample(range(rows), args.numB)
|
||||
|
||||
maskA = np.zeros(rows, dtype=bool)
|
||||
maskA[filterA] = True
|
||||
@@ -82,5 +96,22 @@ def main():
|
||||
print(res)
|
||||
|
||||
|
||||
def get_filter_from_obs(adaptor, obsname, obsval):
|
||||
attrs = adaptor.get_obs_columns()
|
||||
if obsname not in attrs:
|
||||
print(f"Unknown obs attr {obsname}: expected on of {attrs}")
|
||||
sys.exit(1)
|
||||
obsvals = adaptor.query_obs_array(obsname)[:]
|
||||
obsval = type(obsvals[0])(obsval)
|
||||
|
||||
vfilter = np.where(obsvals == obsval)[0]
|
||||
if len(vfilter) == 0:
|
||||
u = np.unique(obsvals)
|
||||
print(f"Unknown value in variable {obsname}:{obsval}: expected one of {list(u)}")
|
||||
sys.exit(1)
|
||||
|
||||
return vfilter
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
@@ -244,6 +244,19 @@ class EndPoints(object):
|
||||
self.assertEqual(df["n_rows"], 2638)
|
||||
self.assertEqual(df["n_cols"], 1)
|
||||
|
||||
def test_data_get_unknown_filter_fbs(self):
|
||||
index_col_name = self.schema["schema"]["annotations"]["var"]["index"]
|
||||
endpoint = "data/var"
|
||||
query = f"var:{index_col_name}=UNKNOWN"
|
||||
url = f"{self.URL_BASE}{endpoint}?{query}"
|
||||
header = {"Accept": "application/octet-stream"}
|
||||
result = self.session.get(url, headers=header)
|
||||
self.assertEqual(result.status_code, HTTPStatus.OK)
|
||||
self.assertEqual(result.headers["Content-Type"], "application/octet-stream")
|
||||
df = decode_fbs.decode_matrix_FBS(result.content)
|
||||
self.assertEqual(df["n_rows"], 2638)
|
||||
self.assertEqual(df["n_cols"], 0)
|
||||
|
||||
def test_data_put_single_var(self):
|
||||
endpoint = "data/var"
|
||||
url = f"{self.URL_BASE}{endpoint}"
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
import unittest
|
||||
from server.common.app_config import AppConfig
|
||||
from server.common.errors import ConfigurationError
|
||||
from server.test import PROJECT_ROOT, test_server
|
||||
import requests
|
||||
|
||||
# NOTE, there are more tests that should be written for AppConfig.
|
||||
# this is just a start.
|
||||
@@ -26,3 +29,43 @@ class AppConfigTest(unittest.TestCase):
|
||||
c.update(server__scripts=("a", "b"), server__inline_scripts=["c", "d"])
|
||||
v = c.changes_from_default()
|
||||
self.assertCountEqual(v, [("server__scripts", ["a", "b"], []), ("server__inline_scripts", ["c", "d"], [])])
|
||||
|
||||
def test_multi_dataset(self):
|
||||
|
||||
c = AppConfig()
|
||||
# test for illegal url_dataroots
|
||||
for illegal in ("a/b", "../b", "!$*", "\\n", "", "(bad)"):
|
||||
c.update(multi_dataset__dataroot={illegal: f"{PROJECT_ROOT}/example-dataset"})
|
||||
with self.assertRaises(ConfigurationError):
|
||||
c.complete_config()
|
||||
|
||||
# test for legal url_dataroots
|
||||
for legal in (
|
||||
"d",
|
||||
"this.is-okay_",
|
||||
):
|
||||
c.update(multi_dataset__dataroot={legal: f"{PROJECT_ROOT}/example-dataset"})
|
||||
c.complete_config()
|
||||
|
||||
# test that multi dataroots work end to end
|
||||
c.update(
|
||||
multi_dataset__dataroot=dict(
|
||||
set1=f"{PROJECT_ROOT}/example-dataset",
|
||||
set2=f"{PROJECT_ROOT}/server/test/test_datasets"
|
||||
)
|
||||
)
|
||||
c.complete_config()
|
||||
|
||||
with test_server(app_config=c) as server:
|
||||
session = requests.Session()
|
||||
|
||||
r = session.get(f"{server}/set1/pbmc3k.h5ad/api/v0.2/config")
|
||||
data_config = r.json()
|
||||
assert data_config["config"]["displayNames"]["dataset"] == "pbmc3k"
|
||||
|
||||
r = session.get(f"{server}/set2/pbmc3k.cxg/api/v0.2/config")
|
||||
data_config = r.json()
|
||||
assert data_config["config"]["displayNames"]["dataset"] == "pbmc3k"
|
||||
|
||||
r = session.get(f"{server}/health")
|
||||
assert r.json()["status"] == "pass"
|
||||
|
||||
+84
-15
@@ -1,24 +1,24 @@
|
||||
import unittest
|
||||
from server.data_common.matrix_loader import MatrixDataLoader
|
||||
from server.common.app_config import AppConfig
|
||||
from server.test import PROJECT_ROOT, app_config
|
||||
import server.compute.diffexp_cxg as diffexp_cxg
|
||||
import server.compute.diffexp_generic as diffexp_generic
|
||||
from server.converters.cxgtool import write_cxg
|
||||
from server.test.create_test_matrix import create_test_h5ad
|
||||
from server.data_common.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
|
||||
import numpy as np
|
||||
|
||||
from server.test import PROJECT_ROOT
|
||||
import tempfile
|
||||
import os
|
||||
|
||||
|
||||
class DiffExpTest(unittest.TestCase):
|
||||
"""Tests the diffexp returns the expected results for one test case, using different
|
||||
adaptor types and different algorithms."""
|
||||
|
||||
def load_dataset(self, path):
|
||||
app_config = AppConfig()
|
||||
app_config.single_dataset__datapath = path
|
||||
app_config.server__verbose = True
|
||||
app_config.complete_config()
|
||||
def load_dataset(self, path, extra={}):
|
||||
config = app_config(path, extra=extra)
|
||||
loader = MatrixDataLoader(path)
|
||||
adaptor = loader.open(app_config)
|
||||
adaptor = loader.open(config)
|
||||
return adaptor
|
||||
|
||||
def get_mask(self, adaptor, start, stride):
|
||||
@@ -29,6 +29,14 @@ class DiffExpTest(unittest.TestCase):
|
||||
mask[sel] = True
|
||||
return mask
|
||||
|
||||
def compare_diffexp_results(self, results, expects):
|
||||
self.assertEqual(len(results), len(expects))
|
||||
for result, expect in zip(results, expects):
|
||||
self.assertEqual(result[0], expect[0])
|
||||
self.assertTrue(np.isclose(result[1], expect[1], 1e-6, 1e-4))
|
||||
self.assertTrue(np.isclose(result[2], expect[2], 1e-6, 1e-4))
|
||||
self.assertTrue(np.isclose(result[3], expect[3], 1e-6, 1e-4))
|
||||
|
||||
def check_1_10_2_10(self, results):
|
||||
"""Checks the results for a specific set of rows selections"""
|
||||
expects = [
|
||||
@@ -43,12 +51,12 @@ class DiffExpTest(unittest.TestCase):
|
||||
[1575, 1.0317602, 0.007830310753043345, 1.0],
|
||||
[576, 0.97873515, 0.008272092578813124, 1.0],
|
||||
]
|
||||
self.assertEqual(len(results), len(expects))
|
||||
for result, expect in zip(results, expects):
|
||||
self.assertEqual(result[0], expect[0])
|
||||
self.assertAlmostEqual(result[1], expect[1])
|
||||
self.assertAlmostEqual(result[2], expect[2])
|
||||
self.assertAlmostEqual(result[3], expect[3])
|
||||
self.compare_diffexp_results(results, expects)
|
||||
|
||||
def get_X_col(self, adaptor, cols):
|
||||
varmask = np.zeros(adaptor.get_shape()[1], dtype=bool)
|
||||
varmask[cols] = True
|
||||
return adaptor.get_X_array(None, varmask)
|
||||
|
||||
def test_anndata_default(self):
|
||||
"""Test an anndata adaptor with its default diffexp algorithm (diffexp_generic)"""
|
||||
@@ -80,3 +88,64 @@ class DiffExpTest(unittest.TestCase):
|
||||
# run it directly
|
||||
results = diffexp_generic.diffexp_ttest(adaptor, maskA, maskB, 10)
|
||||
self.check_1_10_2_10(results)
|
||||
|
||||
def test_cxg_sparse(self):
|
||||
self.sparse_diffexp(False)
|
||||
|
||||
def test_cxg_sparse_col_shift(self):
|
||||
self.sparse_diffexp(True)
|
||||
|
||||
def sparse_diffexp(self, apply_col_shift):
|
||||
with tempfile.TemporaryDirectory() as dirname:
|
||||
# create a sparse matrix
|
||||
h5adfile = os.path.join(dirname, "sparse.h5ad")
|
||||
create_test_h5ad(h5adfile, 2000, 2000, 10, apply_col_shift)
|
||||
adaptor_anndata = self.load_dataset(h5adfile, extra=dict(embeddings__names=[]))
|
||||
adata = adaptor_anndata.data
|
||||
|
||||
sparsename = os.path.join(dirname, "sparse.cxg")
|
||||
write_cxg(adata=adata, container=sparsename, title="sparse", sparse_threshold=11)
|
||||
adaptor_sparse = self.load_dataset(sparsename)
|
||||
assert adaptor_sparse.open_array("X").schema.sparse
|
||||
assert adaptor_sparse.has_array("X_col_shift") == apply_col_shift
|
||||
|
||||
densename = os.path.join(dirname, "dense.cxg")
|
||||
write_cxg(adata=adata, container=densename, title="dense", sparse_threshold=0)
|
||||
adaptor_dense = self.load_dataset(densename)
|
||||
assert not adaptor_dense.open_array("X").schema.sparse
|
||||
assert not adaptor_dense.has_array("X_col_shift")
|
||||
|
||||
maskA = self.get_mask(adaptor_anndata, 1, 10)
|
||||
maskB = self.get_mask(adaptor_anndata, 2, 10)
|
||||
|
||||
diffexp_results_anndata = diffexp_generic.diffexp_ttest(adaptor_anndata, maskA, maskB, 10)
|
||||
diffexp_results_sparse = diffexp_cxg.diffexp_ttest(adaptor_sparse, maskA, maskB, 10)
|
||||
diffexp_results_dense = diffexp_cxg.diffexp_ttest(adaptor_dense, maskA, maskB, 10)
|
||||
|
||||
self.compare_diffexp_results(diffexp_results_anndata, diffexp_results_sparse)
|
||||
self.compare_diffexp_results(diffexp_results_anndata, diffexp_results_dense)
|
||||
|
||||
topcols = np.array([x[0] for x in diffexp_results_anndata])
|
||||
cols_anndata = self.get_X_col(adaptor_anndata, topcols)
|
||||
cols_sparse = self.get_X_col(adaptor_sparse, topcols)
|
||||
cols_dense = self.get_X_col(adaptor_dense, topcols)
|
||||
assert cols_anndata.shape[0] == adaptor_sparse.get_shape()[0]
|
||||
assert cols_anndata.shape[1] == len(diffexp_results_anndata)
|
||||
|
||||
def convert(mat, cols):
|
||||
return decode_matrix_fbs(encode_matrix_fbs(mat, col_idx=cols)).to_numpy()
|
||||
|
||||
cols_anndata = convert(cols_anndata, topcols)
|
||||
cols_sparse = convert(cols_sparse, topcols)
|
||||
cols_dense = convert(cols_dense, topcols)
|
||||
|
||||
x = adaptor_sparse.get_X_array()
|
||||
assert x.shape == adaptor_sparse.get_shape()
|
||||
|
||||
for row in range(cols_anndata.shape[0]):
|
||||
for col in range(cols_anndata.shape[1]):
|
||||
vanndata = cols_anndata[row][col]
|
||||
vsparse = cols_sparse[row][col]
|
||||
vdense = cols_dense[row][col]
|
||||
self.assertTrue(np.isclose(vanndata, vsparse, 1e-6, 1e-6))
|
||||
self.assertTrue(np.isclose(vanndata, vdense, 1e-6, 1e-6))
|
||||
|
||||
Reference in New Issue
Block a user