genesets route for local server (#2079)

* first cut at GET /genesets route

* update existing tests to match code changes

* more GET /genesets and initial tests

* add missing test fixture

* geneset validation accepts OTA format

* genesets route: better error handling, more tests

* lint
This commit is contained in:
Bruce Martin
2021-02-26 17:53:07 -08:00
committed by GitHub
parent 09466a5c32
commit f3a3820ffa
18 changed files with 813 additions and 82 deletions
+5 -1
View File
@@ -46,7 +46,11 @@ def data_with_tmp_annotations(ext: MatrixDataType, annotations_fixture=False):
config.complete_config()
data = MatrixDataLoader(data_locator.abspath()).open(config)
annotations = AnnotationsLocalFile(None, annotations_file)
anno_config = {
"user-annotations": True,
"genesets-save": False,
}
annotations = AnnotationsLocalFile(anno_config, None, annotations_file, None)
return data, tmp_dir, annotations
+3
View File
@@ -16,9 +16,12 @@ dataset:
local_file_csv:
directory: {local_file_csv_directory}
file: {local_file_csv_file}
genesets_file: {local_file_csv_genesets_file}
ontology:
enable: {ontology_enabled}
obo_location: {obo_location}
genesets:
readonly: {genesets_readonly}
embeddings:
names: {embedding_names}
+12
View File
@@ -0,0 +1,12 @@
# Test fixture
geneset_name, geneset_description, gene_symbol, gene_description
first geneset name,,F5, a gene_description
first geneset name,a description, NO_SUCH_GENE, non-existent gene
first geneset name,a description, F5, duplicate gene
first geneset name, a description, SUMO3,
first geneset name,, SRM,
second geneset,,RER1
second geneset,,SIK1
third geneset,,NO_SUCH_GENE
fourth_geneset,fourth description,,gene intentionally missing
fifth_dataset,,,
1 # Test fixture
2 geneset_name, geneset_description, gene_symbol, gene_description
3 first geneset name,,F5, a gene_description
4 first geneset name,a description, NO_SUCH_GENE, non-existent gene
5 first geneset name,a description, F5, duplicate gene
6 first geneset name, a description, SUMO3,
7 first geneset name,, SRM,
8 second geneset,,RER1
9 second geneset,,SIK1
10 third geneset,,NO_SUCH_GENE
11 fourth_geneset,fourth description,,gene intentionally missing
12 fifth_dataset,,,
+1 -1
View File
@@ -14,7 +14,7 @@ class AuthTest(unittest.TestCase):
app_config = AppConfig()
app_config.update_server_config(app__flask_secret_key="secret")
app_config.update_server_config(authentication__type=None, single_dataset__datapath=self.dataset_datapath)
app_config.update_dataset_config(user_annotations__enable=False)
app_config.update_dataset_config(user_annotations__enable=False, user_annotations__genesets__readonly=True)
app_config.complete_config()
@@ -92,8 +92,10 @@ class ConfigTests(unittest.TestCase):
hosted_file_directory="null",
local_file_csv_directory="null",
local_file_csv_file="null",
local_file_csv_genesets_file="null",
ontology_enabled="false",
obo_location="null",
genesets_readonly="false",
embedding_names=[],
enable_reembedding="false",
enable_difexp="true",
@@ -142,8 +144,10 @@ class ConfigTests(unittest.TestCase):
hosted_file_directory=hosted_file_directory,
local_file_csv_directory=local_file_csv_directory,
local_file_csv_file=local_file_csv_file,
local_file_csv_genesets_file=local_file_csv_genesets_file,
ontology_enabled=ontology_enabled,
obo_location=obo_location,
genesets_readonly=genesets_readonly,
embedding_names=embedding_names,
enable_reembedding=enable_reembedding,
enable_difexp=enable_difexp,
@@ -178,8 +182,10 @@ class ConfigTests(unittest.TestCase):
hosted_file_directory="null",
local_file_csv_directory="null",
local_file_csv_file="null",
local_file_csv_genesets_file="null",
ontology_enabled="false",
obo_location="null",
genesets_readonly="false",
embedding_names=[],
enable_reembedding="false",
enable_difexp="true",
@@ -47,7 +47,7 @@ class TestDatasetConfig(ConfigTests):
mock_check_attrs.side_effect = BaseConfig.validate_correct_type_of_configuration_attribute()
self.dataset_config.complete_config(self.context)
self.assertIsNotNone(self.config.server_config.data_adaptor)
self.assertEqual(mock_check_attrs.call_count, 17)
self.assertEqual(mock_check_attrs.call_count, 19)
def test_app_sets_script_vars(self):
config = self.get_config(scripts=["path/to/script"])
@@ -102,7 +102,7 @@ class TestDatasetConfig(ConfigTests):
enable_users_annotations="true", authentication_enable="true", annotation_type="local_file_csv"
)
config.server_config.complete_config(self.context)
config.dataset_config.handle_local_file_csv_annotations()
config.dataset_config.handle_local_file_csv_annotations(self.context)
self.assertIsInstance(config.dataset_config.user_annotations, AnnotationsLocalFile)
cwd = os.getcwd()
self.assertEqual(config.dataset_config.user_annotations._get_output_dir(), cwd)
+264 -2
View File
@@ -3,6 +3,8 @@ import time
import unittest
import zlib
from http import HTTPStatus
import tempfile
from os import path
import pandas as pd
import requests
@@ -13,6 +15,7 @@ from local_server.test import (
data_with_tmp_annotations,
make_fbs,
PROJECT_ROOT,
FIXTURES_ROOT,
start_test_server,
stop_test_server,
)
@@ -26,6 +29,7 @@ BAD_FILTER = {"filter": {"obs": {"annotation_value": [{"name": "xyz"}]}}}
class EndPoints(object):
ANNOTATIONS_ENABLED = True
GENESETS_READONLY = False
def test_initialize(self):
endpoint = "schema"
@@ -49,6 +53,7 @@ class EndPoints(object):
result_data = result.json()
self.assertIn("library_versions", result_data["config"])
self.assertEqual(result_data["config"]["displayNames"]["dataset"], "pbmc3k")
self.assertIsNotNone(result_data["config"]["parameters"])
def test_get_layout_fbs(self):
endpoint = "layout/obs"
@@ -286,6 +291,26 @@ class EndPoints(object):
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.OK)
def test_genesets_config(self):
result = self.session.get(f"{self.URL_BASE}config")
config_data = result.json()
params = config_data["config"]["parameters"]
annotations_genesets = params["annotations_genesets"]
annotations_genesets_readonly = params["annotations_genesets_readonly"]
annotations_genesets_summary_methods = params["annotations_genesets_summary_methods"]
self.assertTrue(annotations_genesets)
self.assertEqual(annotations_genesets_readonly, self.GENESETS_READONLY)
self.assertEqual(annotations_genesets_summary_methods, ["mean"])
def test_get_genesets(self):
endpoint = "genesets"
url = f"{self.URL_BASE}{endpoint}"
result = self.session.get(url, headers={"Accept": "application/json"})
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.headers["Content-Type"], "application/json")
result_data = result.json()
self.assertIsNotNone(result_data["genesets"])
def _setupClass(child_class, command_line):
child_class.ps, child_class.server = start_test_server(command_line)
child_class.URL_BASE = f"{child_class.server}/api/v0.2/"
@@ -304,7 +329,8 @@ class EndPointsAnnotations(EndPoints):
def test_get_user_annotations_existing_obs_keys_fbs(self):
self._test_get_user_annotations_obs_keys_fbs(
"cluster-test", {"unassigned", "one", "two", "three", "four", "five", "six", "seven"},
"cluster-test",
{"unassigned", "one", "two", "three", "four", "five", "six", "seven"},
)
def test_put_user_annotations_obs_fbs(self):
@@ -353,6 +379,7 @@ class EndPointsAnndata(unittest.TestCase, EndPoints):
"""Test Case for endpoints"""
ANNOTATIONS_ENABLED = False
GENESETS_READONLY = True
@classmethod
def setUpClass(cls):
@@ -361,6 +388,7 @@ class EndPointsAnndata(unittest.TestCase, EndPoints):
[
f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad",
"--disable-annotations",
"--disable-genesets-save",
"--experimental-enable-reembedding",
],
)
@@ -408,15 +436,249 @@ class EndPointsAnndataAnnotations(unittest.TestCase, EndPointsAnnotations):
"""Test Case for endpoints"""
ANNOTATIONS_ENABLED = True
GENESETS_READONLY = False
@classmethod
def setUpClass(cls):
cls.data, cls.tmp_dir, cls.annotations = data_with_tmp_annotations(
MatrixDataType.H5AD, annotations_fixture=True
)
cls._setupClass(cls, ["--annotations-file", cls.annotations.output_file, cls.data.get_location()])
cls._setupClass(cls, ["--annotations-file", cls.annotations.label_output_file, cls.data.get_location()])
@classmethod
def tearDownClass(cls):
shutil.rmtree(cls.tmp_dir)
stop_test_server(cls.ps)
class EndPointsAnnDataGenesets(unittest.TestCase, EndPoints):
ANNOTATIONS_ENABLED = False
GENESETS_READONLY = False
@classmethod
def setUpClass(cls):
cls.tmp_dir = tempfile.mkdtemp()
genesets_file = path.join(cls.tmp_dir, "test_genesets.csv")
shutil.copyfile(f"{FIXTURES_ROOT}/pbmc3k-genesets.csv", genesets_file)
cls._setupClass(
cls,
[
f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad",
"--disable-annotations",
"--genesets-file",
genesets_file,
],
)
@classmethod
def tearDownClass(cls):
shutil.rmtree(cls.tmp_dir)
stop_test_server(cls.ps)
def test_get_genesets_json(self):
endpoint = "genesets"
url = f"{self.URL_BASE}{endpoint}"
result = self.session.get(url, headers={"Accept": "application/json"})
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.headers["Content-Type"], "application/json")
result_data = result.json()
self.assertIsNotNone(result_data["genesets"])
self.assertIsNotNone(result_data["tid"])
self.assertEqual(
result_data,
{
"genesets": [
{
"genes": [
{"gene_description": "a gene_description", "gene_symbol": "F5"},
{"gene_description": "", "gene_symbol": "SUMO3"},
{"gene_description": "", "gene_symbol": "SRM"},
],
"geneset_description": "a description",
"geneset_name": "first geneset name",
},
{
"genes": [
{"gene_description": "", "gene_symbol": "RER1"},
{"gene_description": "", "gene_symbol": "SIK1"},
],
"geneset_description": "",
"geneset_name": "second geneset",
},
{"genes": [], "geneset_description": "", "geneset_name": "third geneset"},
{"genes": [], "geneset_description": "fourth description", "geneset_name": "fourth_geneset"},
{"genes": [], "geneset_description": "", "geneset_name": "fifth_dataset"},
],
"tid": 0,
},
)
def test_get_genesets_csv(self):
endpoint = "genesets"
url = f"{self.URL_BASE}{endpoint}"
result = self.session.get(url, headers={"Accept": "text/csv"})
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.headers["Content-Type"], "text/csv")
self.assertEqual(
result.text,
"""geneset_name,geneset_description,gene_symbol,gene_description\r
first geneset name,a description,F5,a gene_description\r
first geneset name,a description,SUMO3,\r
first geneset name,a description,SRM,\r
second geneset,,RER1,\r
second geneset,,SIK1,\r
third geneset,,,\r
fourth_geneset,fourth description,,\r
fifth_dataset,,,\r
""",
)
def test_put_genesets(self):
endpoint = "genesets"
url = f"{self.URL_BASE}{endpoint}"
# assume we start with TID 0
result = self.session.get(url, headers={"Accept": "application/json"})
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.json()["tid"], 0)
test1 = {"tid": 3, "genesets": []}
result = self.session.put(url, json=test1)
self.assertEqual(result.status_code, HTTPStatus.OK)
result = self.session.get(url, headers={"Accept": "application/json"})
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.json(), test1)
# stale TID
result = self.session.put(url, json=test1)
self.assertEqual(result.status_code, HTTPStatus.NOT_FOUND)
test2 = {"tid": 4, "genesets": [{"geneset_name": "foobar", "genes": []}]}
test2_response = {"tid": 4, "genesets": [{"geneset_name": "foobar", "geneset_description": "", "genes": []}]}
result = self.session.put(url, json=test2)
self.assertEqual(result.status_code, HTTPStatus.OK)
result = self.session.get(url, headers={"Accept": "application/json"})
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.json(), test2_response)
test3 = {
"tid": 5,
"genesets": [
{
"geneset_name": "foobar",
"geneset_description": "",
"genes": [
{
"gene_symbol": "F5",
"gene_description": "",
}
],
}
],
}
result = self.session.put(url, json=test3)
self.assertEqual(result.status_code, HTTPStatus.OK)
result = self.session.get(url, headers={"Accept": "application/json"})
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.json(), test3)
def test_put_genesets_malformed(self):
""" test malformed submissions that we expect the backend to catch/tolerate """
endpoint = "genesets"
url = f"{self.URL_BASE}{endpoint}"
result = self.session.get(url, headers={"Accept": "application/json"})
self.assertEqual(result.status_code, HTTPStatus.OK)
original_data = result.json()
tid = original_data["tid"]
def test_case(test, expected_code, original_data):
""" check for expected error AND that no change was made to the original state """
result = self.session.put(url, json=test)
self.assertEqual(result.status_code, expected_code)
result = self.session.get(url, headers={"Accept": "application/json"})
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.json(), original_data)
# missing or malformed genesets
test_case(
{"tid": tid + 1},
HTTPStatus.BAD_REQUEST,
original_data,
)
test_case(
{"tid": tid + 1, "genesets": 99},
HTTPStatus.BAD_REQUEST,
original_data,
)
# illegal geneset_name
test_case(
{"tid": tid + 1, "genesets": [{"geneset_name": """, "genes": []}]},
HTTPStatus.BAD_REQUEST,
original_data,
)
# duplicate geneset_name
test_case(
{
"tid": tid + 1,
"genesets": [
{"geneset_name": "foo", "genes": []},
{"geneset_name": "foo", "genes": []},
],
},
HTTPStatus.BAD_REQUEST,
original_data,
)
# missing geneset_name
test_case(
{"tid": tid + 1, "genesets": [{"genes": []}]},
HTTPStatus.BAD_REQUEST,
original_data,
)
# non-numeric TID
test_case(
{"tid": [], "genesets": [{"geneset_name": "foo", "genes": []}]},
HTTPStatus.BAD_REQUEST,
original_data,
)
test_case(
{"tid": None, "genesets": [{"geneset_name": "foo", "genes": []}]},
HTTPStatus.BAD_REQUEST,
original_data,
)
test_case(
{"tid": "not a number", "genesets": [{"geneset_name": "foo", "genes": []}]},
HTTPStatus.BAD_REQUEST,
original_data,
)
# duplicate gene_symbol
test_case(
{
"tid": "not a number",
"genesets": [{"geneset_name": "foo", "genes": [{"gene_symbol": "SIK1"}, {"gene_symbol": "SIK1"}]}],
},
HTTPStatus.BAD_REQUEST,
original_data,
)
# gene_symbol is not a string
test_case(
{
"tid": "not a number",
"genesets": [{"geneset_name": "foo", "genes": [{"gene_symbol": 99}]}],
},
HTTPStatus.BAD_REQUEST,
original_data,
)
"""
TODO once we have some code to support it:
1. GET genesets_summary
2. genesets_summary obeys tid
"""
@@ -45,8 +45,8 @@ class WritableAnnotationTest(unittest.TestCase):
)
res = self.annotation_put_fbs(fbs)
self.assertEqual(res, json.dumps({"status": "OK"}))
self.assertTrue(path.exists(self.annotations.output_file))
df = pd.read_csv(self.annotations.output_file, index_col=0, header=0, comment="#")
self.assertTrue(path.exists(self.annotations.label_output_file))
df = pd.read_csv(self.annotations.label_output_file, index_col=0, header=0, comment="#")
self.assertEqual(df.shape, (n_rows, 2))
self.assertEqual(set(df.columns), {"cat_A", "cat_B"})
self.assertTrue(self.data.original_obs_index.equals(df.index))
@@ -62,14 +62,14 @@ class WritableAnnotationTest(unittest.TestCase):
)
res = self.annotation_put_fbs(fbs)
self.assertEqual(res, json.dumps({"status": "OK"}))
self.assertTrue(path.exists(self.annotations.output_file))
df = pd.read_csv(self.annotations.output_file, index_col=0, header=0, comment="#")
self.assertTrue(path.exists(self.annotations.label_output_file))
df = pd.read_csv(self.annotations.label_output_file, index_col=0, header=0, comment="#")
self.assertEqual(set(df.columns), {"cat_A", "cat_C"})
self.assertTrue(np.all(df["cat_A"] == ["label_A1"] * n_rows))
self.assertTrue(np.all(df["cat_C"] == ["label_C"] * n_rows))
# rotation
name, ext = path.splitext(self.annotations.output_file)
name, ext = path.splitext(self.annotations.label_output_file)
backup_dir = f"{name}-backups"
self.assertTrue(path.isdir(backup_dir))
found_files = listdir(backup_dir)
@@ -88,7 +88,7 @@ class WritableAnnotationTest(unittest.TestCase):
res = self.annotation_put_fbs(fbs)
self.assertEqual(res, json.dumps({"status": "OK"}))
name, ext = path.splitext(self.annotations.output_file)
name, ext = path.splitext(self.annotations.label_output_file)
backup_dir = f"{name}-backups"
self.assertTrue(path.isdir(backup_dir))
found_files = listdir(backup_dir)