Clean up dead/hosted code [zh2310] (#2430)

* Clean up dead/hosted code

* Remove schema conversion tool and related
* Remove cxg references
* Remove locust

* missed a spot

* Remove aws secret manager

* Merge branch 'main' into brodgers/2310/code-cleanup-v1

* cleanup merge
This commit is contained in:
Ben MR
2021-09-17 20:41:12 +00:00
committed by GitHub
parent 69e159916e
commit ef2ab07ca0
35 changed files with 5 additions and 2882 deletions
@@ -98,8 +98,6 @@ class ConfigTests(unittest.TestCase):
lfc_cutoff=0.01,
top_n=10,
environment=None,
aws_secrets_manager_region=None,
aws_secrets_manager_secrets=[],
X_approximate_distribution="auto",
config_file_name="app_config.yml",
):
@@ -151,8 +149,6 @@ class ConfigTests(unittest.TestCase):
)
external_config = self.custom_external_config(
environment=environment,
aws_secrets_manager_region=aws_secrets_manager_region,
aws_secrets_manager_secrets=aws_secrets_manager_secrets,
config_file_name=f"temp_external_config_{random_num}.yml",
)
@@ -197,8 +193,6 @@ class ConfigTests(unittest.TestCase):
def custom_external_config(
self,
environment=None,
aws_secrets_manager_region=None,
aws_secrets_manager_secrets=[],
config_file_name="external_config.yaml",
):
# set to the default if environment is None
@@ -209,7 +203,6 @@ class ConfigTests(unittest.TestCase):
external_config = {
"external": {
"environment": environment,
"aws_secrets_manager": {"region": aws_secrets_manager_region, "secrets": aws_secrets_manager_secrets},
}
}
@@ -1,5 +1,4 @@
import os
from unittest.mock import patch
import requests
@@ -13,7 +12,7 @@ from backend.test.test_server.unit.common.config import ConfigTests
class TestExternalConfig(ConfigTests):
def test_type_convert(self):
# The values from environment variables and aws secrets are returned as strings.
# The values from environment variables are returned as strings.
# These values need to be converted to the proper types.
self.assertEqual(convert_string_to_value("1"), int(1))
@@ -88,129 +87,3 @@ class TestExternalConfig(ConfigTests):
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(config_error.exception.message, "required environment variable 'THIS_ENV_IS_NOT_SET' not set")
@patch("backend.server.common.config.external_config.get_secret_key")
def test_aws_secrets_manager(self, mock_get_secret_key):
mock_get_secret_key.return_value = {
"flask_secret_key": "mock_flask_secret_key",
}
configfile = self.custom_external_config(
aws_secrets_manager_region="us-west-2",
aws_secrets_manager_secrets=[
dict(
name="my_secret",
values=[
dict(key="flask_secret_key", path=["server", "app", "flask_secret_key"], required=True),
],
)
],
config_file_name="secret_external_config.yaml",
)
app_config = AppConfig()
app_config.update_from_config_file(configfile)
app_config.server_config.single_dataset__datapath = f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad"
app_config.complete_config()
self.assertEqual(app_config.server_config.app__flask_secret_key, "mock_flask_secret_key")
@patch("backend.server.common.config.external_config.get_secret_key")
def test_aws_secrets_manager_error(self, mock_get_secret_key):
mock_get_secret_key.return_value = {
"db_uri": "mock_db_uri",
}
# no region
app_config = AppConfig()
app_config.external_config.aws_secrets_manager__region = None
app_config.external_config.aws_secrets_manager__secrets = [
dict(name="secret1", values=[dict(key="key1", required=True, path=["this", "is", "my", "path"])])
]
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(
config_error.exception.message,
"Invalid type for attribute: aws_secrets_manager__region, expected type str, got NoneType",
)
# missing secret name
app_config = AppConfig()
app_config.external_config.aws_secrets_manager__region = "us-west-2"
app_config.external_config.aws_secrets_manager__secrets = [
dict(values=[dict(key="db_uri", required=True, path=["this", "is", "my", "path"])])
]
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(config_error.exception.message, "aws_secrets_manager: 'name' is missing")
# secret name wrong type
app_config = AppConfig()
app_config.external_config.aws_secrets_manager__region = "us-west-2"
app_config.external_config.aws_secrets_manager__secrets = [
dict(name=1, values=[dict(key="db_uri", required=True, path=["this", "is", "my", "path"])])
]
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(config_error.exception.message, "aws_secrets_manager: 'name' must be a string")
# missing values name
app_config = AppConfig()
app_config.external_config.aws_secrets_manager__region = "us-west-2"
app_config.external_config.aws_secrets_manager__secrets = [dict(name="mysecret")]
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(config_error.exception.message, "aws_secrets_manager: 'values' is missing")
# values wrong type
app_config = AppConfig()
app_config.external_config.aws_secrets_manager__region = "us-west-2"
app_config.external_config.aws_secrets_manager__secrets = [
dict(name="mysecret", values=dict(key="db_uri", required=True, path=["this", "is", "my", "path"]))
]
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(config_error.exception.message, "aws_secrets_manager: 'values' must be a list")
# entry missing key
app_config = AppConfig()
app_config.external_config.aws_secrets_manager__region = "us-west-2"
app_config.external_config.aws_secrets_manager__secrets = [
dict(name="mysecret", values=[dict(required=True, path=["this", "is", "my", "path"])])
]
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(config_error.exception.message, "missing 'key' in secret values: mysecret")
# entry required is wrong type
app_config = AppConfig()
app_config.external_config.aws_secrets_manager__region = "us-west-2"
app_config.external_config.aws_secrets_manager__secrets = [
dict(name="mysecret", values=[dict(key="db_uri", required="optional", path=["this", "is", "my", "path"])])
]
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(config_error.exception.message, "wrong type for 'required' in secret values: mysecret")
# entry missing path
app_config = AppConfig()
app_config.external_config.aws_secrets_manager__region = "us-west-2"
app_config.external_config.aws_secrets_manager__secrets = [
dict(name="mysecret", values=[dict(key="db_uri", required=True)])
]
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(config_error.exception.message, "missing 'path' in secret values: mysecret")
# secret missing required key
app_config = AppConfig()
app_config.external_config.aws_secrets_manager__region = "us-west-2"
app_config.external_config.aws_secrets_manager__secrets = [
dict(
name="mysecret",
values=[dict(key="KEY_DOES_NOT_EXIST", required=True, path=["this", "is", "a", "path"])],
)
]
with self.assertRaises(ConfigurationError) as config_error:
app_config.complete_config()
self.assertEqual(config_error.exception.message, "required secret 'mysecret:KEY_DOES_NOT_EXIST' not set")
@@ -1,61 +0,0 @@
import os
import unittest
import pandas as pd
from backend.test import FIXTURES_ROOT
from backend.server.converters.schema import gene_symbol
class TestHGNCSymbolChecker(unittest.TestCase):
def setUp(self):
self.test_hgnc_path = os.path.join(FIXTURES_ROOT, "hgnc_example.txt.gz")
self.hgnc_checker = gene_symbol.HGNCSymbolChecker.from_hgnc_records(self.test_hgnc_path)
def test_symbol_upgrade(self):
self.assertEqual(self.hgnc_checker.upgrade_symbol("SEPT1"), "SEPTIN1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("ADRB2R"), "ADRB2")
self.assertEqual(self.hgnc_checker.upgrade_symbol("BAR"), "ADRB2")
self.assertEqual(self.hgnc_checker.upgrade_symbol("sept1"), "SEPTIN1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("AdRb2R"), "ADRB2")
self.assertEqual(self.hgnc_checker.upgrade_symbol("bar"), "ADRB2")
# Strip off seurat endings when appropriate
self.assertEqual(self.hgnc_checker.upgrade_symbol("SEPT1.1"), "SEPTIN1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("ADRB2-1"), "ADRB2")
# DIFF6 is ambiguous so don't upgrade it
self.assertEqual(self.hgnc_checker.upgrade_symbol("DIFF6"), "DIFF6")
self.assertEqual(self.hgnc_checker.upgrade_symbol("diff6"), "diff6")
# ARG1 is approved
self.assertEqual(self.hgnc_checker.upgrade_symbol("ARG1"), "ARG1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("arg1"), "ARG1")
# HAP1 is both approved and withdrawn
self.assertEqual(self.hgnc_checker.upgrade_symbol("HAP1"), "HAP1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("hap1"), "HAP1")
# Leave unknown symbols alone
self.assertEqual(self.hgnc_checker.upgrade_symbol("NOTASYMBOL"), "NOTASYMBOL")
self.assertEqual(self.hgnc_checker.upgrade_symbol("notasymbol"), "notasymbol")
# Upgrade HGNC ids unless you can't find it
self.assertEqual(self.hgnc_checker.upgrade_symbol("HGNC:286"), "ADRB2")
self.assertEqual(self.hgnc_checker.upgrade_symbol("HGNC:4812"), "HAP1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("HGNC:123456"), "HGNC:123456")
def test_check_symbol(self):
self.assertEqual(self.hgnc_checker.check_symbol("SEPT1"), gene_symbol.SymbolStatus.UPGRADABLE)
self.assertEqual(self.hgnc_checker.check_symbol("DIFF6"), gene_symbol.SymbolStatus.AMBIGUOUS)
self.assertEqual(self.hgnc_checker.check_symbol("NOTASYMBOL"), gene_symbol.SymbolStatus.UNKNOWN)
# HAP1 is one of the approved and withdrawn symbols
self.assertEqual(self.hgnc_checker.check_symbol("HAP1"), gene_symbol.SymbolStatus.APPROVED)
def test_upgrade_index(self):
index = pd.Index(["SEPT1", "DIFF6", "NOTASYMBOL", "bar", "SEPTIN1"])
var_df = pd.DataFrame([[0] * len(index)], index=index)
upgraded_index = gene_symbol.get_upgraded_var_index(var_df, hgnc_path=self.test_hgnc_path)
self.assertEqual(upgraded_index.tolist(), ["SEPTIN1", "DIFF6", "NOTASYMBOL", "ADRB2", "SEPTIN1"])
@@ -1,128 +0,0 @@
import json
import unittest.mock
from backend.server.converters.schema import ontology
class TestOntologyParsing(unittest.TestCase):
def setUp(self):
self.curies = ["UBERON:0002048", "HsapDv:0000174", "NCBITaxon:9606", "EFO:0008995"]
self.names = ["UBERON", "HsapDv", "NCBITaxon", "EFO"]
self.values = ["0002048", "0000174", "9606", "0008995"]
self.iris = [
"http://purl.obolibrary.org/obo/UBERON_0002048",
"http://purl.obolibrary.org/obo/HsapDv_0000174",
"http://purl.obolibrary.org/obo/NCBITaxon_9606",
"http://www.ebi.ac.uk/efo/EFO_0008995",
]
URL_ROOT = "http://www.ebi.ac.uk/ols/api/ontologies/"
self.urls = [
URL_ROOT + "UBERON/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FUBERON_0002048",
URL_ROOT + "HsapDv/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FHsapDv_0000174",
URL_ROOT + "NCBITaxon/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FNCBITaxon_9606",
URL_ROOT + "EFO/terms/http%253A%252F%252Fwww.ebi.ac.uk%252Fefo%252FEFO_0008995",
]
self.responses = {
"UBERON:0002048": {
"iri": "http://purl.obolibrary.org/obo/UBERON_0002048",
"description": ["Respiration organ that develops as an outpocketing of the esophagus."],
"label": "lung",
},
"HsapDv:0000174": {
"iri": "http://purl.obolibrary.org/obo/HsapDv_0000174",
"description": ["Infant stage that refers to an infant who is over 1 and under 2 months old."],
"label": "1-month-old human stage",
},
"NCBITaxon:9606": {
"iri": "http://purl.obolibrary.org/obo/NCBITaxon_9606",
"description": None,
"label": "Homo sapiens",
},
"EFO:0008995": {
"iri": "http://www.ebi.ac.uk/efo/EFO_0008995",
"description": [
(
'10X is a "synthetic long-read" technology and works by capturing a barcoded oligo-coated '
"gel-bead and 0.3x genome copies into a single emulsion droplet, processing the equivalent "
"of 1 million pipetting steps. Successive versions of the 10x chemistry use different "
"barcode locations to improve the sequencing yield and quality of 10x experiments."
)
],
"label": "10X sequencing",
},
}
def test_ontololgy_name(self):
for curie, expected_name in zip(self.curies, self.names):
self.assertEqual(ontology._ontology_name(curie), expected_name)
def test_ontololgy_value(self):
for curie, expected_value in zip(self.curies, self.values):
self.assertEqual(ontology._ontology_value(curie), expected_value)
def test_iri(self):
for curie, expected_iri in zip(self.curies, self.iris):
self.assertEqual(ontology._iri(curie), expected_iri)
def test_ontology_info_url(self):
for curie, expected_url in zip(self.curies, self.urls):
self.assertEqual(ontology._ontology_info_url(curie), expected_url)
def test_empty_ontology_info_url(self):
self.assertEqual(ontology._ontology_info_url(""), "")
class TestOntologyLookup(unittest.TestCase):
def setUp(self):
self.responses = {
"UBERON:0002048": {
"iri": "http://purl.obolibrary.org/obo/UBERON_0002048",
"description": ["Respiration organ that develops as an outpocketing of the esophagus."],
"label": "lung",
},
"HsapDv:0000174": {
"iri": "http://purl.obolibrary.org/obo/HsapDv_0000174",
"description": ["Infant stage that refers to an infant who is over 1 and under 2 months old."],
"label": "1-month-old human stage",
},
"NCBITaxon:9606": {
"iri": "http://purl.obolibrary.org/obo/NCBITaxon_9606",
"description": None,
"label": "Homo sapiens",
},
"EFO:0008995": {
"iri": "http://www.ebi.ac.uk/efo/EFO_0008995",
"description": [
('10X is a "synthetic long-read" technology and works by capturing a barcoded oligo-coated '
'gel-bead and 0.3x genome copies into a single emulsion droplet, processing the equivalent '
'of 1 million pipetting steps. Successive versions of the 10x chemistry use different barcode '
'locations to improve the sequencing yield and quality of 10x experiments.')
],
"label": "10X sequencing",
},
}
self.labels = {
"UBERON:0002048": "lung",
"HsapDv:0000174": "1-month-old human stage",
"NCBITaxon:9606": "Homo sapiens",
"EFO:0008995": "10X sequencing",
}
@unittest.mock.patch("requests.get")
def test_lookup_label(self, mock_get):
for curie, response in self.responses.items():
mock_get.return_value.content = json.dumps(response)
mock_get.return_value.json.return_value = response
mock_get.return_value.status_code = 200
label = ontology.get_ontology_label(curie)
self.assertEqual(label, self.labels[curie])
@@ -1,256 +0,0 @@
import json
import os
import unittest
import unittest.mock
import anndata
import numpy
import pandas as pd
import scanpy as sc
from backend.server.converters.schema import remix
from backend.test import PROJECT_ROOT
class TestApplySchema(unittest.TestCase):
def setUp(self):
self.source_h5ad_path = f"{PROJECT_ROOT}/backend/test/fixtures/pbmc3k-CSC-gz.h5ad"
self.output_h5ad_path = f"{PROJECT_ROOT}/backend/test/fixtures/test_remix.h5ad"
self.config_path = f"{PROJECT_ROOT}/backend/test/fixtures/test_config.yaml"
self.bad_config_path = f"{PROJECT_ROOT}/backend/test/fixtures/test_bad_config.yaml"
def tearDown(self):
try:
os.remove(self.output_h5ad_path)
except OSError:
pass
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
def test_apply_schema(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "test label"
remix.apply_schema(self.source_h5ad_path, self.config_path, self.output_h5ad_path)
new_adata = sc.read_h5ad(self.output_h5ad_path)
self.assertIn("cell_type", new_adata.obs.columns)
self.assertListEqual(["test label"], new_adata.obs["cell_type"].unique().tolist())
self.assertListEqual(
["CL:00001", "CL:00002", "CL:00003", "CL:00004", "CL:00005", "CL:00006", "CL:00007", "CL:00008"],
sorted(new_adata.obs["cell_type_ontology_term_id"].unique().tolist())
)
self.assertIn("version", new_adata.uns_keys())
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
def test_apply_bad_schema(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "test label"
remix.apply_schema(self.source_h5ad_path, self.bad_config_path, self.output_h5ad_path)
new_adata = sc.read_h5ad(self.output_h5ad_path)
# Should refuse to write the version
self.assertNotIn("version", new_adata.uns_keys())
class TestFieldParsing(unittest.TestCase):
def test_is_curie(self):
self.assertTrue(remix.is_curie("EFO:00001"))
self.assertTrue(remix.is_curie("UBERON:123456"))
self.assertTrue(remix.is_curie("HsapDv:0001"))
self.assertFalse(remix.is_curie("UBERON"))
self.assertFalse(remix.is_curie("UBERON:"))
self.assertFalse(remix.is_curie("123456"))
def test_is_ontology_field(self):
self.assertTrue(remix.is_ontology_field("tissue_ontology_term_id"))
self.assertTrue(remix.is_ontology_field("cell_type_ontology_term_id"))
self.assertFalse(remix.is_ontology_field("cell_ontology"))
self.assertFalse(remix.is_ontology_field("method"))
def test_get_label_field_name(self):
self.assertEqual("tissue", remix.get_label_field_name("tissue_ontology_term_id"))
self.assertEqual("cell_type", remix.get_label_field_name("cell_type_ontology_term_id"))
def test_split_suffix(self):
self.assertEqual(("UBERON:1234", " (organoid)"), remix.split_suffix("UBERON:1234 (organoid)"))
self.assertEqual(("UBERON:1234", " (cell culture)"), remix.split_suffix("UBERON:1234 (cell culture)"))
self.assertEqual(("UBERON:1234", ""), remix.split_suffix("UBERON:1234"))
self.assertEqual(("UBERON:1234 (something)", ""), remix.split_suffix("UBERON:1234 (something)"))
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
def test_get_curie_and_label(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "test label"
self.assertEqual(
remix.get_curie_and_label("UBERON:1234"),
("UBERON:1234", "test label")
)
self.assertEqual(
remix.get_curie_and_label("UBERON:1234 (cell culture)"),
("UBERON:1234 (cell culture)", "test label (cell culture)")
)
self.assertEqual(
remix.get_curie_and_label("whatever"),
("", "whatever")
)
class TestManipulateAnndata(unittest.TestCase):
def setUp(self):
self.cell_count = 20
self.gene_count = 200
X = numpy.random.randint(0, 1000, (self.cell_count, self.gene_count))
uns = {"organism": "monkey", "experiment": "monkey experiment"}
obs = pd.DataFrame(
index=[f"Cell{d}" for d in range(self.cell_count)],
columns=["tissue", "CellType"],
data=[["lung", "epithelial"]] * (self.cell_count // 2) + [["lung", "endothelial"]] * (self.cell_count // 2)
)
var = pd.DataFrame(index=[f"SEPT{d}" for d in range(self.gene_count)])
self.adata = anndata.AnnData(X=X, obs=obs, var=var, uns=uns)
def test_safe_add_field(self):
remix.safe_add_field(self.adata.obs, "tissue", ["monkey lung"] * self.cell_count)
self.assertEqual(self.adata.obs["tissue_original"].tolist(), ["lung"] * self.cell_count)
self.assertEqual(self.adata.obs["tissue"].tolist(), ["monkey lung"] * self.cell_count)
remix.safe_add_field(self.adata.uns, "contributors", [{"name": "contributor1"}, {"name": "contributor2"}])
self.assertEqual(
self.adata.uns["contributors"],
json.dumps([{"name": "contributor1"}, {"name": "contributor2"}])
)
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
def test_remix_uns(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "Pan troglodytes"
uns_config = {
"version": {
"corpora_schema_version": "1.0.0",
"corpora_encoding_version": "0.1.0"
},
"organism_ontology_term_id": "NCBITaxon:9598",
"contributors": [
{
"name": "scientist",
"email": "scientist@science.com"
}
]
}
remix.remix_uns(self.adata, uns_config)
self.assertEqual(
sorted(self.adata.uns_keys()),
sorted(["organism_original", "organism", "organism_ontology_term_id",
"contributors", "version", "experiment"])
)
self.assertEqual(self.adata.uns['organism'], "Pan troglodytes")
self.assertEqual(self.adata.uns['organism_original'], "monkey")
self.assertEqual(self.adata.uns['organism_ontology_term_id'], "NCBITaxon:9598")
self.assertEqual(self.adata.uns['contributors'],
json.dumps([{"name": "scientist", "email": "scientist@science.com"}]))
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
def test_remix_obs(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "lung (in a monkey)"
obs_config = {
"tissue_ontology_term_id": {
"tissue": {
"lung": "UBERON:00000"
}
},
"cell_color": {
"CellType": {
"epithelial": "fuschia",
"endothelial": "khaki"
}
},
"sex": "male"
}
remix.remix_obs(self.adata, obs_config)
self.assertEqual(
sorted(self.adata.obs_keys()),
sorted(["tissue", "tissue_ontology_term_id", "tissue_original", "CellType", "cell_color", "sex"])
)
self.assertTrue(all(v == "lung" for v in self.adata.obs.tissue_original))
self.assertTrue(all(v == "UBERON:00000" for v in self.adata.obs.tissue_ontology_term_id))
self.assertTrue(all(v == "lung (in a monkey)" for v in self.adata.obs.tissue))
self.assertTrue(all(v == "male" for v in self.adata.obs.sex))
self.assertTrue(all(v in (("epithelial", "fuschia"), ("endothelial", "khaki"))
for v in zip(self.adata.obs.CellType, self.adata.obs.cell_color)))
class TestFixupGeneSymbols(unittest.TestCase):
def setUp(self):
self.seurat_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/seurat_tutorial.h5ad"
self.seurat_merged_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/seurat_tutorial_merged.h5ad"
self.sctransform_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/sctransform.h5ad"
self.sctransform_merged_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/sctransform_merged.h5ad"
# There's lots of MALAT1, but it doesn't collide with any other names,
# so it shouldn't change during merging.
self.stable_gene = "MALAT1"
def test_fixup_gene_symbols_seurat(self):
if not os.path.isfile(self.seurat_path):
return unittest.skip(
"Skipping gene symbol conversion tests because test h5ads are not present. To create them, "
"run server/test/fixtures/schema_test_data/generate_test_data.sh"
)
original_adata = sc.read_h5ad(self.seurat_path)
merged_adata = sc.read_h5ad(self.seurat_merged_path)
fixup_config = {"X": "log1p", "counts": "raw", "scale.data": "log1p"}
fixed_adata = remix.fixup_gene_symbols(original_adata, fixup_config)
self.assertEqual(
merged_adata.layers["counts"][:, merged_adata.var.index == self.stable_gene].sum(),
fixed_adata.raw.X[:, fixed_adata.var.index == self.stable_gene].sum()
)
self.assertAlmostEqual(
merged_adata.X[:, merged_adata.var.index == self.stable_gene].sum(),
fixed_adata.X[:, fixed_adata.var.index == self.stable_gene].sum()
)
self.assertAlmostEqual(
merged_adata.layers["scale.data"][:, merged_adata.var.index == self.stable_gene].sum(),
fixed_adata.layers["scale.data"][:, fixed_adata.var.index == self.stable_gene].sum()
)
def test_fixup_gene_symbols_sctransform(self):
if not os.path.isfile(self.sctransform_path):
return unittest.skip(
"Skipping gene symbol conversion tests because test h5ads are not present. To create them, "
"run server/test/fixtures/schema_test_data/generate_test_data.sh"
)
original_adata = sc.read_h5ad(self.sctransform_path)
merged_adata = sc.read_h5ad(self.sctransform_merged_path)
fixup_config = {"X": "log1p", "counts": "raw"}
fixed_adata = remix.fixup_gene_symbols(original_adata, fixup_config)
# sctransform does a bunch of stuff, including slightly modifying the
# raw counts. So we can't assert for exact equality the way we do with
# the vanilla seurat tutorial. But, the results should still be very
# close.
merged_raw_stable = merged_adata.layers["counts"][:, merged_adata.var.index == self.stable_gene].sum()
fixed_raw_stable = fixed_adata.raw.X[:, fixed_adata.var.index == self.stable_gene].sum()
self.assertLess(abs(merged_raw_stable - fixed_raw_stable), .001 * merged_raw_stable)
self.assertAlmostEqual(
merged_adata.X[:, merged_adata.var.index == self.stable_gene].sum(),
fixed_adata.X[:, fixed_adata.var.index == self.stable_gene].sum(),
0
)
@@ -1,434 +0,0 @@
import json
import unittest
import pandas as pd
import scanpy as sc
from backend.server.converters.schema import validate
from backend.test import PROJECT_ROOT
class TestFieldValidation(unittest.TestCase):
def test_validate_stringified_list_of_dicts(self):
good = json.dumps([{"a": 1}, {2: "x", "z": "y"}])
not_stringified = [{"a": 1}, {2: "x", "z": "y"}]
not_a_list = json.dumps({"bad": "dict"})
not_json = "oh hey!"
self.assertTrue(validate._validate_stringified_list_of_dicts(good))
self.assertFalse(validate._validate_stringified_list_of_dicts(not_stringified))
self.assertFalse(validate._validate_stringified_list_of_dicts(not_a_list))
self.assertFalse(validate._validate_stringified_list_of_dicts(not_json))
def test_validate_human_readable_string(self):
good = "oh hey!"
curie = "EFO:0001"
ensg = "ENSG000001234"
enst = "ENST000005678"
self.assertTrue(validate._validate_human_readable_string(good))
self.assertFalse(validate._validate_human_readable_string(curie))
self.assertFalse(validate._validate_human_readable_string(ensg))
self.assertFalse(validate._validate_human_readable_string(enst))
def test_validate_curie(self):
self.assertTrue(validate._validate_curie("UBERON:00001", ["UBERON", "EFO"]))
self.assertTrue(validate._validate_curie("HsapDv:00002", ["HsapDv"]))
self.assertFalse(validate._validate_curie("HsapDv:00002", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("EFO:00002 (organoid)", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("EFO:00002 extra", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("UBERON:ABCD", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("Uberon:00002", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("UBERON:", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("UBERON", ["UBERON", "EFO"]))
def test_validate_suffixed_curie(self):
self.assertTrue(validate._validate_suffixed_curie("EFO:00001", ["UBERON", "EFO"]))
self.assertTrue(validate._validate_suffixed_curie("UBERON:00001 (cell culture)", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002 (organoid)", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002(organoid)", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("EFO:00002 extra", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("UBERON:ABCD", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("Uberon:00002", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("UBERON:", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("UBERON", ["UBERON", "EFO"]))
class TestColumnValidation(unittest.TestCase):
def test_validate_unique(self):
unique = pd.DataFrame([["abc", "def"], ["ghi", "jkl"], ["mnop", "qrs"]],
index=["X", "Y", "Z"], columns=["col1", "col2"])
duped = pd.DataFrame([["abc", "def"], ["ghi", "qrs"], ["abc", "qrs"]],
index=["X", "Y", "X"], columns=["col1", "col2"])
schema_def = {"unique": True}
errors = validate._validate_column(unique.index, "index", "unique_df", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(duped.index, "index", "duped_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("is not unique", errors[0])
errors = validate._validate_column(unique["col1"], "col1", "unique_df", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(duped["col1"], "col1", "duped_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("is not unique", errors[0])
schema_def = {"unique": False}
errors = validate._validate_column(duped["col1"], "col1", "duped_df", schema_def)
self.assertFalse(errors)
def test_validate_nullable(self):
non_null = pd.DataFrame([["abc", "def"], ["ghi", "jkl"], ["mnop", "qrs"]],
index=["X", "Y", "Z"], columns=["col1", "col2"])
has_null = pd.DataFrame([["abc", "", None], ["ghi", "jkl", 1], ["mnop", "qrs", 2]],
index=["X", "Y", "Z"], columns=["col1", "col2", "col3"])
schema_def = {"nullable": False}
errors = validate._validate_column(non_null["col1"], "col1", "nonnull_df", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(has_null["col1"], "col1", "hasnull_df", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(has_null["col2"], "col2", "hasnull_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("contains empty values", errors[0])
errors = validate._validate_column(has_null["col3"], "col3", "hasnull_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("contains empty values", errors[0])
schema_def = {"nullable": True}
errors = validate._validate_column(has_null["col2"], "col2", "hasnull_df", schema_def)
self.assertFalse(errors)
def test_human_readable(self):
hr_df = pd.DataFrame(
[["for you, a human", "UBERON:12345", "UBERON:1234 (thundercat)"],
["hope you're well", "bit of lungs", "brain"]],
index=["ENSG00001", "ENSG00002"],
columns=["good", "curie", "suffixed_curie"])
schema_def = {"type": "human-readable string"}
errors = validate._validate_column(hr_df["good"], "good", "hr", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(hr_df["curie"], "curie", "hr", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("non-human-readable", errors[0])
errors = validate._validate_column(hr_df["suffixed_curie"], "suffixed_curie", "hr", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("non-human-readable", errors[0])
errors = validate._validate_column(hr_df.index, "ensg", "hr", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("non-human-readable", errors[0])
def test_curie(self):
curie_df = pd.DataFrame(
[["EFO:00001", "HsapDv:00001 (cell culture)", "EFO:", "MONDO:0001 cell culture"],
["UBERON:00002", "HsapDv:00002 (organoid)", "EFO:12345", "MONDO:0002 (baba yaga)"],
["EFO:0000000005", "HsapDv:000004 (humanzee)", "EFO:000002", "MONDO:0004 (TMNT)"]],
index=["X", "Y", "Z"],
columns=["good", "good_suffix", "bad", "bad_suffix"])
# Good
schema_def = {"type": "curie", "prefixes": ["EFO", "UBERON"]}
errors = validate._validate_column(curie_df["good"], "good", "curie_df", schema_def)
self.assertFalse(errors)
# Good suffix
schema_def = {"type": "suffixed curie", "prefixes": ["HsapDv", "WHATEVER"]}
errors = validate._validate_column(curie_df["good_suffix"], "good_suffix", "curie_df", schema_def)
self.assertFalse(errors)
# Bad prefix
schema_def = {"type": "curie", "prefixes": ["EFO"]}
errors = validate._validate_column(curie_df["good"], "good", "curie_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("invalid ontology", errors[0])
self.assertIn("must be curies from one of these", errors[0])
# Bad curies
schema_def = {"type": "curie", "prefixes": ["EFO"]}
errors = validate._validate_column(curie_df["bad"], "bad", "curie_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("invalid ontology", errors[0])
# Bad suffixes
schema_def = {"type": "suffixed curie", "prefixes": ["EFO"]}
errors = validate._validate_column(curie_df["bad_suffix"], "bad_suffix", "curie_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("invalid ontology", errors[0])
def test_enum(self):
enum_df = pd.DataFrame(
[["abc", "ghi"],
["def", "jkl"]],
index=["X", "Y"],
columns=["col1", "col2"])
# All match
schema_def = {"type": "string", "enum": ["abc", "def", "xyz"]}
errors = validate._validate_column(enum_df["col1"], "col1", "enum_df", schema_def)
self.assertFalse(errors)
# Missing value
schema_def = {"type": "string", "enum": ["abc", "xyz"]}
errors = validate._validate_column(enum_df["col1"], "col1", "enum_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("unpermitted values", errors[0])
class TestDictValidations(unittest.TestCase):
def test_key_presence(self):
schema_def = {"keys": {"abc": None, "def": None}}
dict_ = {"abc": "123", "def": "456"}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertFalse(errors)
# Missing keys are bad
dict_ = {"abc": "123"}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("missing key", errors[0])
# Extra keys are okay
dict_ = {"abc": "123", "def": "456", "xyz": "789"}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertFalse(errors)
# Better not be empty come on
dict_ = {}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 2)
def test_nullable(self):
schema_def = {"keys": {"abc": {"type": "string", "nullable": False},
"def": {"type": "string", "nullable": True}}}
dict_ = {"abc": "xyz", "def": ""}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertFalse(errors)
dict_ = {"abc": "", "def": ""}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("empty value", errors[0])
def test_recurse(self):
schema_def = {
"keys": {
"subdict": {
"type": "dict",
"keys": {
"subdict_key1": None,
"subdict_key2": None
}
},
"ontology": {
"type": "curie",
"prefixes": ["ONTOLOGY"]
},
"blob": {
"type": "stringified list of dicts"
}
}
}
dict_ = {
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
"ontology": "ONTOLOGY:123456",
"blob": json.dumps([{"abc": 123}, {"def": 456}])
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertFalse(errors)
dict_ = {
"subdict": {"subdict_key1": "any"},
"ontology": "ONTOLOGY:123456",
"blob": json.dumps([{"abc": 123}, {"def": 456}])
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("missing key", errors[0])
dict_ = {
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
"ontology": "oh no not an ontology term",
"blob": json.dumps([{"abc": 123}, {"def": 456}])
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("invalid ontology", errors[0])
dict_ = {
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
"ontology": "ONTOLOGY:123456",
"blob": [{"abc": 123}, {"def": 456}]
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("JSON-encoded list of dicts", errors[0])
# Multiple errors
dict_ = {
"subdict": {"subdict_key1": "any"},
"ontology": "oh no not an ontology term",
"blob": json.dumps([{"abc": 123}, {"def": 456}])
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 2)
class TestDataframeValidation(unittest.TestCase):
def test_column_presence(self):
df = pd.DataFrame(
[["abc", "EFO:123"],
["def", "UBERON:456"]],
columns=["hr_string", "ontology"],
index=["X", "Y"]
)
schema_def = {
"columns": {
"hr_string": {"type": "human-readable string"},
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
}
}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertFalse(errors)
schema_def = {
"columns": {
"hr_string": {"type": "human-readable string"},
"another_hr_string": {"type": "human-readable string"},
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
}
}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("missing column", errors[0])
# Extra is okay
df = pd.DataFrame(
[["abc", "EFO:123", "extra"],
["def", "UBERON:456", "extra"]],
columns=["hr_string", "ontology", "extra"],
index=["X", "Y"]
)
schema_def = {
"columns": {
"hr_string": {"type": "human-readable string"},
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
}
}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertFalse(errors)
def test_index(self):
df = pd.DataFrame(
[["abc", "123"],
["def", "456"]],
columns=["col1", "col2"],
index=["ENSG0001", "ENSG0002"]
)
schema_def = {"index": {"unique": True}}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertFalse(errors)
schema_def = {"index": {"type": "human-readable string"}}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("non-human-readable", errors[0])
df = pd.DataFrame(
[["abc", "123"],
["def", "456"]],
columns=["col1", "col2"],
index=["ENSG0001", "ENSG0001"]
)
schema_def = {"index": {"unique": True}}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("is not unique", errors[0])
def test_recurse(self):
df = pd.DataFrame(
[["abc", "HsapDv:0001"],
["EFO:123", "UBERON:456"]],
columns=["hr_string", "ontology"],
index=["X", "Y"]
)
schema_def = {
"columns": {
"hr_string": {"type": "human-readable string"},
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
}
}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertEqual(len(errors), 2)
self.assertEqual(len([e for e in errors if "non-human-readable" in e]), 1)
self.assertEqual(len([e for e in errors if "invalid ontology" in e]), 1)
class TestGetSchema(unittest.TestCase):
def test_get_schema(self):
self.assertIsInstance(validate.get_schema_definition("1.0.0"), dict)
with self.assertRaises(ValueError):
validate.get_schema_definition("10.1.5")
class TestValidate(unittest.TestCase):
def setUp(self):
self.source_h5ad_path = f"{PROJECT_ROOT}/backend/test/fixtures/pbmc3k-CSC-gz.h5ad"
def test_shallow(self):
adata = sc.read_h5ad(self.source_h5ad_path)
self.assertFalse(validate.validate_adata(adata, True))
adata.uns["version"] = {
"corpora_schema_version": "1.0.0",
"corpora_encoding_version": "0.1.0"
}
self.assertTrue(validate.validate_adata(adata, True))
def test_deep(self):
adata = sc.read_h5ad(self.source_h5ad_path)
self.assertFalse(validate.validate_adata(adata, False))
adata.uns["version"] = {
"corpora_schema_version": "1.0.0",
"corpora_encoding_version": "0.1.0"
}
self.assertFalse(validate.validate_adata(adata, False))