Refactor czi_hosted and server into backend directory, pull common code into backend/common, refactor tests (#2102)

* move local_server -> backend/server server-> backend/czi_hosted, pull common code into backend/common update imports, tests and make commands
This commit is contained in:
Madison Dunitz
2021-03-26 00:27:07 -05:00
committed by GitHub
parent e6e358ddc8
commit 78c9d24ed4
425 changed files with 734 additions and 5317 deletions
@@ -0,0 +1,61 @@
import os
import unittest
import pandas as pd
from backend.czi_hosted.converters.schema import gene_symbol
from backend.test import FIXTURES_ROOT
class TestHGNCSymbolChecker(unittest.TestCase):
def setUp(self):
self.test_hgnc_path = os.path.join(FIXTURES_ROOT, "hgnc_example.txt.gz")
self.hgnc_checker = gene_symbol.HGNCSymbolChecker.from_hgnc_records(self.test_hgnc_path)
def test_symbol_upgrade(self):
self.assertEqual(self.hgnc_checker.upgrade_symbol("SEPT1"), "SEPTIN1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("ADRB2R"), "ADRB2")
self.assertEqual(self.hgnc_checker.upgrade_symbol("BAR"), "ADRB2")
self.assertEqual(self.hgnc_checker.upgrade_symbol("sept1"), "SEPTIN1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("AdRb2R"), "ADRB2")
self.assertEqual(self.hgnc_checker.upgrade_symbol("bar"), "ADRB2")
# Strip off seurat endings when appropriate
self.assertEqual(self.hgnc_checker.upgrade_symbol("SEPT1.1"), "SEPTIN1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("ADRB2-1"), "ADRB2")
# DIFF6 is ambiguous so don't upgrade it
self.assertEqual(self.hgnc_checker.upgrade_symbol("DIFF6"), "DIFF6")
self.assertEqual(self.hgnc_checker.upgrade_symbol("diff6"), "diff6")
# ARG1 is approved
self.assertEqual(self.hgnc_checker.upgrade_symbol("ARG1"), "ARG1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("arg1"), "ARG1")
# HAP1 is both approved and withdrawn
self.assertEqual(self.hgnc_checker.upgrade_symbol("HAP1"), "HAP1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("hap1"), "HAP1")
# Leave unknown symbols alone
self.assertEqual(self.hgnc_checker.upgrade_symbol("NOTASYMBOL"), "NOTASYMBOL")
self.assertEqual(self.hgnc_checker.upgrade_symbol("notasymbol"), "notasymbol")
# Upgrade HGNC ids unless you can't find it
self.assertEqual(self.hgnc_checker.upgrade_symbol("HGNC:286"), "ADRB2")
self.assertEqual(self.hgnc_checker.upgrade_symbol("HGNC:4812"), "HAP1")
self.assertEqual(self.hgnc_checker.upgrade_symbol("HGNC:123456"), "HGNC:123456")
def test_check_symbol(self):
self.assertEqual(self.hgnc_checker.check_symbol("SEPT1"), gene_symbol.SymbolStatus.UPGRADABLE)
self.assertEqual(self.hgnc_checker.check_symbol("DIFF6"), gene_symbol.SymbolStatus.AMBIGUOUS)
self.assertEqual(self.hgnc_checker.check_symbol("NOTASYMBOL"), gene_symbol.SymbolStatus.UNKNOWN)
# HAP1 is one of the approved and withdrawn symbols
self.assertEqual(self.hgnc_checker.check_symbol("HAP1"), gene_symbol.SymbolStatus.APPROVED)
def test_upgrade_index(self):
index = pd.Index(["SEPT1", "DIFF6", "NOTASYMBOL", "bar", "SEPTIN1"])
var_df = pd.DataFrame([[0] * len(index)], index=index)
upgraded_index = gene_symbol.get_upgraded_var_index(var_df, hgnc_path=self.test_hgnc_path)
self.assertEqual(upgraded_index.tolist(), ["SEPTIN1", "DIFF6", "NOTASYMBOL", "ADRB2", "SEPTIN1"])
@@ -0,0 +1,128 @@
import json
import unittest.mock
from backend.czi_hosted.converters.schema import ontology
class TestOntologyParsing(unittest.TestCase):
def setUp(self):
self.curies = ["UBERON:0002048", "HsapDv:0000174", "NCBITaxon:9606", "EFO:0008995"]
self.names = ["UBERON", "HsapDv", "NCBITaxon", "EFO"]
self.values = ["0002048", "0000174", "9606", "0008995"]
self.iris = [
"http://purl.obolibrary.org/obo/UBERON_0002048",
"http://purl.obolibrary.org/obo/HsapDv_0000174",
"http://purl.obolibrary.org/obo/NCBITaxon_9606",
"http://www.ebi.ac.uk/efo/EFO_0008995",
]
URL_ROOT = "http://www.ebi.ac.uk/ols/api/ontologies/"
self.urls = [
URL_ROOT + "UBERON/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FUBERON_0002048",
URL_ROOT + "HsapDv/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FHsapDv_0000174",
URL_ROOT + "NCBITaxon/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FNCBITaxon_9606",
URL_ROOT + "EFO/terms/http%253A%252F%252Fwww.ebi.ac.uk%252Fefo%252FEFO_0008995",
]
self.responses = {
"UBERON:0002048": {
"iri": "http://purl.obolibrary.org/obo/UBERON_0002048",
"description": ["Respiration organ that develops as an outpocketing of the esophagus."],
"label": "lung",
},
"HsapDv:0000174": {
"iri": "http://purl.obolibrary.org/obo/HsapDv_0000174",
"description": ["Infant stage that refers to an infant who is over 1 and under 2 months old."],
"label": "1-month-old human stage",
},
"NCBITaxon:9606": {
"iri": "http://purl.obolibrary.org/obo/NCBITaxon_9606",
"description": None,
"label": "Homo sapiens",
},
"EFO:0008995": {
"iri": "http://www.ebi.ac.uk/efo/EFO_0008995",
"description": [
(
'10X is a "synthetic long-read" technology and works by capturing a barcoded oligo-coated '
"gel-bead and 0.3x genome copies into a single emulsion droplet, processing the equivalent "
"of 1 million pipetting steps. Successive versions of the 10x chemistry use different "
"barcode locations to improve the sequencing yield and quality of 10x experiments."
)
],
"label": "10X sequencing",
},
}
def test_ontololgy_name(self):
for curie, expected_name in zip(self.curies, self.names):
self.assertEqual(ontology._ontology_name(curie), expected_name)
def test_ontololgy_value(self):
for curie, expected_value in zip(self.curies, self.values):
self.assertEqual(ontology._ontology_value(curie), expected_value)
def test_iri(self):
for curie, expected_iri in zip(self.curies, self.iris):
self.assertEqual(ontology._iri(curie), expected_iri)
def test_ontology_info_url(self):
for curie, expected_url in zip(self.curies, self.urls):
self.assertEqual(ontology._ontology_info_url(curie), expected_url)
def test_empty_ontology_info_url(self):
self.assertEqual(ontology._ontology_info_url(""), "")
class TestOntologyLookup(unittest.TestCase):
def setUp(self):
self.responses = {
"UBERON:0002048": {
"iri": "http://purl.obolibrary.org/obo/UBERON_0002048",
"description": ["Respiration organ that develops as an outpocketing of the esophagus."],
"label": "lung",
},
"HsapDv:0000174": {
"iri": "http://purl.obolibrary.org/obo/HsapDv_0000174",
"description": ["Infant stage that refers to an infant who is over 1 and under 2 months old."],
"label": "1-month-old human stage",
},
"NCBITaxon:9606": {
"iri": "http://purl.obolibrary.org/obo/NCBITaxon_9606",
"description": None,
"label": "Homo sapiens",
},
"EFO:0008995": {
"iri": "http://www.ebi.ac.uk/efo/EFO_0008995",
"description": [
('10X is a "synthetic long-read" technology and works by capturing a barcoded oligo-coated '
'gel-bead and 0.3x genome copies into a single emulsion droplet, processing the equivalent '
'of 1 million pipetting steps. Successive versions of the 10x chemistry use different barcode '
'locations to improve the sequencing yield and quality of 10x experiments.')
],
"label": "10X sequencing",
},
}
self.labels = {
"UBERON:0002048": "lung",
"HsapDv:0000174": "1-month-old human stage",
"NCBITaxon:9606": "Homo sapiens",
"EFO:0008995": "10X sequencing",
}
@unittest.mock.patch("requests.get")
def test_lookup_label(self, mock_get):
for curie, response in self.responses.items():
mock_get.return_value.content = json.dumps(response)
mock_get.return_value.json.return_value = response
mock_get.return_value.status_code = 200
label = ontology.get_ontology_label(curie)
self.assertEqual(label, self.labels[curie])
@@ -0,0 +1,257 @@
import json
import os
import unittest
import unittest.mock
import anndata
import numpy
import pandas as pd
import scanpy as sc
from backend.czi_hosted.converters.schema import remix
from backend.test import PROJECT_ROOT, FIXTURES_ROOT
class TestApplySchema(unittest.TestCase):
def setUp(self):
self.source_h5ad_path = f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad"
self.output_h5ad_path = f"{FIXTURES_ROOT}/test_remix.h5ad"
self.config_path = f"{FIXTURES_ROOT}/test_config.yaml"
self.bad_config_path = f"{FIXTURES_ROOT}/test_bad_config.yaml"
def tearDown(self):
try:
os.remove(self.output_h5ad_path)
except OSError:
pass
@unittest.mock.patch("backend.czi_hosted.converters.schema.ontology.get_ontology_label")
def test_apply_schema(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "test label"
remix.apply_schema(self.source_h5ad_path, self.config_path, self.output_h5ad_path)
new_adata = sc.read_h5ad(self.output_h5ad_path)
self.assertIn("cell_type", new_adata.obs.columns)
self.assertListEqual(["test label"], new_adata.obs["cell_type"].unique().tolist())
self.assertListEqual(
["CL:00001", "CL:00002", "CL:00003", "CL:00004", "CL:00005", "CL:00006", "CL:00007", "CL:00008"],
sorted(new_adata.obs["cell_type_ontology_term_id"].unique().tolist())
)
self.assertIn("version", new_adata.uns_keys())
@unittest.mock.patch("backend.czi_hosted.converters.schema.ontology.get_ontology_label")
def test_apply_bad_schema(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "test label"
remix.apply_schema(self.source_h5ad_path, self.bad_config_path, self.output_h5ad_path)
new_adata = sc.read_h5ad(self.output_h5ad_path)
# Should refuse to write the version
self.assertNotIn("version", new_adata.uns_keys())
class TestFieldParsing(unittest.TestCase):
def test_is_curie(self):
self.assertTrue(remix.is_curie("EFO:00001"))
self.assertTrue(remix.is_curie("UBERON:123456"))
self.assertTrue(remix.is_curie("HsapDv:0001"))
self.assertFalse(remix.is_curie("UBERON"))
self.assertFalse(remix.is_curie("UBERON:"))
self.assertFalse(remix.is_curie("123456"))
def test_is_ontology_field(self):
self.assertTrue(remix.is_ontology_field("tissue_ontology_term_id"))
self.assertTrue(remix.is_ontology_field("cell_type_ontology_term_id"))
self.assertFalse(remix.is_ontology_field("cell_ontology"))
self.assertFalse(remix.is_ontology_field("method"))
def test_get_label_field_name(self):
self.assertEqual("tissue", remix.get_label_field_name("tissue_ontology_term_id"))
self.assertEqual("cell_type", remix.get_label_field_name("cell_type_ontology_term_id"))
def test_split_suffix(self):
self.assertEqual(("UBERON:1234", " (organoid)"), remix.split_suffix("UBERON:1234 (organoid)"))
self.assertEqual(("UBERON:1234", " (cell culture)"), remix.split_suffix("UBERON:1234 (cell culture)"))
self.assertEqual(("UBERON:1234", ""), remix.split_suffix("UBERON:1234"))
self.assertEqual(("UBERON:1234 (something)", ""), remix.split_suffix("UBERON:1234 (something)"))
@unittest.mock.patch("backend.czi_hosted.converters.schema.ontology.get_ontology_label")
def test_get_curie_and_label(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "test label"
self.assertEqual(
remix.get_curie_and_label("UBERON:1234"),
("UBERON:1234", "test label")
)
self.assertEqual(
remix.get_curie_and_label("UBERON:1234 (cell culture)"),
("UBERON:1234 (cell culture)", "test label (cell culture)")
)
self.assertEqual(
remix.get_curie_and_label("whatever"),
("", "whatever")
)
class TestManipulateAnndata(unittest.TestCase):
def setUp(self):
self.cell_count = 20
self.gene_count = 200
X = numpy.random.randint(0, 1000, (self.cell_count, self.gene_count))
uns = {"organism": "monkey", "experiment": "monkey experiment"}
obs = pd.DataFrame(
index=[f"Cell{d}" for d in range(self.cell_count)],
columns=["tissue", "CellType"],
data=[["lung", "epithelial"]] * (self.cell_count // 2) + [["lung", "endothelial"]] * (self.cell_count // 2)
)
var = pd.DataFrame(index=[f"SEPT{d}" for d in range(self.gene_count)])
self.adata = anndata.AnnData(X=X, obs=obs, var=var, uns=uns)
def test_safe_add_field(self):
remix.safe_add_field(self.adata.obs, "tissue", ["monkey lung"] * self.cell_count)
self.assertEqual(self.adata.obs["tissue_original"].tolist(), ["lung"] * self.cell_count)
self.assertEqual(self.adata.obs["tissue"].tolist(), ["monkey lung"] * self.cell_count)
remix.safe_add_field(self.adata.uns, "contributors", [{"name": "contributor1"}, {"name": "contributor2"}])
self.assertEqual(
self.adata.uns["contributors"],
json.dumps([{"name": "contributor1"}, {"name": "contributor2"}])
)
@unittest.mock.patch("backend.czi_hosted.converters.schema.ontology.get_ontology_label")
def test_remix_uns(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "Pan troglodytes"
uns_config = {
"version": {
"corpora_schema_version": "1.0.0",
"corpora_encoding_version": "0.1.0"
},
"organism_ontology_term_id": "NCBITaxon:9598",
"contributors": [
{
"name": "scientist",
"email": "scientist@science.com"
}
]
}
remix.remix_uns(self.adata, uns_config)
self.assertEqual(
sorted(self.adata.uns_keys()),
sorted(["organism_original", "organism", "organism_ontology_term_id",
"contributors", "version", "experiment"])
)
self.assertEqual(self.adata.uns['organism'], "Pan troglodytes")
self.assertEqual(self.adata.uns['organism_original'], "monkey")
self.assertEqual(self.adata.uns['organism_ontology_term_id'], "NCBITaxon:9598")
self.assertEqual(self.adata.uns['contributors'],
json.dumps([{"name": "scientist", "email": "scientist@science.com"}]))
@unittest.mock.patch("backend.czi_hosted.converters.schema.ontology.get_ontology_label")
def test_remix_obs(self, mock_get_ontology_label):
mock_get_ontology_label.return_value = "lung (in a monkey)"
obs_config = {
"tissue_ontology_term_id": {
"tissue": {
"lung": "UBERON:00000"
}
},
"cell_color": {
"CellType": {
"epithelial": "fuschia",
"endothelial": "khaki"
}
},
"sex": "male"
}
remix.remix_obs(self.adata, obs_config)
self.assertEqual(
sorted(self.adata.obs_keys()),
sorted(["tissue", "tissue_ontology_term_id", "tissue_original", "CellType", "cell_color", "sex"])
)
self.assertTrue(all(v == "lung" for v in self.adata.obs.tissue_original))
self.assertTrue(all(v == "UBERON:00000" for v in self.adata.obs.tissue_ontology_term_id))
self.assertTrue(all(v == "lung (in a monkey)" for v in self.adata.obs.tissue))
self.assertTrue(all(v == "male" for v in self.adata.obs.sex))
self.assertTrue(all(v in (("epithelial", "fuschia"), ("endothelial", "khaki"))
for v in zip(self.adata.obs.CellType, self.adata.obs.cell_color)))
class TestFixupGeneSymbols(unittest.TestCase):
def setUp(self):
self.seurat_path = f"{PROJECT_ROOT}/czi_hosted/test/fixtures/schema_test_data/seurat_tutorial.h5ad"
self.seurat_merged_path = f"{PROJECT_ROOT}/czi_hosted/test/fixtures/schema_test_data/seurat_tutorial_merged.h5ad"
self.sctransform_path = f"{PROJECT_ROOT}/czi_hosted/test/fixtures/schema_test_data/sctransform.h5ad"
self.sctransform_merged_path = f"{PROJECT_ROOT}/czi_hosted/test/fixtures/schema_test_data/sctransform_merged.h5ad"
# There's lots of MALAT1, but it doesn't collide with any other names,
# so it shouldn't change during merging.
self.stable_gene = "MALAT1"
def test_fixup_gene_symbols_seurat(self):
if not os.path.isfile(self.seurat_path):
return unittest.skip(
"Skipping gene symbol conversion tests because test h5ads are not present. To create them, "
"run czi_hosted/test/fixtures/schema_test_data/generate_test_data.sh"
)
original_adata = sc.read_h5ad(self.seurat_path)
merged_adata = sc.read_h5ad(self.seurat_merged_path)
fixup_config = {"X": "log1p", "counts": "raw", "scale.data": "log1p"}
fixed_adata = remix.fixup_gene_symbols(original_adata, fixup_config)
self.assertEqual(
merged_adata.layers["counts"][:, merged_adata.var.index == self.stable_gene].sum(),
fixed_adata.raw.X[:, fixed_adata.var.index == self.stable_gene].sum()
)
self.assertAlmostEqual(
merged_adata.X[:, merged_adata.var.index == self.stable_gene].sum(),
fixed_adata.X[:, fixed_adata.var.index == self.stable_gene].sum()
)
self.assertAlmostEqual(
merged_adata.layers["scale.data"][:, merged_adata.var.index == self.stable_gene].sum(),
fixed_adata.layers["scale.data"][:, fixed_adata.var.index == self.stable_gene].sum()
)
def test_fixup_gene_symbols_sctransform(self):
if not os.path.isfile(self.sctransform_path):
return unittest.skip(
"Skipping gene symbol conversion tests because test h5ads are not present. To create them, "
"run czi_hosted/test/fixtures/schema_test_data/generate_test_data.sh"
)
original_adata = sc.read_h5ad(self.sctransform_path)
merged_adata = sc.read_h5ad(self.sctransform_merged_path)
fixup_config = {"X": "log1p", "counts": "raw"}
fixed_adata = remix.fixup_gene_symbols(original_adata, fixup_config)
# sctransform does a bunch of stuff, including slightly modifying the
# raw counts. So we can't assert for exact equality the way we do with
# the vanilla seurat tutorial. But, the results should still be very
# close.
merged_raw_stable = merged_adata.layers["counts"][:, merged_adata.var.index == self.stable_gene].sum()
fixed_raw_stable = fixed_adata.raw.X[:, fixed_adata.var.index == self.stable_gene].sum()
self.assertLess(abs(merged_raw_stable - fixed_raw_stable), .001 * merged_raw_stable)
self.assertAlmostEqual(
merged_adata.X[:, merged_adata.var.index == self.stable_gene].sum(),
fixed_adata.X[:, fixed_adata.var.index == self.stable_gene].sum(),
0
)
@@ -0,0 +1,434 @@
import json
import unittest
import pandas as pd
import scanpy as sc
from backend.czi_hosted.converters.schema import validate
from backend.test import PROJECT_ROOT
class TestFieldValidation(unittest.TestCase):
def test_validate_stringified_list_of_dicts(self):
good = json.dumps([{"a": 1}, {2: "x", "z": "y"}])
not_stringified = [{"a": 1}, {2: "x", "z": "y"}]
not_a_list = json.dumps({"bad": "dict"})
not_json = "oh hey!"
self.assertTrue(validate._validate_stringified_list_of_dicts(good))
self.assertFalse(validate._validate_stringified_list_of_dicts(not_stringified))
self.assertFalse(validate._validate_stringified_list_of_dicts(not_a_list))
self.assertFalse(validate._validate_stringified_list_of_dicts(not_json))
def test_validate_human_readable_string(self):
good = "oh hey!"
curie = "EFO:0001"
ensg = "ENSG000001234"
enst = "ENST000005678"
self.assertTrue(validate._validate_human_readable_string(good))
self.assertFalse(validate._validate_human_readable_string(curie))
self.assertFalse(validate._validate_human_readable_string(ensg))
self.assertFalse(validate._validate_human_readable_string(enst))
def test_validate_curie(self):
self.assertTrue(validate._validate_curie("UBERON:00001", ["UBERON", "EFO"]))
self.assertTrue(validate._validate_curie("HsapDv:00002", ["HsapDv"]))
self.assertFalse(validate._validate_curie("HsapDv:00002", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("EFO:00002 (organoid)", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("EFO:00002 extra", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("UBERON:ABCD", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("Uberon:00002", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("UBERON:", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_curie("UBERON", ["UBERON", "EFO"]))
def test_validate_suffixed_curie(self):
self.assertTrue(validate._validate_suffixed_curie("EFO:00001", ["UBERON", "EFO"]))
self.assertTrue(validate._validate_suffixed_curie("UBERON:00001 (cell culture)", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002 (organoid)", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002(organoid)", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("EFO:00002 extra", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("UBERON:ABCD", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("Uberon:00002", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("UBERON:", ["UBERON", "EFO"]))
self.assertFalse(validate._validate_suffixed_curie("UBERON", ["UBERON", "EFO"]))
class TestColumnValidation(unittest.TestCase):
def test_validate_unique(self):
unique = pd.DataFrame([["abc", "def"], ["ghi", "jkl"], ["mnop", "qrs"]],
index=["X", "Y", "Z"], columns=["col1", "col2"])
duped = pd.DataFrame([["abc", "def"], ["ghi", "qrs"], ["abc", "qrs"]],
index=["X", "Y", "X"], columns=["col1", "col2"])
schema_def = {"unique": True}
errors = validate._validate_column(unique.index, "index", "unique_df", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(duped.index, "index", "duped_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("is not unique", errors[0])
errors = validate._validate_column(unique["col1"], "col1", "unique_df", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(duped["col1"], "col1", "duped_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("is not unique", errors[0])
schema_def = {"unique": False}
errors = validate._validate_column(duped["col1"], "col1", "duped_df", schema_def)
self.assertFalse(errors)
def test_validate_nullable(self):
non_null = pd.DataFrame([["abc", "def"], ["ghi", "jkl"], ["mnop", "qrs"]],
index=["X", "Y", "Z"], columns=["col1", "col2"])
has_null = pd.DataFrame([["abc", "", None], ["ghi", "jkl", 1], ["mnop", "qrs", 2]],
index=["X", "Y", "Z"], columns=["col1", "col2", "col3"])
schema_def = {"nullable": False}
errors = validate._validate_column(non_null["col1"], "col1", "nonnull_df", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(has_null["col1"], "col1", "hasnull_df", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(has_null["col2"], "col2", "hasnull_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("contains empty values", errors[0])
errors = validate._validate_column(has_null["col3"], "col3", "hasnull_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("contains empty values", errors[0])
schema_def = {"nullable": True}
errors = validate._validate_column(has_null["col2"], "col2", "hasnull_df", schema_def)
self.assertFalse(errors)
def test_human_readable(self):
hr_df = pd.DataFrame(
[["for you, a human", "UBERON:12345", "UBERON:1234 (thundercat)"],
["hope you're well", "bit of lungs", "brain"]],
index=["ENSG00001", "ENSG00002"],
columns=["good", "curie", "suffixed_curie"])
schema_def = {"type": "human-readable string"}
errors = validate._validate_column(hr_df["good"], "good", "hr", schema_def)
self.assertFalse(errors)
errors = validate._validate_column(hr_df["curie"], "curie", "hr", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("non-human-readable", errors[0])
errors = validate._validate_column(hr_df["suffixed_curie"], "suffixed_curie", "hr", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("non-human-readable", errors[0])
errors = validate._validate_column(hr_df.index, "ensg", "hr", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("non-human-readable", errors[0])
def test_curie(self):
curie_df = pd.DataFrame(
[["EFO:00001", "HsapDv:00001 (cell culture)", "EFO:", "MONDO:0001 cell culture"],
["UBERON:00002", "HsapDv:00002 (organoid)", "EFO:12345", "MONDO:0002 (baba yaga)"],
["EFO:0000000005", "HsapDv:000004 (humanzee)", "EFO:000002", "MONDO:0004 (TMNT)"]],
index=["X", "Y", "Z"],
columns=["good", "good_suffix", "bad", "bad_suffix"])
# Good
schema_def = {"type": "curie", "prefixes": ["EFO", "UBERON"]}
errors = validate._validate_column(curie_df["good"], "good", "curie_df", schema_def)
self.assertFalse(errors)
# Good suffix
schema_def = {"type": "suffixed curie", "prefixes": ["HsapDv", "WHATEVER"]}
errors = validate._validate_column(curie_df["good_suffix"], "good_suffix", "curie_df", schema_def)
self.assertFalse(errors)
# Bad prefix
schema_def = {"type": "curie", "prefixes": ["EFO"]}
errors = validate._validate_column(curie_df["good"], "good", "curie_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("invalid ontology", errors[0])
self.assertIn("must be curies from one of these", errors[0])
# Bad curies
schema_def = {"type": "curie", "prefixes": ["EFO"]}
errors = validate._validate_column(curie_df["bad"], "bad", "curie_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("invalid ontology", errors[0])
# Bad suffixes
schema_def = {"type": "suffixed curie", "prefixes": ["EFO"]}
errors = validate._validate_column(curie_df["bad_suffix"], "bad_suffix", "curie_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("invalid ontology", errors[0])
def test_enum(self):
enum_df = pd.DataFrame(
[["abc", "ghi"],
["def", "jkl"]],
index=["X", "Y"],
columns=["col1", "col2"])
# All match
schema_def = {"type": "string", "enum": ["abc", "def", "xyz"]}
errors = validate._validate_column(enum_df["col1"], "col1", "enum_df", schema_def)
self.assertFalse(errors)
# Missing value
schema_def = {"type": "string", "enum": ["abc", "xyz"]}
errors = validate._validate_column(enum_df["col1"], "col1", "enum_df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("unpermitted values", errors[0])
class TestDictValidations(unittest.TestCase):
def test_key_presence(self):
schema_def = {"keys": {"abc": None, "def": None}}
dict_ = {"abc": "123", "def": "456"}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertFalse(errors)
# Missing keys are bad
dict_ = {"abc": "123"}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("missing key", errors[0])
# Extra keys are okay
dict_ = {"abc": "123", "def": "456", "xyz": "789"}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertFalse(errors)
# Better not be empty come on
dict_ = {}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 2)
def test_nullable(self):
schema_def = {"keys": {"abc": {"type": "string", "nullable": False},
"def": {"type": "string", "nullable": True}}}
dict_ = {"abc": "xyz", "def": ""}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertFalse(errors)
dict_ = {"abc": "", "def": ""}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("empty value", errors[0])
def test_recurse(self):
schema_def = {
"keys": {
"subdict": {
"type": "dict",
"keys": {
"subdict_key1": None,
"subdict_key2": None
}
},
"ontology": {
"type": "curie",
"prefixes": ["ONTOLOGY"]
},
"blob": {
"type": "stringified list of dicts"
}
}
}
dict_ = {
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
"ontology": "ONTOLOGY:123456",
"blob": json.dumps([{"abc": 123}, {"def": 456}])
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertFalse(errors)
dict_ = {
"subdict": {"subdict_key1": "any"},
"ontology": "ONTOLOGY:123456",
"blob": json.dumps([{"abc": 123}, {"def": 456}])
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("missing key", errors[0])
dict_ = {
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
"ontology": "oh no not an ontology term",
"blob": json.dumps([{"abc": 123}, {"def": 456}])
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("invalid ontology", errors[0])
dict_ = {
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
"ontology": "ONTOLOGY:123456",
"blob": [{"abc": 123}, {"def": 456}]
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("JSON-encoded list of dicts", errors[0])
# Multiple errors
dict_ = {
"subdict": {"subdict_key1": "any"},
"ontology": "oh no not an ontology term",
"blob": json.dumps([{"abc": 123}, {"def": 456}])
}
errors = validate._validate_dict(dict_, "d", schema_def)
self.assertEqual(len(errors), 2)
class TestDataframeValidation(unittest.TestCase):
def test_column_presence(self):
df = pd.DataFrame(
[["abc", "EFO:123"],
["def", "UBERON:456"]],
columns=["hr_string", "ontology"],
index=["X", "Y"]
)
schema_def = {
"columns": {
"hr_string": {"type": "human-readable string"},
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
}
}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertFalse(errors)
schema_def = {
"columns": {
"hr_string": {"type": "human-readable string"},
"another_hr_string": {"type": "human-readable string"},
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
}
}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("missing column", errors[0])
# Extra is okay
df = pd.DataFrame(
[["abc", "EFO:123", "extra"],
["def", "UBERON:456", "extra"]],
columns=["hr_string", "ontology", "extra"],
index=["X", "Y"]
)
schema_def = {
"columns": {
"hr_string": {"type": "human-readable string"},
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
}
}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertFalse(errors)
def test_index(self):
df = pd.DataFrame(
[["abc", "123"],
["def", "456"]],
columns=["col1", "col2"],
index=["ENSG0001", "ENSG0002"]
)
schema_def = {"index": {"unique": True}}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertFalse(errors)
schema_def = {"index": {"type": "human-readable string"}}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("non-human-readable", errors[0])
df = pd.DataFrame(
[["abc", "123"],
["def", "456"]],
columns=["col1", "col2"],
index=["ENSG0001", "ENSG0001"]
)
schema_def = {"index": {"unique": True}}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertEqual(len(errors), 1)
self.assertIn("is not unique", errors[0])
def test_recurse(self):
df = pd.DataFrame(
[["abc", "HsapDv:0001"],
["EFO:123", "UBERON:456"]],
columns=["hr_string", "ontology"],
index=["X", "Y"]
)
schema_def = {
"columns": {
"hr_string": {"type": "human-readable string"},
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
}
}
errors = validate._validate_dataframe(df, "df", schema_def)
self.assertEqual(len(errors), 2)
self.assertEqual(len([e for e in errors if "non-human-readable" in e]), 1)
self.assertEqual(len([e for e in errors if "invalid ontology" in e]), 1)
class TestGetSchema(unittest.TestCase):
def test_get_schema(self):
self.assertIsInstance(validate.get_schema_definition("1.0.0"), dict)
with self.assertRaises(ValueError):
validate.get_schema_definition("10.1.5")
class TestValidate(unittest.TestCase):
def setUp(self):
self.source_h5ad_path = f"{PROJECT_ROOT}/backend/test/fixtures/pbmc3k-CSC-gz.h5ad"
def test_shallow(self):
adata = sc.read_h5ad(self.source_h5ad_path)
self.assertFalse(validate.validate_adata(adata, True))
adata.uns["version"] = {
"corpora_schema_version": "1.0.0",
"corpora_encoding_version": "0.1.0"
}
self.assertTrue(validate.validate_adata(adata, True))
def test_deep(self):
adata = sc.read_h5ad(self.source_h5ad_path)
self.assertFalse(validate.validate_adata(adata, False))
adata.uns["version"] = {
"corpora_schema_version": "1.0.0",
"corpora_encoding_version": "0.1.0"
}
self.assertFalse(validate.validate_adata(adata, False))
@@ -0,0 +1,268 @@
import json
import unittest
from glob import glob
from os import remove, path
from shutil import rmtree
from uuid import uuid4
import anndata
import numpy as np
from pandas import Series, DataFrame
from backend.czi_hosted.common.corpora import CorporaConstants
from backend.czi_hosted.converters.h5ad_data_file import H5ADDataFile
from backend.test import PROJECT_ROOT
class TestH5ADDataFile(unittest.TestCase):
def setUp(self):
self.sample_anndata = self._create_sample_anndata_dataset()
self.sample_h5ad_filename = self._write_anndata_to_file(self.sample_anndata)
self.sample_output_directory = path.splitext(self.sample_h5ad_filename)[0] + ".cxg"
def tearDown(self):
if self.sample_h5ad_filename:
remove(self.sample_h5ad_filename)
if path.isdir(self.sample_output_directory):
rmtree(self.sample_output_directory)
def test__create_h5ad_data_file__non_h5ad_raises_exception(self):
non_h5ad_filename = "my_fancy_dataset.csv"
with self.assertRaises(Exception) as exception_context:
H5ADDataFile(non_h5ad_filename)
self.assertIn("File must be an H5AD", str(exception_context.exception))
def test__create_h5ad_data_file__assert_warning_outputted_if_dataset_title_or_about_given(self):
with self.assertLogs(level="WARN") as logger:
H5ADDataFile(
self.sample_h5ad_filename,
dataset_title="My Awesome Dataset",
dataset_about="http://www.awesomedataset.com",
use_corpora_schema=False,
)
self.assertIn("will override any metadata that is extracted", logger.output[0])
def test__create_h5ad_data_file__reads_anndata_successfully(self):
h5ad_file = H5ADDataFile(self.sample_h5ad_filename, use_corpora_schema=False)
self.assertTrue((h5ad_file.anndata.X == self.sample_anndata.X).all())
self.assertEqual(
h5ad_file.anndata.obs.sort_index(inplace=True), self.sample_anndata.obs.sort_index(inplace=True)
)
self.assertEqual(
h5ad_file.anndata.var.sort_index(inplace=True), self.sample_anndata.var.sort_index(inplace=True)
)
for key in h5ad_file.anndata.obsm.keys():
self.assertIn(key, self.sample_anndata.obsm.keys())
self.assertTrue((h5ad_file.anndata.obsm[key] == self.sample_anndata.obsm[key]).all())
for key in self.sample_anndata.obsm.keys():
self.assertIn(key, h5ad_file.anndata.obsm.keys())
self.assertTrue((h5ad_file.anndata.obsm[key] == self.sample_anndata.obsm[key]).all())
def test__create_h5ad_data_file__copies_index_of_obs_and_var_to_column(self):
h5ad_file = H5ADDataFile(self.sample_h5ad_filename, use_corpora_schema=False)
# The automatic name chosen for the index should be "name_0"
self.assertNotIn("name_0", self.sample_anndata.obs.columns)
self.assertIn("name_0", h5ad_file.obs.columns)
self.assertNotIn("name_0", self.sample_anndata.var.columns)
self.assertIn("name_0", h5ad_file.var.columns)
def test__create_h5ad_data_file__no_copy_if_obs_and_var_index_names_specified(self):
h5ad_file = H5ADDataFile(
self.sample_h5ad_filename,
use_corpora_schema=False,
obs_index_column_name="float_category",
vars_index_column_name="int_category",
)
self.assertNotIn("name_0", h5ad_file.obs.columns)
self.assertNotIn("name_0", h5ad_file.var.columns)
def test__create_h5ad_data_file__obs_and_var_index_names_specified_not_unique_raises_exception(self):
with self.assertRaises(Exception) as exception_context:
H5ADDataFile(
self.sample_h5ad_filename,
use_corpora_schema=False,
obs_index_column_name="float_category",
vars_index_column_name="bool_category",
)
self.assertIn("Please prepare data to contain unique values", str(exception_context.exception))
def test__create_h5ad_data_file__obs_and_var_index_names_specified_doesnt_exist_raises_exception(self):
with self.assertRaises(Exception) as exception_context:
H5ADDataFile(
self.sample_h5ad_filename,
use_corpora_schema=False,
obs_index_column_name="unknown_category",
vars_index_column_name="i_dont_exist",
)
self.assertIn("does not exist", str(exception_context.exception))
def test__create_h5ad_data_file__extract_about_and_title_from_dataset(self):
h5ad_file = H5ADDataFile(self.sample_h5ad_filename)
self.assertEqual(h5ad_file.dataset_title, "random_link_name")
self.assertEqual(h5ad_file.dataset_about, "www.link.com")
def test__create_h5ad_data_file__inputted_dataset_title_and_about_overrides_extracted(self):
h5ad_file = H5ADDataFile(
self.sample_h5ad_filename, dataset_about="override_about", dataset_title="override_title"
)
self.assertEqual(h5ad_file.dataset_title, "override_title")
self.assertEqual(h5ad_file.dataset_about, "override_about")
def test__to_cxg__simple_anndata_no_corpora_and_sparse(self):
h5ad_file = H5ADDataFile(self.sample_h5ad_filename, use_corpora_schema=False)
h5ad_file.to_cxg(self.sample_output_directory, 100)
self._validate_expected_generated_list_of_tiledb_files()
def test__to_cxg__simple_anndata_with_corpora_and_sparse(self):
h5ad_file = H5ADDataFile(self.sample_h5ad_filename)
h5ad_file.to_cxg(self.sample_output_directory, 100)
self._validate_expected_generated_list_of_tiledb_files()
def test__to_cxg__simple_anndata_no_corpora_and_dense(self):
h5ad_file = H5ADDataFile(self.sample_h5ad_filename, use_corpora_schema=False)
h5ad_file.to_cxg(self.sample_output_directory, 0)
self._validate_expected_generated_list_of_tiledb_files()
def test__to_cxg__simple_anndata_with_corpora_and_dense(self):
h5ad_file = H5ADDataFile(self.sample_h5ad_filename)
h5ad_file.to_cxg(self.sample_output_directory, 0)
self._validate_expected_generated_list_of_tiledb_files()
def test__to_cxg__with_sparse_column_encoding(self):
anndata = self._create_sample_anndata_dataset()
anndata.X = np.ones((3, 4))
sparse_with_column_shift_filename = self._write_anndata_to_file(anndata)
h5ad_file = H5ADDataFile(sparse_with_column_shift_filename)
h5ad_file.to_cxg(self.sample_output_directory, 50)
self._validate_expected_generated_list_of_tiledb_files(has_column_encoding=True)
# Clean up
remove(sparse_with_column_shift_filename)
def _validate_expected_generated_list_of_tiledb_files(self, has_column_encoding=False):
(
expected_directories,
expected_obs_files,
expected_var_files,
) = self._get_expected_generated_list_of_tiledb_files()
for directory in expected_directories:
self.assertTrue(path.isdir(directory))
for obs_file in expected_obs_files:
expected_location_of_obs_file = f"{self.sample_output_directory}/obs/*/{obs_file}"
self.assertTrue(path.isfile(glob(expected_location_of_obs_file)[0]))
for var_file in expected_var_files:
expected_location_of_var_file = f"{self.sample_output_directory}/var/*/{var_file}"
self.assertTrue(path.isfile(glob(expected_location_of_var_file)[0]))
if has_column_encoding:
self.assertTrue(path.isdir(f"{self.sample_output_directory}/X_col_shift"))
def _get_expected_generated_list_of_tiledb_files(self):
# Expected directories
metadata_directory = f"{self.sample_output_directory}/cxg_group_metadata"
main_x_directory = f"{self.sample_output_directory}/X"
overall_embedding_directory = f"{self.sample_output_directory}/emb"
specific_embedding_directory = f"{self.sample_output_directory}/emb/awesome_embedding"
obs_directory = f"{self.sample_output_directory}/obs"
var_directory = f"{self.sample_output_directory}/var"
# Obs files
obs_files = []
obs_files.append("name_0.tdb")
obs_files.append("name_0_var.tdb")
obs_files.append("string_category.tdb")
obs_files.append("string_category_var.tdb")
obs_files.append("float_category.tdb")
# Var files
var_files = []
var_files.append("name_0.tdb")
var_files.append("name_0_var.tdb")
var_files.append("bool_category.tdb")
var_files.append("int_category.tdb")
return (
[
metadata_directory,
main_x_directory,
overall_embedding_directory,
specific_embedding_directory,
obs_directory,
var_directory,
],
obs_files,
var_files,
)
def _write_anndata_to_file(self, anndata):
temporary_filename = f"{PROJECT_ROOT}/backend/test/fixtures/{uuid4()}.h5ad"
anndata.write(temporary_filename)
return temporary_filename
def _create_sample_anndata_dataset(self):
# Create X
X = np.random.rand(3, 4)
# Create obs
random_string_category = Series(data=["a", "b", "b"], dtype="category")
random_float_category = Series(data=[3.2, 1.1, 2.2], dtype=np.float32)
obs_dataframe = DataFrame(
data={"string_category": random_string_category, "float_category": random_float_category}
)
obs = obs_dataframe
# Create vars
random_int_category = Series(data=[3, 1, 2, 4], dtype=np.int32)
random_bool_category = Series(data=[True, True, False, True], dtype=np.bool_)
var_dataframe = DataFrame(data={"int_category": random_int_category, "bool_category": random_bool_category})
var = var_dataframe
# Create embeddings
random_embedding = np.random.rand(3, 2)
obsm = {"X_awesome_embedding": random_embedding}
# Create uns corpora metadata
uns = {}
for metadata_field in CorporaConstants.REQUIRED_SIMPLE_METADATA_FIELDS:
uns[metadata_field] = "random"
for metadata_field in CorporaConstants.OPTIONAL_JSON_ENCODED_METADATA_FIELD:
uns[metadata_field] = json.dumps({"random_key": "random_value"})
# Need to carefully set the corpora schema versions in order for tests to pass.
uns["version"] = {"corpora_schema_version": "1.0.0", "corpora_encoding_version": "0.1.0"}
# Set project links to be a dictionary
uns["project_links"] = json.dumps(
[{"link_name": "random_link_name", "link_url": "www.link.com", "link_type": "SUMMARY"}]
)
return anndata.AnnData(X=X, obs=obs, var=var, obsm=obsm, uns=uns)