mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-29 04:18:11 +08:00
Clean up dead/hosted code [zh2310] (#2430)
* Clean up dead/hosted code * Remove schema conversion tool and related * Remove cxg references * Remove locust * missed a spot * Remove aws secret manager * Merge branch 'main' into brodgers/2310/code-cleanup-v1 * cleanup merge
This commit is contained in:
@@ -1,39 +0,0 @@
|
||||
# Locust Load Test
|
||||
|
||||
This directory contains scripts to load test cellxgene's backend. It
|
||||
primary simulates initial data loading and expression data fetch, which
|
||||
are the most common data routes. It currently does not include tests
|
||||
for differential expression or re-clustering routes.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
You need:
|
||||
|
||||
- Python 3.6+, and pip
|
||||
- cellxgene installed
|
||||
- install the locust dependencies in `requirements-locust.txt`
|
||||
|
||||
## To test
|
||||
|
||||
1. Choose to run cellxgene in either single dataset or data root mode.
|
||||
2. Edit config.py to indicate which datasets to load:
|
||||
- in single dataset mode, just set `DataSets=[""]`
|
||||
- in dataroot (multi-dataset) mode, add the route names, eg, `DataSets=['foo.cxg', 'bar.cxg']`
|
||||
3. Launch cellxgene in the appropriate mode
|
||||
4. launch locust, specifying the correct --host argument
|
||||
5. point your web browser to the locust http server, usually `http://localhost:8089/`
|
||||
|
||||
### Single dataset mode
|
||||
|
||||
- Edit config.py and set `DataSets=[""]`
|
||||
- in a shell, run `cellxgene launch somefile.h5ad`
|
||||
- launch locust in another shell, `locust --host http://localhost:5005/` (or wherever you are running cellxgene)
|
||||
- point a browser to the locust port, usually http://localhost:8089/
|
||||
- run test
|
||||
|
||||
### Multi-dataset mode
|
||||
|
||||
- Edit config.py and set `DataSets=["datapath1", ...]`
|
||||
- in a shell, run `cellxgene launch --dataroot path`
|
||||
|
||||
The remainder of the steps are same as single dataset.
|
||||
@@ -1,15 +0,0 @@
|
||||
"""
|
||||
Locust test config
|
||||
"""
|
||||
|
||||
|
||||
""" Data routes that will be tested """
|
||||
|
||||
# single dataset, for non-dataroot tests
|
||||
# DataSets = [""]
|
||||
|
||||
# multi-dataset, for dataroot tests. these are varied in size/shape
|
||||
DataSets = [
|
||||
"GSE60361.cxg",
|
||||
"WongAdultRetina.cxg",
|
||||
]
|
||||
@@ -1,165 +0,0 @@
|
||||
import json
|
||||
import random
|
||||
|
||||
import requests
|
||||
from config import DataSets
|
||||
from locust import HttpUser, SequentialTaskSet, task, between, TaskSet
|
||||
from locust.clients import HttpSession
|
||||
from requests.packages.urllib3.exceptions import InsecureRequestWarning
|
||||
|
||||
import backend.test.decode_fbs as decode_fbs
|
||||
|
||||
requests.packages.urllib3.disable_warnings(InsecureRequestWarning)
|
||||
|
||||
"""
|
||||
Simple locust stress test definition for cellxgene
|
||||
"""
|
||||
|
||||
API_SUFFIX = "api/v0.2"
|
||||
|
||||
|
||||
class CellXGeneTasks(TaskSet):
|
||||
"""
|
||||
Simulate use against a single dataset
|
||||
"""
|
||||
|
||||
def on_start(self):
|
||||
|
||||
self.client.verify = False
|
||||
self.dataset = random.choice(DataSets)
|
||||
|
||||
with self.client.get(
|
||||
f"{self.dataset}/{API_SUFFIX}/schema", stream=True, catch_response=True
|
||||
) as schema_response:
|
||||
if schema_response.status_code == 200:
|
||||
self.schema = schema_response.json()["schema"]
|
||||
else:
|
||||
self.schema = None
|
||||
|
||||
with self.client.get(
|
||||
f"{self.dataset}/{API_SUFFIX}/config", stream=True, catch_response=True
|
||||
) as config_response:
|
||||
if config_response.status_code == 200:
|
||||
self.config = config_response.json()["config"]
|
||||
else:
|
||||
self.config = None
|
||||
|
||||
with self.client.get(
|
||||
f"{self.dataset}/{API_SUFFIX}/annotations/var?annotation-name={self.var_index_name()}",
|
||||
headers={"Accept": "application/octet-stream"},
|
||||
catch_response=True,
|
||||
) as var_index_response:
|
||||
if var_index_response.status_code == 200:
|
||||
df = decode_fbs.decode_matrix_FBS(var_index_response.content)
|
||||
gene_names_idx = df["col_idx"].index(self.var_index_name())
|
||||
self.gene_names = df["columns"][gene_names_idx]
|
||||
else:
|
||||
self.gene_names = []
|
||||
|
||||
def var_index_name(self):
|
||||
if self.schema is not None:
|
||||
return self.schema["annotations"]["var"]["index"]
|
||||
return None
|
||||
|
||||
def obs_annotation_names(self):
|
||||
if self.schema is not None:
|
||||
return [col["name"] for col in self.schema["annotations"]["obs"]["columns"]]
|
||||
return []
|
||||
|
||||
def layout_names(self):
|
||||
if self.schema is not None:
|
||||
return [layout["name"] for layout in self.schema["layout"]["obs"]]
|
||||
else:
|
||||
return []
|
||||
|
||||
@task(2)
|
||||
class InitializeClient(SequentialTaskSet):
|
||||
"""
|
||||
Initial loading of cellxgene - when the user hits the main route.
|
||||
|
||||
Currently this sequence skips some of the static assets, which are quite small and should be served by the
|
||||
HTTP server directly.
|
||||
|
||||
1. Load index.html, etc.
|
||||
2. Concurrently load /config, /schema
|
||||
3. Concurrently load /layout/obs, /annotations/var?annotation-name=<the index>
|
||||
-- Does initial render --
|
||||
4. Concurrently load all /annotations/obs and all /layouts/obs
|
||||
-- Fully initialized --
|
||||
"""
|
||||
|
||||
# Users hit all of the init routes as fast as they can, subject to the ordering constraints and network latency.
|
||||
wait_time = between(0.01, 0.1)
|
||||
|
||||
def on_start(self):
|
||||
self.dataset = self.parent.dataset
|
||||
self.client.verify = False
|
||||
self.api_less_client = HttpSession(
|
||||
base_url=self.client.base_url.replace("api.", "").replace("cellxgene/", ""),
|
||||
request_success=self.client.request_success,
|
||||
request_failure=self.client.request_failure,
|
||||
)
|
||||
|
||||
@task
|
||||
def index(self):
|
||||
self.api_less_client.get(f"{self.dataset}", stream=True)
|
||||
|
||||
@task
|
||||
def loadConfigAndSchema(self):
|
||||
self.client.get(f"{self.dataset}/{API_SUFFIX}/schema", stream=True, catch_response=True)
|
||||
self.client.get(f"{self.dataset}/{API_SUFFIX}/config", stream=True, catch_response=True)
|
||||
|
||||
@task
|
||||
def loadBootstrapData(self):
|
||||
self.client.get(
|
||||
f"{self.dataset}/{API_SUFFIX}/layout/obs", headers={"Accept": "application/octet-stream"}, stream=True
|
||||
)
|
||||
self.client.get(
|
||||
f"{self.dataset}/{API_SUFFIX}/annotations/var?annotation-name={self.parent.var_index_name()}",
|
||||
headers={"Accept": "application/octet-stream"},
|
||||
catch_response=True,
|
||||
)
|
||||
|
||||
@task
|
||||
def loadObsAnnotationsAndLayouts(self):
|
||||
obs_names = self.parent.obs_annotation_names()
|
||||
for name in obs_names:
|
||||
self.client.get(
|
||||
f"{self.dataset}/{API_SUFFIX}/annotations/obs?annotation-name={name}",
|
||||
headers={"Accept": "application/octet-stream"},
|
||||
stream=True,
|
||||
)
|
||||
|
||||
layouts = self.parent.layout_names()
|
||||
for name in layouts:
|
||||
self.client.get(
|
||||
f"{self.dataset}/{API_SUFFIX}/annotations/obs?layout-name={name}",
|
||||
headers={"Accept": "application/octet-stream"},
|
||||
stream=True,
|
||||
)
|
||||
|
||||
@task
|
||||
def done(self):
|
||||
self.interrupt()
|
||||
|
||||
@task(1)
|
||||
def load_expression(self):
|
||||
"""
|
||||
Simulate user occasionally loading some expression data for a gene
|
||||
"""
|
||||
|
||||
gene_name = random.choice(self.gene_names)
|
||||
filter = {"filter": {"var": {"annotation_value": [{"name": self.var_index_name(), "values": [gene_name]}]}}}
|
||||
self.client.put(
|
||||
f"{self.dataset}/{API_SUFFIX}/data/var",
|
||||
data=json.dumps(filter),
|
||||
headers={"Content-Type": "application/json", "Accept": "application/octet-stream"},
|
||||
stream=True,
|
||||
).close()
|
||||
|
||||
|
||||
class CellxgeneUser(HttpUser):
|
||||
tasks = [CellXGeneTasks]
|
||||
|
||||
# Most ops do not require back-end interaction, so slow cadence for users
|
||||
wait_time = between(10, 60)
|
||||
@@ -1,2 +0,0 @@
|
||||
locust
|
||||
-r ../../requirements.txt
|
||||
@@ -98,8 +98,6 @@ class ConfigTests(unittest.TestCase):
|
||||
lfc_cutoff=0.01,
|
||||
top_n=10,
|
||||
environment=None,
|
||||
aws_secrets_manager_region=None,
|
||||
aws_secrets_manager_secrets=[],
|
||||
X_approximate_distribution="auto",
|
||||
config_file_name="app_config.yml",
|
||||
):
|
||||
@@ -151,8 +149,6 @@ class ConfigTests(unittest.TestCase):
|
||||
)
|
||||
external_config = self.custom_external_config(
|
||||
environment=environment,
|
||||
aws_secrets_manager_region=aws_secrets_manager_region,
|
||||
aws_secrets_manager_secrets=aws_secrets_manager_secrets,
|
||||
config_file_name=f"temp_external_config_{random_num}.yml",
|
||||
)
|
||||
|
||||
@@ -197,8 +193,6 @@ class ConfigTests(unittest.TestCase):
|
||||
def custom_external_config(
|
||||
self,
|
||||
environment=None,
|
||||
aws_secrets_manager_region=None,
|
||||
aws_secrets_manager_secrets=[],
|
||||
config_file_name="external_config.yaml",
|
||||
):
|
||||
# set to the default if environment is None
|
||||
@@ -209,7 +203,6 @@ class ConfigTests(unittest.TestCase):
|
||||
external_config = {
|
||||
"external": {
|
||||
"environment": environment,
|
||||
"aws_secrets_manager": {"region": aws_secrets_manager_region, "secrets": aws_secrets_manager_secrets},
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import requests
|
||||
|
||||
@@ -13,7 +12,7 @@ from backend.test.test_server.unit.common.config import ConfigTests
|
||||
|
||||
class TestExternalConfig(ConfigTests):
|
||||
def test_type_convert(self):
|
||||
# The values from environment variables and aws secrets are returned as strings.
|
||||
# The values from environment variables are returned as strings.
|
||||
# These values need to be converted to the proper types.
|
||||
|
||||
self.assertEqual(convert_string_to_value("1"), int(1))
|
||||
@@ -88,129 +87,3 @@ class TestExternalConfig(ConfigTests):
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(config_error.exception.message, "required environment variable 'THIS_ENV_IS_NOT_SET' not set")
|
||||
|
||||
@patch("backend.server.common.config.external_config.get_secret_key")
|
||||
def test_aws_secrets_manager(self, mock_get_secret_key):
|
||||
mock_get_secret_key.return_value = {
|
||||
"flask_secret_key": "mock_flask_secret_key",
|
||||
}
|
||||
configfile = self.custom_external_config(
|
||||
aws_secrets_manager_region="us-west-2",
|
||||
aws_secrets_manager_secrets=[
|
||||
dict(
|
||||
name="my_secret",
|
||||
values=[
|
||||
dict(key="flask_secret_key", path=["server", "app", "flask_secret_key"], required=True),
|
||||
],
|
||||
)
|
||||
],
|
||||
config_file_name="secret_external_config.yaml",
|
||||
)
|
||||
|
||||
app_config = AppConfig()
|
||||
app_config.update_from_config_file(configfile)
|
||||
app_config.server_config.single_dataset__datapath = f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad"
|
||||
|
||||
app_config.complete_config()
|
||||
|
||||
self.assertEqual(app_config.server_config.app__flask_secret_key, "mock_flask_secret_key")
|
||||
|
||||
@patch("backend.server.common.config.external_config.get_secret_key")
|
||||
def test_aws_secrets_manager_error(self, mock_get_secret_key):
|
||||
mock_get_secret_key.return_value = {
|
||||
"db_uri": "mock_db_uri",
|
||||
}
|
||||
|
||||
# no region
|
||||
app_config = AppConfig()
|
||||
app_config.external_config.aws_secrets_manager__region = None
|
||||
app_config.external_config.aws_secrets_manager__secrets = [
|
||||
dict(name="secret1", values=[dict(key="key1", required=True, path=["this", "is", "my", "path"])])
|
||||
]
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(
|
||||
config_error.exception.message,
|
||||
"Invalid type for attribute: aws_secrets_manager__region, expected type str, got NoneType",
|
||||
)
|
||||
|
||||
# missing secret name
|
||||
app_config = AppConfig()
|
||||
app_config.external_config.aws_secrets_manager__region = "us-west-2"
|
||||
app_config.external_config.aws_secrets_manager__secrets = [
|
||||
dict(values=[dict(key="db_uri", required=True, path=["this", "is", "my", "path"])])
|
||||
]
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(config_error.exception.message, "aws_secrets_manager: 'name' is missing")
|
||||
|
||||
# secret name wrong type
|
||||
app_config = AppConfig()
|
||||
app_config.external_config.aws_secrets_manager__region = "us-west-2"
|
||||
app_config.external_config.aws_secrets_manager__secrets = [
|
||||
dict(name=1, values=[dict(key="db_uri", required=True, path=["this", "is", "my", "path"])])
|
||||
]
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(config_error.exception.message, "aws_secrets_manager: 'name' must be a string")
|
||||
|
||||
# missing values name
|
||||
app_config = AppConfig()
|
||||
app_config.external_config.aws_secrets_manager__region = "us-west-2"
|
||||
app_config.external_config.aws_secrets_manager__secrets = [dict(name="mysecret")]
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(config_error.exception.message, "aws_secrets_manager: 'values' is missing")
|
||||
|
||||
# values wrong type
|
||||
app_config = AppConfig()
|
||||
app_config.external_config.aws_secrets_manager__region = "us-west-2"
|
||||
app_config.external_config.aws_secrets_manager__secrets = [
|
||||
dict(name="mysecret", values=dict(key="db_uri", required=True, path=["this", "is", "my", "path"]))
|
||||
]
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(config_error.exception.message, "aws_secrets_manager: 'values' must be a list")
|
||||
|
||||
# entry missing key
|
||||
app_config = AppConfig()
|
||||
app_config.external_config.aws_secrets_manager__region = "us-west-2"
|
||||
app_config.external_config.aws_secrets_manager__secrets = [
|
||||
dict(name="mysecret", values=[dict(required=True, path=["this", "is", "my", "path"])])
|
||||
]
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(config_error.exception.message, "missing 'key' in secret values: mysecret")
|
||||
|
||||
# entry required is wrong type
|
||||
app_config = AppConfig()
|
||||
app_config.external_config.aws_secrets_manager__region = "us-west-2"
|
||||
app_config.external_config.aws_secrets_manager__secrets = [
|
||||
dict(name="mysecret", values=[dict(key="db_uri", required="optional", path=["this", "is", "my", "path"])])
|
||||
]
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(config_error.exception.message, "wrong type for 'required' in secret values: mysecret")
|
||||
|
||||
# entry missing path
|
||||
app_config = AppConfig()
|
||||
app_config.external_config.aws_secrets_manager__region = "us-west-2"
|
||||
app_config.external_config.aws_secrets_manager__secrets = [
|
||||
dict(name="mysecret", values=[dict(key="db_uri", required=True)])
|
||||
]
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(config_error.exception.message, "missing 'path' in secret values: mysecret")
|
||||
|
||||
# secret missing required key
|
||||
app_config = AppConfig()
|
||||
app_config.external_config.aws_secrets_manager__region = "us-west-2"
|
||||
app_config.external_config.aws_secrets_manager__secrets = [
|
||||
dict(
|
||||
name="mysecret",
|
||||
values=[dict(key="KEY_DOES_NOT_EXIST", required=True, path=["this", "is", "a", "path"])],
|
||||
)
|
||||
]
|
||||
with self.assertRaises(ConfigurationError) as config_error:
|
||||
app_config.complete_config()
|
||||
self.assertEqual(config_error.exception.message, "required secret 'mysecret:KEY_DOES_NOT_EXIST' not set")
|
||||
|
||||
@@ -1,61 +0,0 @@
|
||||
import os
|
||||
import unittest
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from backend.test import FIXTURES_ROOT
|
||||
from backend.server.converters.schema import gene_symbol
|
||||
|
||||
|
||||
class TestHGNCSymbolChecker(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.test_hgnc_path = os.path.join(FIXTURES_ROOT, "hgnc_example.txt.gz")
|
||||
self.hgnc_checker = gene_symbol.HGNCSymbolChecker.from_hgnc_records(self.test_hgnc_path)
|
||||
|
||||
def test_symbol_upgrade(self):
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("SEPT1"), "SEPTIN1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("ADRB2R"), "ADRB2")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("BAR"), "ADRB2")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("sept1"), "SEPTIN1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("AdRb2R"), "ADRB2")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("bar"), "ADRB2")
|
||||
|
||||
# Strip off seurat endings when appropriate
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("SEPT1.1"), "SEPTIN1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("ADRB2-1"), "ADRB2")
|
||||
|
||||
# DIFF6 is ambiguous so don't upgrade it
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("DIFF6"), "DIFF6")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("diff6"), "diff6")
|
||||
|
||||
# ARG1 is approved
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("ARG1"), "ARG1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("arg1"), "ARG1")
|
||||
|
||||
# HAP1 is both approved and withdrawn
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("HAP1"), "HAP1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("hap1"), "HAP1")
|
||||
|
||||
# Leave unknown symbols alone
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("NOTASYMBOL"), "NOTASYMBOL")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("notasymbol"), "notasymbol")
|
||||
|
||||
# Upgrade HGNC ids unless you can't find it
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("HGNC:286"), "ADRB2")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("HGNC:4812"), "HAP1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("HGNC:123456"), "HGNC:123456")
|
||||
|
||||
def test_check_symbol(self):
|
||||
self.assertEqual(self.hgnc_checker.check_symbol("SEPT1"), gene_symbol.SymbolStatus.UPGRADABLE)
|
||||
self.assertEqual(self.hgnc_checker.check_symbol("DIFF6"), gene_symbol.SymbolStatus.AMBIGUOUS)
|
||||
self.assertEqual(self.hgnc_checker.check_symbol("NOTASYMBOL"), gene_symbol.SymbolStatus.UNKNOWN)
|
||||
|
||||
# HAP1 is one of the approved and withdrawn symbols
|
||||
self.assertEqual(self.hgnc_checker.check_symbol("HAP1"), gene_symbol.SymbolStatus.APPROVED)
|
||||
|
||||
def test_upgrade_index(self):
|
||||
index = pd.Index(["SEPT1", "DIFF6", "NOTASYMBOL", "bar", "SEPTIN1"])
|
||||
var_df = pd.DataFrame([[0] * len(index)], index=index)
|
||||
upgraded_index = gene_symbol.get_upgraded_var_index(var_df, hgnc_path=self.test_hgnc_path)
|
||||
self.assertEqual(upgraded_index.tolist(), ["SEPTIN1", "DIFF6", "NOTASYMBOL", "ADRB2", "SEPTIN1"])
|
||||
@@ -1,128 +0,0 @@
|
||||
import json
|
||||
|
||||
import unittest.mock
|
||||
|
||||
from backend.server.converters.schema import ontology
|
||||
|
||||
|
||||
class TestOntologyParsing(unittest.TestCase):
|
||||
def setUp(self):
|
||||
|
||||
self.curies = ["UBERON:0002048", "HsapDv:0000174", "NCBITaxon:9606", "EFO:0008995"]
|
||||
|
||||
self.names = ["UBERON", "HsapDv", "NCBITaxon", "EFO"]
|
||||
|
||||
self.values = ["0002048", "0000174", "9606", "0008995"]
|
||||
|
||||
self.iris = [
|
||||
"http://purl.obolibrary.org/obo/UBERON_0002048",
|
||||
"http://purl.obolibrary.org/obo/HsapDv_0000174",
|
||||
"http://purl.obolibrary.org/obo/NCBITaxon_9606",
|
||||
"http://www.ebi.ac.uk/efo/EFO_0008995",
|
||||
]
|
||||
|
||||
URL_ROOT = "http://www.ebi.ac.uk/ols/api/ontologies/"
|
||||
self.urls = [
|
||||
URL_ROOT + "UBERON/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FUBERON_0002048",
|
||||
URL_ROOT + "HsapDv/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FHsapDv_0000174",
|
||||
URL_ROOT + "NCBITaxon/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FNCBITaxon_9606",
|
||||
URL_ROOT + "EFO/terms/http%253A%252F%252Fwww.ebi.ac.uk%252Fefo%252FEFO_0008995",
|
||||
]
|
||||
|
||||
self.responses = {
|
||||
"UBERON:0002048": {
|
||||
"iri": "http://purl.obolibrary.org/obo/UBERON_0002048",
|
||||
"description": ["Respiration organ that develops as an outpocketing of the esophagus."],
|
||||
"label": "lung",
|
||||
},
|
||||
"HsapDv:0000174": {
|
||||
"iri": "http://purl.obolibrary.org/obo/HsapDv_0000174",
|
||||
"description": ["Infant stage that refers to an infant who is over 1 and under 2 months old."],
|
||||
"label": "1-month-old human stage",
|
||||
},
|
||||
"NCBITaxon:9606": {
|
||||
"iri": "http://purl.obolibrary.org/obo/NCBITaxon_9606",
|
||||
"description": None,
|
||||
"label": "Homo sapiens",
|
||||
},
|
||||
"EFO:0008995": {
|
||||
"iri": "http://www.ebi.ac.uk/efo/EFO_0008995",
|
||||
"description": [
|
||||
(
|
||||
'10X is a "synthetic long-read" technology and works by capturing a barcoded oligo-coated '
|
||||
"gel-bead and 0.3x genome copies into a single emulsion droplet, processing the equivalent "
|
||||
"of 1 million pipetting steps. Successive versions of the 10x chemistry use different "
|
||||
"barcode locations to improve the sequencing yield and quality of 10x experiments."
|
||||
)
|
||||
],
|
||||
"label": "10X sequencing",
|
||||
},
|
||||
}
|
||||
|
||||
def test_ontololgy_name(self):
|
||||
for curie, expected_name in zip(self.curies, self.names):
|
||||
self.assertEqual(ontology._ontology_name(curie), expected_name)
|
||||
|
||||
def test_ontololgy_value(self):
|
||||
for curie, expected_value in zip(self.curies, self.values):
|
||||
self.assertEqual(ontology._ontology_value(curie), expected_value)
|
||||
|
||||
def test_iri(self):
|
||||
for curie, expected_iri in zip(self.curies, self.iris):
|
||||
self.assertEqual(ontology._iri(curie), expected_iri)
|
||||
|
||||
def test_ontology_info_url(self):
|
||||
for curie, expected_url in zip(self.curies, self.urls):
|
||||
self.assertEqual(ontology._ontology_info_url(curie), expected_url)
|
||||
|
||||
def test_empty_ontology_info_url(self):
|
||||
self.assertEqual(ontology._ontology_info_url(""), "")
|
||||
|
||||
|
||||
class TestOntologyLookup(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.responses = {
|
||||
"UBERON:0002048": {
|
||||
"iri": "http://purl.obolibrary.org/obo/UBERON_0002048",
|
||||
"description": ["Respiration organ that develops as an outpocketing of the esophagus."],
|
||||
"label": "lung",
|
||||
},
|
||||
"HsapDv:0000174": {
|
||||
"iri": "http://purl.obolibrary.org/obo/HsapDv_0000174",
|
||||
"description": ["Infant stage that refers to an infant who is over 1 and under 2 months old."],
|
||||
"label": "1-month-old human stage",
|
||||
},
|
||||
"NCBITaxon:9606": {
|
||||
"iri": "http://purl.obolibrary.org/obo/NCBITaxon_9606",
|
||||
"description": None,
|
||||
"label": "Homo sapiens",
|
||||
},
|
||||
"EFO:0008995": {
|
||||
"iri": "http://www.ebi.ac.uk/efo/EFO_0008995",
|
||||
"description": [
|
||||
('10X is a "synthetic long-read" technology and works by capturing a barcoded oligo-coated '
|
||||
'gel-bead and 0.3x genome copies into a single emulsion droplet, processing the equivalent '
|
||||
'of 1 million pipetting steps. Successive versions of the 10x chemistry use different barcode '
|
||||
'locations to improve the sequencing yield and quality of 10x experiments.')
|
||||
],
|
||||
"label": "10X sequencing",
|
||||
},
|
||||
}
|
||||
|
||||
self.labels = {
|
||||
"UBERON:0002048": "lung",
|
||||
"HsapDv:0000174": "1-month-old human stage",
|
||||
"NCBITaxon:9606": "Homo sapiens",
|
||||
"EFO:0008995": "10X sequencing",
|
||||
}
|
||||
|
||||
@unittest.mock.patch("requests.get")
|
||||
def test_lookup_label(self, mock_get):
|
||||
|
||||
for curie, response in self.responses.items():
|
||||
mock_get.return_value.content = json.dumps(response)
|
||||
mock_get.return_value.json.return_value = response
|
||||
mock_get.return_value.status_code = 200
|
||||
|
||||
label = ontology.get_ontology_label(curie)
|
||||
self.assertEqual(label, self.labels[curie])
|
||||
@@ -1,256 +0,0 @@
|
||||
import json
|
||||
import os
|
||||
import unittest
|
||||
import unittest.mock
|
||||
|
||||
import anndata
|
||||
import numpy
|
||||
import pandas as pd
|
||||
import scanpy as sc
|
||||
|
||||
from backend.server.converters.schema import remix
|
||||
from backend.test import PROJECT_ROOT
|
||||
|
||||
|
||||
class TestApplySchema(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.source_h5ad_path = f"{PROJECT_ROOT}/backend/test/fixtures/pbmc3k-CSC-gz.h5ad"
|
||||
self.output_h5ad_path = f"{PROJECT_ROOT}/backend/test/fixtures/test_remix.h5ad"
|
||||
self.config_path = f"{PROJECT_ROOT}/backend/test/fixtures/test_config.yaml"
|
||||
self.bad_config_path = f"{PROJECT_ROOT}/backend/test/fixtures/test_bad_config.yaml"
|
||||
|
||||
def tearDown(self):
|
||||
try:
|
||||
os.remove(self.output_h5ad_path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
|
||||
def test_apply_schema(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "test label"
|
||||
remix.apply_schema(self.source_h5ad_path, self.config_path, self.output_h5ad_path)
|
||||
new_adata = sc.read_h5ad(self.output_h5ad_path)
|
||||
|
||||
self.assertIn("cell_type", new_adata.obs.columns)
|
||||
self.assertListEqual(["test label"], new_adata.obs["cell_type"].unique().tolist())
|
||||
self.assertListEqual(
|
||||
["CL:00001", "CL:00002", "CL:00003", "CL:00004", "CL:00005", "CL:00006", "CL:00007", "CL:00008"],
|
||||
sorted(new_adata.obs["cell_type_ontology_term_id"].unique().tolist())
|
||||
)
|
||||
|
||||
self.assertIn("version", new_adata.uns_keys())
|
||||
|
||||
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
|
||||
def test_apply_bad_schema(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "test label"
|
||||
remix.apply_schema(self.source_h5ad_path, self.bad_config_path, self.output_h5ad_path)
|
||||
new_adata = sc.read_h5ad(self.output_h5ad_path)
|
||||
|
||||
# Should refuse to write the version
|
||||
self.assertNotIn("version", new_adata.uns_keys())
|
||||
|
||||
class TestFieldParsing(unittest.TestCase):
|
||||
|
||||
def test_is_curie(self):
|
||||
self.assertTrue(remix.is_curie("EFO:00001"))
|
||||
self.assertTrue(remix.is_curie("UBERON:123456"))
|
||||
self.assertTrue(remix.is_curie("HsapDv:0001"))
|
||||
self.assertFalse(remix.is_curie("UBERON"))
|
||||
self.assertFalse(remix.is_curie("UBERON:"))
|
||||
self.assertFalse(remix.is_curie("123456"))
|
||||
|
||||
def test_is_ontology_field(self):
|
||||
self.assertTrue(remix.is_ontology_field("tissue_ontology_term_id"))
|
||||
self.assertTrue(remix.is_ontology_field("cell_type_ontology_term_id"))
|
||||
self.assertFalse(remix.is_ontology_field("cell_ontology"))
|
||||
self.assertFalse(remix.is_ontology_field("method"))
|
||||
|
||||
def test_get_label_field_name(self):
|
||||
self.assertEqual("tissue", remix.get_label_field_name("tissue_ontology_term_id"))
|
||||
self.assertEqual("cell_type", remix.get_label_field_name("cell_type_ontology_term_id"))
|
||||
|
||||
def test_split_suffix(self):
|
||||
self.assertEqual(("UBERON:1234", " (organoid)"), remix.split_suffix("UBERON:1234 (organoid)"))
|
||||
self.assertEqual(("UBERON:1234", " (cell culture)"), remix.split_suffix("UBERON:1234 (cell culture)"))
|
||||
self.assertEqual(("UBERON:1234", ""), remix.split_suffix("UBERON:1234"))
|
||||
self.assertEqual(("UBERON:1234 (something)", ""), remix.split_suffix("UBERON:1234 (something)"))
|
||||
|
||||
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
|
||||
def test_get_curie_and_label(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "test label"
|
||||
self.assertEqual(
|
||||
remix.get_curie_and_label("UBERON:1234"),
|
||||
("UBERON:1234", "test label")
|
||||
)
|
||||
self.assertEqual(
|
||||
remix.get_curie_and_label("UBERON:1234 (cell culture)"),
|
||||
("UBERON:1234 (cell culture)", "test label (cell culture)")
|
||||
)
|
||||
self.assertEqual(
|
||||
remix.get_curie_and_label("whatever"),
|
||||
("", "whatever")
|
||||
)
|
||||
|
||||
|
||||
class TestManipulateAnndata(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
|
||||
self.cell_count = 20
|
||||
self.gene_count = 200
|
||||
X = numpy.random.randint(0, 1000, (self.cell_count, self.gene_count))
|
||||
uns = {"organism": "monkey", "experiment": "monkey experiment"}
|
||||
obs = pd.DataFrame(
|
||||
index=[f"Cell{d}" for d in range(self.cell_count)],
|
||||
columns=["tissue", "CellType"],
|
||||
data=[["lung", "epithelial"]] * (self.cell_count // 2) + [["lung", "endothelial"]] * (self.cell_count // 2)
|
||||
)
|
||||
var = pd.DataFrame(index=[f"SEPT{d}" for d in range(self.gene_count)])
|
||||
|
||||
self.adata = anndata.AnnData(X=X, obs=obs, var=var, uns=uns)
|
||||
|
||||
def test_safe_add_field(self):
|
||||
|
||||
remix.safe_add_field(self.adata.obs, "tissue", ["monkey lung"] * self.cell_count)
|
||||
self.assertEqual(self.adata.obs["tissue_original"].tolist(), ["lung"] * self.cell_count)
|
||||
self.assertEqual(self.adata.obs["tissue"].tolist(), ["monkey lung"] * self.cell_count)
|
||||
|
||||
remix.safe_add_field(self.adata.uns, "contributors", [{"name": "contributor1"}, {"name": "contributor2"}])
|
||||
self.assertEqual(
|
||||
self.adata.uns["contributors"],
|
||||
json.dumps([{"name": "contributor1"}, {"name": "contributor2"}])
|
||||
)
|
||||
|
||||
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
|
||||
def test_remix_uns(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "Pan troglodytes"
|
||||
uns_config = {
|
||||
"version": {
|
||||
"corpora_schema_version": "1.0.0",
|
||||
"corpora_encoding_version": "0.1.0"
|
||||
},
|
||||
"organism_ontology_term_id": "NCBITaxon:9598",
|
||||
"contributors": [
|
||||
{
|
||||
"name": "scientist",
|
||||
"email": "scientist@science.com"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
remix.remix_uns(self.adata, uns_config)
|
||||
|
||||
self.assertEqual(
|
||||
sorted(self.adata.uns_keys()),
|
||||
sorted(["organism_original", "organism", "organism_ontology_term_id",
|
||||
"contributors", "version", "experiment"])
|
||||
)
|
||||
|
||||
self.assertEqual(self.adata.uns['organism'], "Pan troglodytes")
|
||||
self.assertEqual(self.adata.uns['organism_original'], "monkey")
|
||||
self.assertEqual(self.adata.uns['organism_ontology_term_id'], "NCBITaxon:9598")
|
||||
self.assertEqual(self.adata.uns['contributors'],
|
||||
json.dumps([{"name": "scientist", "email": "scientist@science.com"}]))
|
||||
|
||||
@unittest.mock.patch("backend.server.converters.schema.ontology.get_ontology_label")
|
||||
def test_remix_obs(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "lung (in a monkey)"
|
||||
obs_config = {
|
||||
"tissue_ontology_term_id": {
|
||||
"tissue": {
|
||||
"lung": "UBERON:00000"
|
||||
}
|
||||
},
|
||||
"cell_color": {
|
||||
"CellType": {
|
||||
"epithelial": "fuschia",
|
||||
"endothelial": "khaki"
|
||||
}
|
||||
},
|
||||
"sex": "male"
|
||||
}
|
||||
|
||||
remix.remix_obs(self.adata, obs_config)
|
||||
self.assertEqual(
|
||||
sorted(self.adata.obs_keys()),
|
||||
sorted(["tissue", "tissue_ontology_term_id", "tissue_original", "CellType", "cell_color", "sex"])
|
||||
)
|
||||
|
||||
self.assertTrue(all(v == "lung" for v in self.adata.obs.tissue_original))
|
||||
self.assertTrue(all(v == "UBERON:00000" for v in self.adata.obs.tissue_ontology_term_id))
|
||||
self.assertTrue(all(v == "lung (in a monkey)" for v in self.adata.obs.tissue))
|
||||
self.assertTrue(all(v == "male" for v in self.adata.obs.sex))
|
||||
self.assertTrue(all(v in (("epithelial", "fuschia"), ("endothelial", "khaki"))
|
||||
for v in zip(self.adata.obs.CellType, self.adata.obs.cell_color)))
|
||||
|
||||
|
||||
class TestFixupGeneSymbols(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.seurat_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/seurat_tutorial.h5ad"
|
||||
self.seurat_merged_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/seurat_tutorial_merged.h5ad"
|
||||
self.sctransform_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/sctransform.h5ad"
|
||||
self.sctransform_merged_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/sctransform_merged.h5ad"
|
||||
|
||||
# There's lots of MALAT1, but it doesn't collide with any other names,
|
||||
# so it shouldn't change during merging.
|
||||
self.stable_gene = "MALAT1"
|
||||
|
||||
def test_fixup_gene_symbols_seurat(self):
|
||||
|
||||
if not os.path.isfile(self.seurat_path):
|
||||
return unittest.skip(
|
||||
"Skipping gene symbol conversion tests because test h5ads are not present. To create them, "
|
||||
"run server/test/fixtures/schema_test_data/generate_test_data.sh"
|
||||
)
|
||||
|
||||
original_adata = sc.read_h5ad(self.seurat_path)
|
||||
merged_adata = sc.read_h5ad(self.seurat_merged_path)
|
||||
|
||||
fixup_config = {"X": "log1p", "counts": "raw", "scale.data": "log1p"}
|
||||
|
||||
fixed_adata = remix.fixup_gene_symbols(original_adata, fixup_config)
|
||||
|
||||
self.assertEqual(
|
||||
merged_adata.layers["counts"][:, merged_adata.var.index == self.stable_gene].sum(),
|
||||
fixed_adata.raw.X[:, fixed_adata.var.index == self.stable_gene].sum()
|
||||
)
|
||||
self.assertAlmostEqual(
|
||||
merged_adata.X[:, merged_adata.var.index == self.stable_gene].sum(),
|
||||
fixed_adata.X[:, fixed_adata.var.index == self.stable_gene].sum()
|
||||
)
|
||||
|
||||
self.assertAlmostEqual(
|
||||
merged_adata.layers["scale.data"][:, merged_adata.var.index == self.stable_gene].sum(),
|
||||
fixed_adata.layers["scale.data"][:, fixed_adata.var.index == self.stable_gene].sum()
|
||||
)
|
||||
|
||||
def test_fixup_gene_symbols_sctransform(self):
|
||||
|
||||
if not os.path.isfile(self.sctransform_path):
|
||||
return unittest.skip(
|
||||
"Skipping gene symbol conversion tests because test h5ads are not present. To create them, "
|
||||
"run server/test/fixtures/schema_test_data/generate_test_data.sh"
|
||||
)
|
||||
|
||||
original_adata = sc.read_h5ad(self.sctransform_path)
|
||||
merged_adata = sc.read_h5ad(self.sctransform_merged_path)
|
||||
|
||||
fixup_config = {"X": "log1p", "counts": "raw"}
|
||||
|
||||
fixed_adata = remix.fixup_gene_symbols(original_adata, fixup_config)
|
||||
|
||||
# sctransform does a bunch of stuff, including slightly modifying the
|
||||
# raw counts. So we can't assert for exact equality the way we do with
|
||||
# the vanilla seurat tutorial. But, the results should still be very
|
||||
# close.
|
||||
merged_raw_stable = merged_adata.layers["counts"][:, merged_adata.var.index == self.stable_gene].sum()
|
||||
fixed_raw_stable = fixed_adata.raw.X[:, fixed_adata.var.index == self.stable_gene].sum()
|
||||
self.assertLess(abs(merged_raw_stable - fixed_raw_stable), .001 * merged_raw_stable)
|
||||
|
||||
self.assertAlmostEqual(
|
||||
merged_adata.X[:, merged_adata.var.index == self.stable_gene].sum(),
|
||||
fixed_adata.X[:, fixed_adata.var.index == self.stable_gene].sum(),
|
||||
0
|
||||
)
|
||||
@@ -1,434 +0,0 @@
|
||||
import json
|
||||
import unittest
|
||||
|
||||
import pandas as pd
|
||||
import scanpy as sc
|
||||
|
||||
from backend.server.converters.schema import validate
|
||||
|
||||
from backend.test import PROJECT_ROOT
|
||||
|
||||
|
||||
class TestFieldValidation(unittest.TestCase):
|
||||
|
||||
def test_validate_stringified_list_of_dicts(self):
|
||||
|
||||
good = json.dumps([{"a": 1}, {2: "x", "z": "y"}])
|
||||
not_stringified = [{"a": 1}, {2: "x", "z": "y"}]
|
||||
not_a_list = json.dumps({"bad": "dict"})
|
||||
not_json = "oh hey!"
|
||||
|
||||
self.assertTrue(validate._validate_stringified_list_of_dicts(good))
|
||||
|
||||
self.assertFalse(validate._validate_stringified_list_of_dicts(not_stringified))
|
||||
self.assertFalse(validate._validate_stringified_list_of_dicts(not_a_list))
|
||||
self.assertFalse(validate._validate_stringified_list_of_dicts(not_json))
|
||||
|
||||
def test_validate_human_readable_string(self):
|
||||
|
||||
good = "oh hey!"
|
||||
curie = "EFO:0001"
|
||||
ensg = "ENSG000001234"
|
||||
enst = "ENST000005678"
|
||||
|
||||
self.assertTrue(validate._validate_human_readable_string(good))
|
||||
|
||||
self.assertFalse(validate._validate_human_readable_string(curie))
|
||||
self.assertFalse(validate._validate_human_readable_string(ensg))
|
||||
self.assertFalse(validate._validate_human_readable_string(enst))
|
||||
|
||||
def test_validate_curie(self):
|
||||
|
||||
self.assertTrue(validate._validate_curie("UBERON:00001", ["UBERON", "EFO"]))
|
||||
self.assertTrue(validate._validate_curie("HsapDv:00002", ["HsapDv"]))
|
||||
|
||||
self.assertFalse(validate._validate_curie("HsapDv:00002", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("EFO:00002 (organoid)", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("EFO:00002 extra", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("UBERON:ABCD", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("Uberon:00002", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("UBERON:", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("UBERON", ["UBERON", "EFO"]))
|
||||
|
||||
def test_validate_suffixed_curie(self):
|
||||
|
||||
self.assertTrue(validate._validate_suffixed_curie("EFO:00001", ["UBERON", "EFO"]))
|
||||
self.assertTrue(validate._validate_suffixed_curie("UBERON:00001 (cell culture)", ["UBERON", "EFO"]))
|
||||
|
||||
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002 (organoid)", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002(organoid)", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("EFO:00002 extra", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("UBERON:ABCD", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("Uberon:00002", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("UBERON:", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("UBERON", ["UBERON", "EFO"]))
|
||||
|
||||
|
||||
class TestColumnValidation(unittest.TestCase):
|
||||
|
||||
def test_validate_unique(self):
|
||||
unique = pd.DataFrame([["abc", "def"], ["ghi", "jkl"], ["mnop", "qrs"]],
|
||||
index=["X", "Y", "Z"], columns=["col1", "col2"])
|
||||
duped = pd.DataFrame([["abc", "def"], ["ghi", "qrs"], ["abc", "qrs"]],
|
||||
index=["X", "Y", "X"], columns=["col1", "col2"])
|
||||
|
||||
schema_def = {"unique": True}
|
||||
|
||||
errors = validate._validate_column(unique.index, "index", "unique_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
errors = validate._validate_column(duped.index, "index", "duped_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("is not unique", errors[0])
|
||||
|
||||
errors = validate._validate_column(unique["col1"], "col1", "unique_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
errors = validate._validate_column(duped["col1"], "col1", "duped_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("is not unique", errors[0])
|
||||
|
||||
schema_def = {"unique": False}
|
||||
errors = validate._validate_column(duped["col1"], "col1", "duped_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
def test_validate_nullable(self):
|
||||
non_null = pd.DataFrame([["abc", "def"], ["ghi", "jkl"], ["mnop", "qrs"]],
|
||||
index=["X", "Y", "Z"], columns=["col1", "col2"])
|
||||
has_null = pd.DataFrame([["abc", "", None], ["ghi", "jkl", 1], ["mnop", "qrs", 2]],
|
||||
index=["X", "Y", "Z"], columns=["col1", "col2", "col3"])
|
||||
|
||||
schema_def = {"nullable": False}
|
||||
errors = validate._validate_column(non_null["col1"], "col1", "nonnull_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
errors = validate._validate_column(has_null["col1"], "col1", "hasnull_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
errors = validate._validate_column(has_null["col2"], "col2", "hasnull_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("contains empty values", errors[0])
|
||||
errors = validate._validate_column(has_null["col3"], "col3", "hasnull_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("contains empty values", errors[0])
|
||||
|
||||
schema_def = {"nullable": True}
|
||||
errors = validate._validate_column(has_null["col2"], "col2", "hasnull_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
def test_human_readable(self):
|
||||
hr_df = pd.DataFrame(
|
||||
[["for you, a human", "UBERON:12345", "UBERON:1234 (thundercat)"],
|
||||
["hope you're well", "bit of lungs", "brain"]],
|
||||
index=["ENSG00001", "ENSG00002"],
|
||||
columns=["good", "curie", "suffixed_curie"])
|
||||
|
||||
schema_def = {"type": "human-readable string"}
|
||||
errors = validate._validate_column(hr_df["good"], "good", "hr", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
errors = validate._validate_column(hr_df["curie"], "curie", "hr", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("non-human-readable", errors[0])
|
||||
|
||||
errors = validate._validate_column(hr_df["suffixed_curie"], "suffixed_curie", "hr", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("non-human-readable", errors[0])
|
||||
|
||||
errors = validate._validate_column(hr_df.index, "ensg", "hr", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("non-human-readable", errors[0])
|
||||
|
||||
def test_curie(self):
|
||||
|
||||
curie_df = pd.DataFrame(
|
||||
[["EFO:00001", "HsapDv:00001 (cell culture)", "EFO:", "MONDO:0001 cell culture"],
|
||||
["UBERON:00002", "HsapDv:00002 (organoid)", "EFO:12345", "MONDO:0002 (baba yaga)"],
|
||||
["EFO:0000000005", "HsapDv:000004 (humanzee)", "EFO:000002", "MONDO:0004 (TMNT)"]],
|
||||
index=["X", "Y", "Z"],
|
||||
columns=["good", "good_suffix", "bad", "bad_suffix"])
|
||||
|
||||
# Good
|
||||
schema_def = {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
errors = validate._validate_column(curie_df["good"], "good", "curie_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Good suffix
|
||||
schema_def = {"type": "suffixed curie", "prefixes": ["HsapDv", "WHATEVER"]}
|
||||
errors = validate._validate_column(curie_df["good_suffix"], "good_suffix", "curie_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Bad prefix
|
||||
schema_def = {"type": "curie", "prefixes": ["EFO"]}
|
||||
errors = validate._validate_column(curie_df["good"], "good", "curie_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("invalid ontology", errors[0])
|
||||
self.assertIn("must be curies from one of these", errors[0])
|
||||
|
||||
# Bad curies
|
||||
schema_def = {"type": "curie", "prefixes": ["EFO"]}
|
||||
errors = validate._validate_column(curie_df["bad"], "bad", "curie_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("invalid ontology", errors[0])
|
||||
|
||||
# Bad suffixes
|
||||
schema_def = {"type": "suffixed curie", "prefixes": ["EFO"]}
|
||||
errors = validate._validate_column(curie_df["bad_suffix"], "bad_suffix", "curie_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("invalid ontology", errors[0])
|
||||
|
||||
def test_enum(self):
|
||||
enum_df = pd.DataFrame(
|
||||
[["abc", "ghi"],
|
||||
["def", "jkl"]],
|
||||
index=["X", "Y"],
|
||||
columns=["col1", "col2"])
|
||||
|
||||
# All match
|
||||
schema_def = {"type": "string", "enum": ["abc", "def", "xyz"]}
|
||||
errors = validate._validate_column(enum_df["col1"], "col1", "enum_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Missing value
|
||||
schema_def = {"type": "string", "enum": ["abc", "xyz"]}
|
||||
errors = validate._validate_column(enum_df["col1"], "col1", "enum_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("unpermitted values", errors[0])
|
||||
|
||||
|
||||
class TestDictValidations(unittest.TestCase):
|
||||
|
||||
|
||||
def test_key_presence(self):
|
||||
|
||||
schema_def = {"keys": {"abc": None, "def": None}}
|
||||
|
||||
dict_ = {"abc": "123", "def": "456"}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Missing keys are bad
|
||||
dict_ = {"abc": "123"}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("missing key", errors[0])
|
||||
|
||||
# Extra keys are okay
|
||||
dict_ = {"abc": "123", "def": "456", "xyz": "789"}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Better not be empty come on
|
||||
dict_ = {}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 2)
|
||||
|
||||
def test_nullable(self):
|
||||
|
||||
schema_def = {"keys": {"abc": {"type": "string", "nullable": False},
|
||||
"def": {"type": "string", "nullable": True}}}
|
||||
|
||||
dict_ = {"abc": "xyz", "def": ""}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
dict_ = {"abc": "", "def": ""}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("empty value", errors[0])
|
||||
|
||||
def test_recurse(self):
|
||||
|
||||
schema_def = {
|
||||
"keys": {
|
||||
"subdict": {
|
||||
"type": "dict",
|
||||
"keys": {
|
||||
"subdict_key1": None,
|
||||
"subdict_key2": None
|
||||
}
|
||||
},
|
||||
"ontology": {
|
||||
"type": "curie",
|
||||
"prefixes": ["ONTOLOGY"]
|
||||
},
|
||||
"blob": {
|
||||
"type": "stringified list of dicts"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
|
||||
"ontology": "ONTOLOGY:123456",
|
||||
"blob": json.dumps([{"abc": 123}, {"def": 456}])
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any"},
|
||||
"ontology": "ONTOLOGY:123456",
|
||||
"blob": json.dumps([{"abc": 123}, {"def": 456}])
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("missing key", errors[0])
|
||||
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
|
||||
"ontology": "oh no not an ontology term",
|
||||
"blob": json.dumps([{"abc": 123}, {"def": 456}])
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("invalid ontology", errors[0])
|
||||
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
|
||||
"ontology": "ONTOLOGY:123456",
|
||||
"blob": [{"abc": 123}, {"def": 456}]
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("JSON-encoded list of dicts", errors[0])
|
||||
|
||||
# Multiple errors
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any"},
|
||||
"ontology": "oh no not an ontology term",
|
||||
"blob": json.dumps([{"abc": 123}, {"def": 456}])
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 2)
|
||||
|
||||
|
||||
class TestDataframeValidation(unittest.TestCase):
|
||||
|
||||
def test_column_presence(self):
|
||||
df = pd.DataFrame(
|
||||
[["abc", "EFO:123"],
|
||||
["def", "UBERON:456"]],
|
||||
columns=["hr_string", "ontology"],
|
||||
index=["X", "Y"]
|
||||
)
|
||||
|
||||
schema_def = {
|
||||
"columns": {
|
||||
"hr_string": {"type": "human-readable string"},
|
||||
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
}
|
||||
}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
schema_def = {
|
||||
"columns": {
|
||||
"hr_string": {"type": "human-readable string"},
|
||||
"another_hr_string": {"type": "human-readable string"},
|
||||
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
}
|
||||
}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("missing column", errors[0])
|
||||
|
||||
# Extra is okay
|
||||
df = pd.DataFrame(
|
||||
[["abc", "EFO:123", "extra"],
|
||||
["def", "UBERON:456", "extra"]],
|
||||
columns=["hr_string", "ontology", "extra"],
|
||||
index=["X", "Y"]
|
||||
)
|
||||
schema_def = {
|
||||
"columns": {
|
||||
"hr_string": {"type": "human-readable string"},
|
||||
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
}
|
||||
}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
|
||||
def test_index(self):
|
||||
df = pd.DataFrame(
|
||||
[["abc", "123"],
|
||||
["def", "456"]],
|
||||
columns=["col1", "col2"],
|
||||
index=["ENSG0001", "ENSG0002"]
|
||||
)
|
||||
|
||||
schema_def = {"index": {"unique": True}}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
schema_def = {"index": {"type": "human-readable string"}}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("non-human-readable", errors[0])
|
||||
|
||||
df = pd.DataFrame(
|
||||
[["abc", "123"],
|
||||
["def", "456"]],
|
||||
columns=["col1", "col2"],
|
||||
index=["ENSG0001", "ENSG0001"]
|
||||
)
|
||||
schema_def = {"index": {"unique": True}}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("is not unique", errors[0])
|
||||
|
||||
def test_recurse(self):
|
||||
|
||||
df = pd.DataFrame(
|
||||
[["abc", "HsapDv:0001"],
|
||||
["EFO:123", "UBERON:456"]],
|
||||
columns=["hr_string", "ontology"],
|
||||
index=["X", "Y"]
|
||||
)
|
||||
schema_def = {
|
||||
"columns": {
|
||||
"hr_string": {"type": "human-readable string"},
|
||||
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
}
|
||||
}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertEqual(len(errors), 2)
|
||||
self.assertEqual(len([e for e in errors if "non-human-readable" in e]), 1)
|
||||
self.assertEqual(len([e for e in errors if "invalid ontology" in e]), 1)
|
||||
|
||||
|
||||
class TestGetSchema(unittest.TestCase):
|
||||
|
||||
def test_get_schema(self):
|
||||
self.assertIsInstance(validate.get_schema_definition("1.0.0"), dict)
|
||||
|
||||
with self.assertRaises(ValueError):
|
||||
validate.get_schema_definition("10.1.5")
|
||||
|
||||
|
||||
class TestValidate(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.source_h5ad_path = f"{PROJECT_ROOT}/backend/test/fixtures/pbmc3k-CSC-gz.h5ad"
|
||||
|
||||
def test_shallow(self):
|
||||
|
||||
adata = sc.read_h5ad(self.source_h5ad_path)
|
||||
self.assertFalse(validate.validate_adata(adata, True))
|
||||
|
||||
adata.uns["version"] = {
|
||||
"corpora_schema_version": "1.0.0",
|
||||
"corpora_encoding_version": "0.1.0"
|
||||
}
|
||||
self.assertTrue(validate.validate_adata(adata, True))
|
||||
|
||||
def test_deep(self):
|
||||
adata = sc.read_h5ad(self.source_h5ad_path)
|
||||
self.assertFalse(validate.validate_adata(adata, False))
|
||||
|
||||
adata.uns["version"] = {
|
||||
"corpora_schema_version": "1.0.0",
|
||||
"corpora_encoding_version": "0.1.0"
|
||||
}
|
||||
self.assertFalse(validate.validate_adata(adata, False))
|
||||
Reference in New Issue
Block a user