mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-02 08:08:11 +08:00
Add schema subcommand (#1939)
Add the `cellxgene schema apply` and `cellxgene schema validate` subcommands. The first takes an h5ad file and a yaml with config information and produces a new h5ad that follows the cellxgene data integration schema. The second takes an h5ad and checks if it follows the schema version written into its metadata. Both are currently marked as "experimental" as the primary intended users are still at CZI.
This commit is contained in:
BIN
Binary file not shown.
+139
@@ -0,0 +1,139 @@
|
||||
#!/bin/bash
|
||||
wget "https://s3-us-west-2.amazonaws.com/10x.files/samples/cell/pbmc3k/pbmc3k_filtered_gene_bc_matrices.tar.gz"
|
||||
tar xf "pbmc3k_filtered_gene_bc_matrices.tar.gz"
|
||||
|
||||
python3 - <<MERGE_GENES
|
||||
import os
|
||||
from scipy.io import mmread, mmwrite
|
||||
import scipy.sparse
|
||||
import pandas as pd
|
||||
from server.converters.schema import gene_symbol
|
||||
|
||||
mat = mmread("filtered_gene_bc_matrices/hg19/matrix.mtx").todense()
|
||||
genes = pd.read_csv("filtered_gene_bc_matrices/hg19/genes.tsv", sep='\t', names=["gene_id", "gene_symbol"])
|
||||
|
||||
upgraded_genes = gene_symbol.get_upgraded_var_index(pd.DataFrame(index=genes["gene_symbol"]))
|
||||
df = pd.DataFrame(data=mat, index=upgraded_genes).T
|
||||
merged = df.sum(axis=1, level=0, skipna=False)
|
||||
|
||||
os.makedirs("merged")
|
||||
merged.columns.to_frame().to_csv("merged/genes.tsv", index=False, header=False)
|
||||
mmwrite("merged/matrix.mtx", scipy.sparse.coo_matrix(merged).T)
|
||||
MERGE_GENES
|
||||
|
||||
cp "filtered_gene_bc_matrices/hg19/barcodes.tsv" "merged/barcodes.tsv"
|
||||
awk '{print $1"\t"$1}' merged/genes.tsv > genes_tmp.tsv; mv genes_tmp.tsv merged/genes.tsv
|
||||
|
||||
echo -e "\n\n\nRunning tutorial on original\n\n\n"
|
||||
Rscript - <<TUTORIAL
|
||||
library(Seurat)
|
||||
|
||||
pbmc.data <- Read10X(data.dir = "filtered_gene_bc_matrices/hg19/")
|
||||
pbmc <- CreateSeuratObject(counts = pbmc.data, project = "pbmc3k", min.features = 200)
|
||||
pbmc <- NormalizeData(pbmc, normalization.method = "LogNormalize", scale.factor = 10000)
|
||||
pbmc <- FindVariableFeatures(pbmc, selection.method = "vst", nfeatures = 2000)
|
||||
pbmc[["percent.mt"]] <- PercentageFeatureSet(pbmc, pattern = "^MT-")
|
||||
all.genes <- rownames(pbmc)
|
||||
pbmc <- ScaleData(pbmc, features = all.genes)
|
||||
|
||||
pbmc <- RunPCA(pbmc, features = VariableFeatures(object = pbmc))
|
||||
pbmc <- FindNeighbors(pbmc, dims = 1:10)
|
||||
pbmc <- FindClusters(pbmc, resolution = 0.5)
|
||||
pbmc <- RunUMAP(pbmc, dims = 1:10)
|
||||
saveRDS(pbmc, file = "./seurat_tutorial.rds")
|
||||
TUTORIAL
|
||||
|
||||
echo -e "\n\n\nRunning tutorial on merged\n\n\n"
|
||||
Rscript - <<TUTORIAL_MERGED
|
||||
library(Seurat)
|
||||
|
||||
pbmc.data <- Read10X(data.dir = "merged/")
|
||||
pbmc <- CreateSeuratObject(counts = pbmc.data, project = "pbmc3k", min.features = 200)
|
||||
pbmc <- NormalizeData(pbmc, normalization.method = "LogNormalize", scale.factor = 10000)
|
||||
pbmc <- FindVariableFeatures(pbmc, selection.method = "vst", nfeatures = 2000)
|
||||
pbmc[["percent.mt"]] <- PercentageFeatureSet(pbmc, pattern = "^MT-")
|
||||
all.genes <- rownames(pbmc)
|
||||
pbmc <- ScaleData(pbmc, features = all.genes)
|
||||
|
||||
pbmc <- RunPCA(pbmc, features = VariableFeatures(object = pbmc))
|
||||
pbmc <- FindNeighbors(pbmc, dims = 1:10)
|
||||
pbmc <- FindClusters(pbmc, resolution = 0.5)
|
||||
pbmc <- RunUMAP(pbmc, dims = 1:10)
|
||||
saveRDS(pbmc, file = "./seurat_tutorial_merged.rds")
|
||||
TUTORIAL_MERGED
|
||||
|
||||
echo -e "\n\n\nRunning SCTransform on original\n\n\n"
|
||||
Rscript - <<SCTRANSFORM
|
||||
library(Seurat)
|
||||
library(sctransform)
|
||||
|
||||
pbmc.data <- Read10X(data.dir = "filtered_gene_bc_matrices/hg19/")
|
||||
pbmc <- CreateSeuratObject(counts = pbmc.data)
|
||||
pbmc <- PercentageFeatureSet(pbmc, pattern = "^MT-", col.name = "percent.mt")
|
||||
pbmc <- SCTransform(pbmc, vars.to.regress = "percent.mt", verbose = FALSE)
|
||||
pbmc <- RunPCA(pbmc, verbose = FALSE)
|
||||
pbmc <- RunUMAP(pbmc, dims = 1:30, verbose = FALSE)
|
||||
pbmc <- FindNeighbors(pbmc, dims = 1:30, verbose = FALSE)
|
||||
pbmc <- FindClusters(pbmc, verbose = FALSE)
|
||||
saveRDS(pbmc, file = "./sctransform.rds")
|
||||
SCTRANSFORM
|
||||
|
||||
echo -e "\n\n\nRunning SCTransform on merged\n\n\n"
|
||||
Rscript - <<SCTRANSFORM_MERGED
|
||||
library(Seurat)
|
||||
library(sctransform)
|
||||
|
||||
pbmc.data <- Read10X(data.dir = "merged/")
|
||||
pbmc <- CreateSeuratObject(counts = pbmc.data)
|
||||
pbmc <- PercentageFeatureSet(pbmc, pattern = "^MT-", col.name = "percent.mt")
|
||||
pbmc <- SCTransform(pbmc, vars.to.regress = "percent.mt", verbose = FALSE)
|
||||
pbmc <- RunPCA(pbmc, verbose = FALSE)
|
||||
pbmc <- RunUMAP(pbmc, dims = 1:30, verbose = FALSE)
|
||||
pbmc <- FindNeighbors(pbmc, dims = 1:30, verbose = FALSE)
|
||||
pbmc <- FindClusters(pbmc, verbose = FALSE)
|
||||
saveRDS(pbmc, file = "./sctransform_merged.rds")
|
||||
SCTRANSFORM_MERGED
|
||||
|
||||
echo -e "\n\n\nConverting\n\n\n"
|
||||
Rscript - <<SCEASY
|
||||
library(sceasy)
|
||||
srt <- readRDS("seurat_tutorial.rds")
|
||||
sceasy::convertFormat(srt,
|
||||
outFile = "seurat_tutorial.h5ad",
|
||||
from = "seurat",
|
||||
to = "anndata",
|
||||
assay = "RNA",
|
||||
main_layer = "data",
|
||||
transfer_layers = c("data", "counts", "scale.data"),
|
||||
drop_single_values = FALSE)
|
||||
|
||||
srt <- readRDS("seurat_tutorial_merged.rds")
|
||||
sceasy::convertFormat(srt,
|
||||
outFile = "seurat_tutorial_merged.h5ad",
|
||||
from = "seurat",
|
||||
to = "anndata",
|
||||
assay = "RNA",
|
||||
main_layer = "data",
|
||||
transfer_layers = c("data", "counts", "scale.data"),
|
||||
drop_single_values = FALSE)
|
||||
|
||||
srt <- readRDS("sctransform.rds")
|
||||
sceasy::convertFormat(srt,
|
||||
outFile = "sctransform.h5ad",
|
||||
from = "seurat",
|
||||
to = "anndata",
|
||||
assay = "SCT",
|
||||
main_layer = "data",
|
||||
transfer_layers = c("data", "counts", "scale.data"),
|
||||
drop_single_values = FALSE)
|
||||
|
||||
srt <- readRDS("sctransform_merged.rds")
|
||||
sceasy::convertFormat(srt,
|
||||
outFile = "sctransform_merged.h5ad",
|
||||
from = "seurat",
|
||||
to = "anndata",
|
||||
assay = "SCT",
|
||||
main_layer = "data",
|
||||
transfer_layers = c("data", "counts", "scale.data"),
|
||||
drop_single_values = FALSE)
|
||||
SCEASY
|
||||
+34
@@ -0,0 +1,34 @@
|
||||
fixup_gene_symbols:
|
||||
X: log1p
|
||||
obs:
|
||||
cell_type_ontology_term_id:
|
||||
louvain:
|
||||
CD4 T cells: CL:00001
|
||||
B cells: CL:00002
|
||||
CD14+ Monocytes: CL:00003
|
||||
NK cells: CL:00004
|
||||
CD8 T cells: CL:00005
|
||||
FCGR3A+ Monocytes: CL:00006
|
||||
Dendritic cells: CL:00007
|
||||
Megakaryocytes: CL:00008
|
||||
tissue_ontology_term_id: UBERON:12345
|
||||
assay_ontology_term_id: EFO:12345
|
||||
disease_ontology_term_id: MONDO:12345
|
||||
ethnicity_ontology_term_id: MANCESTRO:12345
|
||||
development_stage_ontology_term_id: HsapDv:12345
|
||||
sex: other
|
||||
uns:
|
||||
version:
|
||||
corpora_schema_version: 1.0.0
|
||||
corpora_encoding_version: 0.1.0
|
||||
organism_ontology_term_id: NCBITaxon:9606
|
||||
title: Test dataset
|
||||
contributors:
|
||||
- name: Marcus
|
||||
institution: CZI
|
||||
layer_descriptions:
|
||||
X: raw
|
||||
project_links:
|
||||
- link_url: https://chanzuckerberg.com/
|
||||
link_name: CZI
|
||||
link_type: SUMMARY
|
||||
+34
@@ -0,0 +1,34 @@
|
||||
fixup_gene_symbols:
|
||||
X: log1p
|
||||
obs:
|
||||
cell_type_ontology_term_id:
|
||||
louvain:
|
||||
CD4 T cells: CL:00001
|
||||
B cells: CL:00002
|
||||
CD14+ Monocytes: CL:00003
|
||||
NK cells: CL:00004
|
||||
CD8 T cells: CL:00005
|
||||
FCGR3A+ Monocytes: CL:00006
|
||||
Dendritic cells: CL:00007
|
||||
Megakaryocytes: CL:00008
|
||||
tissue_ontology_term_id: UBERON:12345
|
||||
assay_ontology_term_id: EFO:12345
|
||||
disease_ontology_term_id: MONDO:12345
|
||||
ethnicity_ontology_term_id: HANCESTRO:12345
|
||||
development_stage_ontology_term_id: HsapDv:12345
|
||||
sex: other
|
||||
uns:
|
||||
version:
|
||||
corpora_schema_version: 1.0.0
|
||||
corpora_encoding_version: 0.1.0
|
||||
organism_ontology_term_id: NCBITaxon:9606
|
||||
title: Test dataset
|
||||
contributors:
|
||||
- name: Marcus
|
||||
institution: CZI
|
||||
layer_descriptions:
|
||||
X: raw
|
||||
project_links:
|
||||
- link_url: https://chanzuckerberg.com/
|
||||
link_name: CZI
|
||||
link_type: SUMMARY
|
||||
@@ -0,0 +1,56 @@
|
||||
import os
|
||||
import unittest
|
||||
|
||||
import pandas as pd
|
||||
|
||||
from server.test import FIXTURES_ROOT
|
||||
from server.converters.schema import gene_symbol
|
||||
|
||||
|
||||
class TestHGNCSymbolChecker(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.test_hgnc_path = os.path.join(FIXTURES_ROOT, "hgnc_example.txt.gz")
|
||||
self.hgnc_checker = gene_symbol.HGNCSymbolChecker.from_hgnc_records(self.test_hgnc_path)
|
||||
|
||||
def test_symbol_upgrade(self):
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("SEPT1"), "SEPTIN1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("ADRB2R"), "ADRB2")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("BAR"), "ADRB2")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("sept1"), "SEPTIN1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("AdRb2R"), "ADRB2")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("bar"), "ADRB2")
|
||||
|
||||
# Strip off seurat endings when appropriate
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("SEPT1.1"), "SEPTIN1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("ADRB2-1"), "ADRB2")
|
||||
|
||||
# DIFF6 is ambiguous so don't upgrade it
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("DIFF6"), "DIFF6")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("diff6"), "diff6")
|
||||
|
||||
# ARG1 is approved
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("ARG1"), "ARG1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("arg1"), "ARG1")
|
||||
|
||||
# HAP1 is both approved and withdrawn
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("HAP1"), "HAP1")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("hap1"), "HAP1")
|
||||
|
||||
# Leave unknown symbols alone
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("NOTASYMBOL"), "NOTASYMBOL")
|
||||
self.assertEqual(self.hgnc_checker.upgrade_symbol("notasymbol"), "notasymbol")
|
||||
|
||||
def test_check_symbol(self):
|
||||
self.assertEqual(self.hgnc_checker.check_symbol("SEPT1"), gene_symbol.SymbolStatus.UPGRADABLE)
|
||||
self.assertEqual(self.hgnc_checker.check_symbol("DIFF6"), gene_symbol.SymbolStatus.AMBIGUOUS)
|
||||
self.assertEqual(self.hgnc_checker.check_symbol("NOTASYMBOL"), gene_symbol.SymbolStatus.UNKNOWN)
|
||||
|
||||
# HAP1 is one of the approved and withdrawn symbols
|
||||
self.assertEqual(self.hgnc_checker.check_symbol("HAP1"), gene_symbol.SymbolStatus.APPROVED)
|
||||
|
||||
def test_upgrade_index(self):
|
||||
index = pd.Index(["SEPT1", "DIFF6", "NOTASYMBOL", "bar", "SEPTIN1"])
|
||||
var_df = pd.DataFrame([[0] * len(index)], index=index)
|
||||
upgraded_index = gene_symbol.get_upgraded_var_index(var_df, hgnc_path=self.test_hgnc_path)
|
||||
self.assertEqual(upgraded_index.tolist(), ["SEPTIN1", "DIFF6", "NOTASYMBOL", "ADRB2", "SEPTIN1"])
|
||||
@@ -0,0 +1,129 @@
|
||||
import json
|
||||
|
||||
import unittest
|
||||
import unittest.mock
|
||||
|
||||
from server.converters.schema import ontology
|
||||
|
||||
|
||||
class TestOntologyParsing(unittest.TestCase):
|
||||
def setUp(self):
|
||||
|
||||
self.curies = ["UBERON:0002048", "HsapDv:0000174", "NCBITaxon:9606", "EFO:0008995"]
|
||||
|
||||
self.names = ["UBERON", "HsapDv", "NCBITaxon", "EFO"]
|
||||
|
||||
self.values = ["0002048", "0000174", "9606", "0008995"]
|
||||
|
||||
self.iris = [
|
||||
"http://purl.obolibrary.org/obo/UBERON_0002048",
|
||||
"http://purl.obolibrary.org/obo/HsapDv_0000174",
|
||||
"http://purl.obolibrary.org/obo/NCBITaxon_9606",
|
||||
"http://www.ebi.ac.uk/efo/EFO_0008995",
|
||||
]
|
||||
|
||||
URL_ROOT = "http://www.ebi.ac.uk/ols/api/ontologies/"
|
||||
self.urls = [
|
||||
URL_ROOT + "UBERON/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FUBERON_0002048",
|
||||
URL_ROOT + "HsapDv/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FHsapDv_0000174",
|
||||
URL_ROOT + "NCBITaxon/terms/http%253A%252F%252Fpurl.obolibrary.org%252Fobo%252FNCBITaxon_9606",
|
||||
URL_ROOT + "EFO/terms/http%253A%252F%252Fwww.ebi.ac.uk%252Fefo%252FEFO_0008995",
|
||||
]
|
||||
|
||||
self.responses = {
|
||||
"UBERON:0002048": {
|
||||
"iri": "http://purl.obolibrary.org/obo/UBERON_0002048",
|
||||
"description": ["Respiration organ that develops as an outpocketing of the esophagus."],
|
||||
"label": "lung",
|
||||
},
|
||||
"HsapDv:0000174": {
|
||||
"iri": "http://purl.obolibrary.org/obo/HsapDv_0000174",
|
||||
"description": ["Infant stage that refers to an infant who is over 1 and under 2 months old."],
|
||||
"label": "1-month-old human stage",
|
||||
},
|
||||
"NCBITaxon:9606": {
|
||||
"iri": "http://purl.obolibrary.org/obo/NCBITaxon_9606",
|
||||
"description": None,
|
||||
"label": "Homo sapiens",
|
||||
},
|
||||
"EFO:0008995": {
|
||||
"iri": "http://www.ebi.ac.uk/efo/EFO_0008995",
|
||||
"description": [
|
||||
(
|
||||
'10X is a "synthetic long-read" technology and works by capturing a barcoded oligo-coated '
|
||||
"gel-bead and 0.3x genome copies into a single emulsion droplet, processing the equivalent "
|
||||
"of 1 million pipetting steps. Successive versions of the 10x chemistry use different "
|
||||
"barcode locations to improve the sequencing yield and quality of 10x experiments."
|
||||
)
|
||||
],
|
||||
"label": "10X sequencing",
|
||||
},
|
||||
}
|
||||
|
||||
def test_ontololgy_name(self):
|
||||
for curie, expected_name in zip(self.curies, self.names):
|
||||
self.assertEqual(ontology._ontology_name(curie), expected_name)
|
||||
|
||||
def test_ontololgy_value(self):
|
||||
for curie, expected_value in zip(self.curies, self.values):
|
||||
self.assertEqual(ontology._ontology_value(curie), expected_value)
|
||||
|
||||
def test_iri(self):
|
||||
for curie, expected_iri in zip(self.curies, self.iris):
|
||||
self.assertEqual(ontology._iri(curie), expected_iri)
|
||||
|
||||
def test_ontology_info_url(self):
|
||||
for curie, expected_url in zip(self.curies, self.urls):
|
||||
self.assertEqual(ontology._ontology_info_url(curie), expected_url)
|
||||
|
||||
def test_empty_ontology_info_url(self):
|
||||
self.assertEqual(ontology._ontology_info_url(""), "")
|
||||
|
||||
|
||||
class TestOntologyLookup(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.responses = {
|
||||
"UBERON:0002048": {
|
||||
"iri": "http://purl.obolibrary.org/obo/UBERON_0002048",
|
||||
"description": ["Respiration organ that develops as an outpocketing of the esophagus."],
|
||||
"label": "lung",
|
||||
},
|
||||
"HsapDv:0000174": {
|
||||
"iri": "http://purl.obolibrary.org/obo/HsapDv_0000174",
|
||||
"description": ["Infant stage that refers to an infant who is over 1 and under 2 months old."],
|
||||
"label": "1-month-old human stage",
|
||||
},
|
||||
"NCBITaxon:9606": {
|
||||
"iri": "http://purl.obolibrary.org/obo/NCBITaxon_9606",
|
||||
"description": None,
|
||||
"label": "Homo sapiens",
|
||||
},
|
||||
"EFO:0008995": {
|
||||
"iri": "http://www.ebi.ac.uk/efo/EFO_0008995",
|
||||
"description": [
|
||||
('10X is a "synthetic long-read" technology and works by capturing a barcoded oligo-coated '
|
||||
'gel-bead and 0.3x genome copies into a single emulsion droplet, processing the equivalent '
|
||||
'of 1 million pipetting steps. Successive versions of the 10x chemistry use different barcode '
|
||||
'locations to improve the sequencing yield and quality of 10x experiments.')
|
||||
],
|
||||
"label": "10X sequencing",
|
||||
},
|
||||
}
|
||||
|
||||
self.labels = {
|
||||
"UBERON:0002048": "lung",
|
||||
"HsapDv:0000174": "1-month-old human stage",
|
||||
"NCBITaxon:9606": "Homo sapiens",
|
||||
"EFO:0008995": "10X sequencing",
|
||||
}
|
||||
|
||||
@unittest.mock.patch("requests.get")
|
||||
def test_lookup_label(self, mock_get):
|
||||
|
||||
for curie, response in self.responses.items():
|
||||
mock_get.return_value.content = json.dumps(response)
|
||||
mock_get.return_value.json.return_value = response
|
||||
mock_get.return_value.status_code = 200
|
||||
|
||||
label = ontology.get_ontology_label(curie)
|
||||
self.assertEqual(label, self.labels[curie])
|
||||
@@ -0,0 +1,257 @@
|
||||
import json
|
||||
import os
|
||||
import unittest
|
||||
import unittest.mock
|
||||
|
||||
import anndata
|
||||
import numpy
|
||||
import pandas as pd
|
||||
import scanpy as sc
|
||||
|
||||
from server.converters.schema import remix
|
||||
|
||||
PROJECT_ROOT = os.popen("git rev-parse --show-toplevel").read().strip()
|
||||
|
||||
|
||||
class TestApplySchema(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.source_h5ad_path = f"{PROJECT_ROOT}/server/test/fixtures/pbmc3k-CSC-gz.h5ad"
|
||||
self.output_h5ad_path = f"{PROJECT_ROOT}/server/test/fixtures/test_remix.h5ad"
|
||||
self.config_path = f"{PROJECT_ROOT}/server/test/fixtures/test_config.yaml"
|
||||
self.bad_config_path = f"{PROJECT_ROOT}/server/test/fixtures/test_bad_config.yaml"
|
||||
|
||||
def tearDown(self):
|
||||
try:
|
||||
os.remove(self.output_h5ad_path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
@unittest.mock.patch("server.converters.schema.ontology.get_ontology_label")
|
||||
def test_apply_schema(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "test label"
|
||||
remix.apply_schema(self.source_h5ad_path, self.config_path, self.output_h5ad_path)
|
||||
new_adata = sc.read_h5ad(self.output_h5ad_path)
|
||||
|
||||
self.assertIn("cell_type", new_adata.obs.columns)
|
||||
self.assertListEqual(["test label"], new_adata.obs["cell_type"].unique().tolist())
|
||||
self.assertListEqual(
|
||||
["CL:00001", "CL:00002", "CL:00003", "CL:00004", "CL:00005", "CL:00006", "CL:00007", "CL:00008"],
|
||||
sorted(new_adata.obs["cell_type_ontology_term_id"].unique().tolist())
|
||||
)
|
||||
|
||||
self.assertIn("version", new_adata.uns_keys())
|
||||
|
||||
@unittest.mock.patch("server.converters.schema.ontology.get_ontology_label")
|
||||
def test_apply_bad_schema(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "test label"
|
||||
remix.apply_schema(self.source_h5ad_path, self.bad_config_path, self.output_h5ad_path)
|
||||
new_adata = sc.read_h5ad(self.output_h5ad_path)
|
||||
|
||||
# Should refuse to write the version
|
||||
self.assertNotIn("version", new_adata.uns_keys())
|
||||
|
||||
class TestFieldParsing(unittest.TestCase):
|
||||
|
||||
def test_is_curie(self):
|
||||
self.assertTrue(remix.is_curie("EFO:00001"))
|
||||
self.assertTrue(remix.is_curie("UBERON:123456"))
|
||||
self.assertTrue(remix.is_curie("HsapDv:0001"))
|
||||
self.assertFalse(remix.is_curie("UBERON"))
|
||||
self.assertFalse(remix.is_curie("UBERON:"))
|
||||
self.assertFalse(remix.is_curie("123456"))
|
||||
|
||||
def test_is_ontology_field(self):
|
||||
self.assertTrue(remix.is_ontology_field("tissue_ontology_term_id"))
|
||||
self.assertTrue(remix.is_ontology_field("cell_type_ontology_term_id"))
|
||||
self.assertFalse(remix.is_ontology_field("cell_ontology"))
|
||||
self.assertFalse(remix.is_ontology_field("method"))
|
||||
|
||||
def test_get_label_field_name(self):
|
||||
self.assertEqual("tissue", remix.get_label_field_name("tissue_ontology_term_id"))
|
||||
self.assertEqual("cell_type", remix.get_label_field_name("cell_type_ontology_term_id"))
|
||||
|
||||
def test_split_suffix(self):
|
||||
self.assertEqual(("UBERON:1234", " (organoid)"), remix.split_suffix("UBERON:1234 (organoid)"))
|
||||
self.assertEqual(("UBERON:1234", " (cell culture)"), remix.split_suffix("UBERON:1234 (cell culture)"))
|
||||
self.assertEqual(("UBERON:1234", ""), remix.split_suffix("UBERON:1234"))
|
||||
self.assertEqual(("UBERON:1234 (something)", ""), remix.split_suffix("UBERON:1234 (something)"))
|
||||
|
||||
@unittest.mock.patch("server.converters.schema.ontology.get_ontology_label")
|
||||
def test_get_curie_and_label(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "test label"
|
||||
self.assertEqual(
|
||||
remix.get_curie_and_label("UBERON:1234"),
|
||||
("UBERON:1234", "test label")
|
||||
)
|
||||
self.assertEqual(
|
||||
remix.get_curie_and_label("UBERON:1234 (cell culture)"),
|
||||
("UBERON:1234 (cell culture)", "test label (cell culture)")
|
||||
)
|
||||
self.assertEqual(
|
||||
remix.get_curie_and_label("whatever"),
|
||||
("", "whatever")
|
||||
)
|
||||
|
||||
|
||||
class TestManipulateAnndata(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
|
||||
self.cell_count = 20
|
||||
self.gene_count = 200
|
||||
X = numpy.random.randint(0, 1000, (self.cell_count, self.gene_count))
|
||||
uns = {"organism": "monkey", "experiment": "monkey experiment"}
|
||||
obs = pd.DataFrame(
|
||||
index=[f"Cell{d}" for d in range(self.cell_count)],
|
||||
columns=["tissue", "CellType"],
|
||||
data=[["lung", "epithelial"]] * (self.cell_count // 2) + [["lung", "endothelial"]] * (self.cell_count // 2)
|
||||
)
|
||||
var = pd.DataFrame(index=[f"SEPT{d}" for d in range(self.gene_count)])
|
||||
|
||||
self.adata = anndata.AnnData(X=X, obs=obs, var=var, uns=uns)
|
||||
|
||||
def test_safe_add_field(self):
|
||||
|
||||
remix.safe_add_field(self.adata.obs, "tissue", ["monkey lung"] * self.cell_count)
|
||||
self.assertEqual(self.adata.obs["tissue_original"].tolist(), ["lung"] * self.cell_count)
|
||||
self.assertEqual(self.adata.obs["tissue"].tolist(), ["monkey lung"] * self.cell_count)
|
||||
|
||||
remix.safe_add_field(self.adata.uns, "contributors", [{"name": "contributor1"}, {"name": "contributor2"}])
|
||||
self.assertEqual(
|
||||
self.adata.uns["contributors"],
|
||||
json.dumps([{"name": "contributor1"}, {"name": "contributor2"}])
|
||||
)
|
||||
|
||||
@unittest.mock.patch("server.converters.schema.ontology.get_ontology_label")
|
||||
def test_remix_uns(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "Pan troglodytes"
|
||||
uns_config = {
|
||||
"version": {
|
||||
"corpora_schema_version": "1.0.0",
|
||||
"corpora_encoding_version": "0.1.0"
|
||||
},
|
||||
"organism_ontology_term_id": "NCBITaxon:9598",
|
||||
"contributors": [
|
||||
{
|
||||
"name": "scientist",
|
||||
"email": "scientist@science.com"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
remix.remix_uns(self.adata, uns_config)
|
||||
|
||||
self.assertEqual(
|
||||
sorted(self.adata.uns_keys()),
|
||||
sorted(["organism_original", "organism", "organism_ontology_term_id",
|
||||
"contributors", "version", "experiment"])
|
||||
)
|
||||
|
||||
self.assertEqual(self.adata.uns['organism'], "Pan troglodytes")
|
||||
self.assertEqual(self.adata.uns['organism_original'], "monkey")
|
||||
self.assertEqual(self.adata.uns['organism_ontology_term_id'], "NCBITaxon:9598")
|
||||
self.assertEqual(self.adata.uns['contributors'],
|
||||
json.dumps([{"name": "scientist", "email": "scientist@science.com"}]))
|
||||
|
||||
@unittest.mock.patch("server.converters.schema.ontology.get_ontology_label")
|
||||
def test_remix_obs(self, mock_get_ontology_label):
|
||||
mock_get_ontology_label.return_value = "lung (in a monkey)"
|
||||
obs_config = {
|
||||
"tissue_ontology_term_id": {
|
||||
"tissue": {
|
||||
"lung": "UBERON:00000"
|
||||
}
|
||||
},
|
||||
"cell_color": {
|
||||
"CellType": {
|
||||
"epithelial": "fuschia",
|
||||
"endothelial": "khaki"
|
||||
}
|
||||
},
|
||||
"sex": "male"
|
||||
}
|
||||
|
||||
remix.remix_obs(self.adata, obs_config)
|
||||
self.assertEqual(
|
||||
sorted(self.adata.obs_keys()),
|
||||
sorted(["tissue", "tissue_ontology_term_id", "tissue_original", "CellType", "cell_color", "sex"])
|
||||
)
|
||||
|
||||
self.assertTrue(all(v == "lung" for v in self.adata.obs.tissue_original))
|
||||
self.assertTrue(all(v == "UBERON:00000" for v in self.adata.obs.tissue_ontology_term_id))
|
||||
self.assertTrue(all(v == "lung (in a monkey)" for v in self.adata.obs.tissue))
|
||||
self.assertTrue(all(v == "male" for v in self.adata.obs.sex))
|
||||
self.assertTrue(all(v in (("epithelial", "fuschia"), ("endothelial", "khaki"))
|
||||
for v in zip(self.adata.obs.CellType, self.adata.obs.cell_color)))
|
||||
|
||||
|
||||
class TestFixupGeneSymbols(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.seurat_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/seurat_tutorial.h5ad"
|
||||
self.seurat_merged_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/seurat_tutorial_merged.h5ad"
|
||||
self.sctransform_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/sctransform.h5ad"
|
||||
self.sctransform_merged_path = f"{PROJECT_ROOT}/server/test/fixtures/schema_test_data/sctransform_merged.h5ad"
|
||||
|
||||
# There's lots of MALAT1, but it doesn't collide with any other names,
|
||||
# so it shouldn't change during merging.
|
||||
self.stable_gene = "MALAT1"
|
||||
|
||||
def test_fixup_gene_symbols_seurat(self):
|
||||
|
||||
if not os.path.isfile(self.seurat_path):
|
||||
return unittest.skip(
|
||||
"Skipping gene symbol conversion tests because test h5ads are not present. To create them, "
|
||||
"run server/test/fixtures/schema_test_data/generate_test_data.sh"
|
||||
)
|
||||
|
||||
original_adata = sc.read_h5ad(self.seurat_path)
|
||||
merged_adata = sc.read_h5ad(self.seurat_merged_path)
|
||||
|
||||
fixup_config = {"X": "log1p", "counts": "raw", "scale.data": "log1p"}
|
||||
|
||||
fixed_adata = remix.fixup_gene_symbols(original_adata, fixup_config)
|
||||
|
||||
self.assertEqual(
|
||||
merged_adata.layers["counts"][:, merged_adata.var.index == self.stable_gene].sum(),
|
||||
fixed_adata.raw.X[:, fixed_adata.var.index == self.stable_gene].sum()
|
||||
)
|
||||
self.assertAlmostEqual(
|
||||
merged_adata.X[:, merged_adata.var.index == self.stable_gene].sum(),
|
||||
fixed_adata.X[:, fixed_adata.var.index == self.stable_gene].sum()
|
||||
)
|
||||
|
||||
self.assertAlmostEqual(
|
||||
merged_adata.layers["scale.data"][:, merged_adata.var.index == self.stable_gene].sum(),
|
||||
fixed_adata.layers["scale.data"][:, fixed_adata.var.index == self.stable_gene].sum()
|
||||
)
|
||||
|
||||
def test_fixup_gene_symbols_sctransform(self):
|
||||
|
||||
if not os.path.isfile(self.sctransform_path):
|
||||
return unittest.skip(
|
||||
"Skipping gene symbol conversion tests because test h5ads are not present. To create them, "
|
||||
"run server/test/fixtures/schema_test_data/generate_test_data.sh"
|
||||
)
|
||||
|
||||
original_adata = sc.read_h5ad(self.sctransform_path)
|
||||
merged_adata = sc.read_h5ad(self.sctransform_merged_path)
|
||||
|
||||
fixup_config = {"X": "log1p", "counts": "raw"}
|
||||
|
||||
fixed_adata = remix.fixup_gene_symbols(original_adata, fixup_config)
|
||||
|
||||
# sctransform does a bunch of stuff, including slightly modifying the
|
||||
# raw counts. So we can't assert for exact equality the way we do with
|
||||
# the vanilla seurat tutorial. But, the results should still be very
|
||||
# close.
|
||||
merged_raw_stable = merged_adata.layers["counts"][:, merged_adata.var.index == self.stable_gene].sum()
|
||||
fixed_raw_stable = fixed_adata.raw.X[:, fixed_adata.var.index == self.stable_gene].sum()
|
||||
self.assertLess(abs(merged_raw_stable - fixed_raw_stable), .001 * merged_raw_stable)
|
||||
|
||||
self.assertAlmostEqual(
|
||||
merged_adata.X[:, merged_adata.var.index == self.stable_gene].sum(),
|
||||
fixed_adata.X[:, fixed_adata.var.index == self.stable_gene].sum(),
|
||||
0
|
||||
)
|
||||
@@ -0,0 +1,435 @@
|
||||
import json
|
||||
import os
|
||||
import unittest
|
||||
|
||||
import pandas as pd
|
||||
import scanpy as sc
|
||||
|
||||
from server.converters.schema import validate
|
||||
|
||||
PROJECT_ROOT = os.popen("git rev-parse --show-toplevel").read().strip()
|
||||
|
||||
|
||||
class TestFieldValidation(unittest.TestCase):
|
||||
|
||||
def test_validate_stringified_list_of_dicts(self):
|
||||
|
||||
good = json.dumps([{"a": 1}, {2: "x", "z": "y"}])
|
||||
not_stringified = [{"a": 1}, {2: "x", "z": "y"}]
|
||||
not_a_list = json.dumps({"bad": "dict"})
|
||||
not_json = "oh hey!"
|
||||
|
||||
self.assertTrue(validate._validate_stringified_list_of_dicts(good))
|
||||
|
||||
self.assertFalse(validate._validate_stringified_list_of_dicts(not_stringified))
|
||||
self.assertFalse(validate._validate_stringified_list_of_dicts(not_a_list))
|
||||
self.assertFalse(validate._validate_stringified_list_of_dicts(not_json))
|
||||
|
||||
def test_validate_human_readable_string(self):
|
||||
|
||||
good = "oh hey!"
|
||||
curie = "EFO:0001"
|
||||
ensg = "ENSG000001234"
|
||||
enst = "ENST000005678"
|
||||
|
||||
self.assertTrue(validate._validate_human_readable_string(good))
|
||||
|
||||
self.assertFalse(validate._validate_human_readable_string(curie))
|
||||
self.assertFalse(validate._validate_human_readable_string(ensg))
|
||||
self.assertFalse(validate._validate_human_readable_string(enst))
|
||||
|
||||
def test_validate_curie(self):
|
||||
|
||||
self.assertTrue(validate._validate_curie("UBERON:00001", ["UBERON", "EFO"]))
|
||||
self.assertTrue(validate._validate_curie("HsapDv:00002", ["HsapDv"]))
|
||||
|
||||
self.assertFalse(validate._validate_curie("HsapDv:00002", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("EFO:00002 (organoid)", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("EFO:00002 extra", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("UBERON:ABCD", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("Uberon:00002", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("UBERON:", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_curie("UBERON", ["UBERON", "EFO"]))
|
||||
|
||||
def test_validate_suffixed_curie(self):
|
||||
|
||||
self.assertTrue(validate._validate_suffixed_curie("EFO:00001", ["UBERON", "EFO"]))
|
||||
self.assertTrue(validate._validate_suffixed_curie("UBERON:00001 (cell culture)", ["UBERON", "EFO"]))
|
||||
|
||||
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002 (organoid)", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002(organoid)", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("HsapDv:00002", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("EFO:00002 extra", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("UBERON:ABCD", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("Uberon:00002", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("UBERON:", ["UBERON", "EFO"]))
|
||||
self.assertFalse(validate._validate_suffixed_curie("UBERON", ["UBERON", "EFO"]))
|
||||
|
||||
|
||||
class TestColumnValidation(unittest.TestCase):
|
||||
|
||||
def test_validate_unique(self):
|
||||
unique = pd.DataFrame([["abc", "def"], ["ghi", "jkl"], ["mnop", "qrs"]],
|
||||
index=["X", "Y", "Z"], columns=["col1", "col2"])
|
||||
duped = pd.DataFrame([["abc", "def"], ["ghi", "qrs"], ["abc", "qrs"]],
|
||||
index=["X", "Y", "X"], columns=["col1", "col2"])
|
||||
|
||||
schema_def = {"unique": True}
|
||||
|
||||
errors = validate._validate_column(unique.index, "index", "unique_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
errors = validate._validate_column(duped.index, "index", "duped_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("is not unique", errors[0])
|
||||
|
||||
errors = validate._validate_column(unique["col1"], "col1", "unique_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
errors = validate._validate_column(duped["col1"], "col1", "duped_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("is not unique", errors[0])
|
||||
|
||||
schema_def = {"unique": False}
|
||||
errors = validate._validate_column(duped["col1"], "col1", "duped_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
def test_validate_nullable(self):
|
||||
non_null = pd.DataFrame([["abc", "def"], ["ghi", "jkl"], ["mnop", "qrs"]],
|
||||
index=["X", "Y", "Z"], columns=["col1", "col2"])
|
||||
has_null = pd.DataFrame([["abc", "", None], ["ghi", "jkl", 1], ["mnop", "qrs", 2]],
|
||||
index=["X", "Y", "Z"], columns=["col1", "col2", "col3"])
|
||||
|
||||
schema_def = {"nullable": False}
|
||||
errors = validate._validate_column(non_null["col1"], "col1", "nonnull_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
errors = validate._validate_column(has_null["col1"], "col1", "hasnull_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
errors = validate._validate_column(has_null["col2"], "col2", "hasnull_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("contains empty values", errors[0])
|
||||
errors = validate._validate_column(has_null["col3"], "col3", "hasnull_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("contains empty values", errors[0])
|
||||
|
||||
schema_def = {"nullable": True}
|
||||
errors = validate._validate_column(has_null["col2"], "col2", "hasnull_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
def test_human_readable(self):
|
||||
hr_df = pd.DataFrame(
|
||||
[["for you, a human", "UBERON:12345", "UBERON:1234 (thundercat)"],
|
||||
["hope you're well", "bit of lungs", "brain"]],
|
||||
index=["ENSG00001", "ENSG00002"],
|
||||
columns=["good", "curie", "suffixed_curie"])
|
||||
|
||||
schema_def = {"type": "human-readable string"}
|
||||
errors = validate._validate_column(hr_df["good"], "good", "hr", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
errors = validate._validate_column(hr_df["curie"], "curie", "hr", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("non-human-readable", errors[0])
|
||||
|
||||
errors = validate._validate_column(hr_df["suffixed_curie"], "suffixed_curie", "hr", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("non-human-readable", errors[0])
|
||||
|
||||
errors = validate._validate_column(hr_df.index, "ensg", "hr", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("non-human-readable", errors[0])
|
||||
|
||||
def test_curie(self):
|
||||
|
||||
curie_df = pd.DataFrame(
|
||||
[["EFO:00001", "HsapDv:00001 (cell culture)", "EFO:", "MONDO:0001 cell culture"],
|
||||
["UBERON:00002", "HsapDv:00002 (organoid)", "EFO:12345", "MONDO:0002 (baba yaga)"],
|
||||
["EFO:0000000005", "HsapDv:000004 (humanzee)", "EFO:000002", "MONDO:0004 (TMNT)"]],
|
||||
index=["X", "Y", "Z"],
|
||||
columns=["good", "good_suffix", "bad", "bad_suffix"])
|
||||
|
||||
# Good
|
||||
schema_def = {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
errors = validate._validate_column(curie_df["good"], "good", "curie_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Good suffix
|
||||
schema_def = {"type": "suffixed curie", "prefixes": ["HsapDv", "WHATEVER"]}
|
||||
errors = validate._validate_column(curie_df["good_suffix"], "good_suffix", "curie_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Bad prefix
|
||||
schema_def = {"type": "curie", "prefixes": ["EFO"]}
|
||||
errors = validate._validate_column(curie_df["good"], "good", "curie_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("invalid ontology", errors[0])
|
||||
self.assertIn("must be curies from one of these", errors[0])
|
||||
|
||||
# Bad curies
|
||||
schema_def = {"type": "curie", "prefixes": ["EFO"]}
|
||||
errors = validate._validate_column(curie_df["bad"], "bad", "curie_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("invalid ontology", errors[0])
|
||||
|
||||
# Bad suffixes
|
||||
schema_def = {"type": "suffixed curie", "prefixes": ["EFO"]}
|
||||
errors = validate._validate_column(curie_df["bad_suffix"], "bad_suffix", "curie_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("invalid ontology", errors[0])
|
||||
|
||||
def test_enum(self):
|
||||
enum_df = pd.DataFrame(
|
||||
[["abc", "ghi"],
|
||||
["def", "jkl"]],
|
||||
index=["X", "Y"],
|
||||
columns=["col1", "col2"])
|
||||
|
||||
# All match
|
||||
schema_def = {"type": "string", "enum": ["abc", "def", "xyz"]}
|
||||
errors = validate._validate_column(enum_df["col1"], "col1", "enum_df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Missing value
|
||||
schema_def = {"type": "string", "enum": ["abc", "xyz"]}
|
||||
errors = validate._validate_column(enum_df["col1"], "col1", "enum_df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("unpermitted values", errors[0])
|
||||
|
||||
|
||||
class TestDictValidations(unittest.TestCase):
|
||||
|
||||
|
||||
def test_key_presence(self):
|
||||
|
||||
schema_def = {"keys": {"abc": None, "def": None}}
|
||||
|
||||
dict_ = {"abc": "123", "def": "456"}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Missing keys are bad
|
||||
dict_ = {"abc": "123"}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("missing key", errors[0])
|
||||
|
||||
# Extra keys are okay
|
||||
dict_ = {"abc": "123", "def": "456", "xyz": "789"}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
# Better not be empty come on
|
||||
dict_ = {}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 2)
|
||||
|
||||
def test_nullable(self):
|
||||
|
||||
schema_def = {"keys": {"abc": {"type": "string", "nullable": False},
|
||||
"def": {"type": "string", "nullable": True}}}
|
||||
|
||||
dict_ = {"abc": "xyz", "def": ""}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
dict_ = {"abc": "", "def": ""}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("empty value", errors[0])
|
||||
|
||||
def test_recurse(self):
|
||||
|
||||
schema_def = {
|
||||
"keys": {
|
||||
"subdict": {
|
||||
"type": "dict",
|
||||
"keys": {
|
||||
"subdict_key1": None,
|
||||
"subdict_key2": None
|
||||
}
|
||||
},
|
||||
"ontology": {
|
||||
"type": "curie",
|
||||
"prefixes": ["ONTOLOGY"]
|
||||
},
|
||||
"blob": {
|
||||
"type": "stringified list of dicts"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
|
||||
"ontology": "ONTOLOGY:123456",
|
||||
"blob": json.dumps([{"abc": 123}, {"def": 456}])
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any"},
|
||||
"ontology": "ONTOLOGY:123456",
|
||||
"blob": json.dumps([{"abc": 123}, {"def": 456}])
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("missing key", errors[0])
|
||||
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
|
||||
"ontology": "oh no not an ontology term",
|
||||
"blob": json.dumps([{"abc": 123}, {"def": 456}])
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("invalid ontology", errors[0])
|
||||
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any", "subdict_key2": "any"},
|
||||
"ontology": "ONTOLOGY:123456",
|
||||
"blob": [{"abc": 123}, {"def": 456}]
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("JSON-encoded list of dicts", errors[0])
|
||||
|
||||
# Multiple errors
|
||||
dict_ = {
|
||||
"subdict": {"subdict_key1": "any"},
|
||||
"ontology": "oh no not an ontology term",
|
||||
"blob": json.dumps([{"abc": 123}, {"def": 456}])
|
||||
}
|
||||
errors = validate._validate_dict(dict_, "d", schema_def)
|
||||
self.assertEqual(len(errors), 2)
|
||||
|
||||
|
||||
class TestDataframeValidation(unittest.TestCase):
|
||||
|
||||
def test_column_presence(self):
|
||||
df = pd.DataFrame(
|
||||
[["abc", "EFO:123"],
|
||||
["def", "UBERON:456"]],
|
||||
columns=["hr_string", "ontology"],
|
||||
index=["X", "Y"]
|
||||
)
|
||||
|
||||
schema_def = {
|
||||
"columns": {
|
||||
"hr_string": {"type": "human-readable string"},
|
||||
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
}
|
||||
}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
schema_def = {
|
||||
"columns": {
|
||||
"hr_string": {"type": "human-readable string"},
|
||||
"another_hr_string": {"type": "human-readable string"},
|
||||
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
}
|
||||
}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("missing column", errors[0])
|
||||
|
||||
# Extra is okay
|
||||
df = pd.DataFrame(
|
||||
[["abc", "EFO:123", "extra"],
|
||||
["def", "UBERON:456", "extra"]],
|
||||
columns=["hr_string", "ontology", "extra"],
|
||||
index=["X", "Y"]
|
||||
)
|
||||
schema_def = {
|
||||
"columns": {
|
||||
"hr_string": {"type": "human-readable string"},
|
||||
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
}
|
||||
}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
|
||||
def test_index(self):
|
||||
df = pd.DataFrame(
|
||||
[["abc", "123"],
|
||||
["def", "456"]],
|
||||
columns=["col1", "col2"],
|
||||
index=["ENSG0001", "ENSG0002"]
|
||||
)
|
||||
|
||||
schema_def = {"index": {"unique": True}}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertFalse(errors)
|
||||
|
||||
schema_def = {"index": {"type": "human-readable string"}}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("non-human-readable", errors[0])
|
||||
|
||||
df = pd.DataFrame(
|
||||
[["abc", "123"],
|
||||
["def", "456"]],
|
||||
columns=["col1", "col2"],
|
||||
index=["ENSG0001", "ENSG0001"]
|
||||
)
|
||||
schema_def = {"index": {"unique": True}}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertEqual(len(errors), 1)
|
||||
self.assertIn("is not unique", errors[0])
|
||||
|
||||
def test_recurse(self):
|
||||
|
||||
df = pd.DataFrame(
|
||||
[["abc", "HsapDv:0001"],
|
||||
["EFO:123", "UBERON:456"]],
|
||||
columns=["hr_string", "ontology"],
|
||||
index=["X", "Y"]
|
||||
)
|
||||
schema_def = {
|
||||
"columns": {
|
||||
"hr_string": {"type": "human-readable string"},
|
||||
"ontology": {"type": "curie", "prefixes": ["EFO", "UBERON"]}
|
||||
}
|
||||
}
|
||||
errors = validate._validate_dataframe(df, "df", schema_def)
|
||||
self.assertEqual(len(errors), 2)
|
||||
self.assertEqual(len([e for e in errors if "non-human-readable" in e]), 1)
|
||||
self.assertEqual(len([e for e in errors if "invalid ontology" in e]), 1)
|
||||
|
||||
|
||||
class TestGetSchema(unittest.TestCase):
|
||||
|
||||
def test_get_schema(self):
|
||||
self.assertIsInstance(validate.get_schema_definition("1.0.0"), dict)
|
||||
|
||||
with self.assertRaises(ValueError):
|
||||
validate.get_schema_definition("10.1.5")
|
||||
|
||||
|
||||
class TestValidate(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
self.source_h5ad_path = f"{PROJECT_ROOT}/server/test/fixtures/pbmc3k-CSC-gz.h5ad"
|
||||
|
||||
def test_shallow(self):
|
||||
|
||||
adata = sc.read_h5ad(self.source_h5ad_path)
|
||||
self.assertFalse(validate.validate_adata(adata, True))
|
||||
|
||||
adata.uns["version"] = {
|
||||
"corpora_schema_version": "1.0.0",
|
||||
"corpora_encoding_version": "0.1.0"
|
||||
}
|
||||
self.assertTrue(validate.validate_adata(adata, True))
|
||||
|
||||
def test_deep(self):
|
||||
adata = sc.read_h5ad(self.source_h5ad_path)
|
||||
self.assertFalse(validate.validate_adata(adata, False))
|
||||
|
||||
adata.uns["version"] = {
|
||||
"corpora_schema_version": "1.0.0",
|
||||
"corpora_encoding_version": "0.1.0"
|
||||
}
|
||||
self.assertFalse(validate.validate_adata(adata, False))
|
||||
Reference in New Issue
Block a user