do not hard-wire column names in annotations (#785)

* enforce column name uniqueness for obs and var

* parameterize the column name containing obs and var user-readable names

* use the new annotation index value from schema

* update f/e unit tests

* PR review suggestions

* lint
This commit is contained in:
Bruce Martin
2019-05-24 21:00:54 -07:00
committed by GitHub
parent a8c2e408d1
commit 3dc45d6330
16 changed files with 246 additions and 155 deletions
@@ -34,28 +34,34 @@ const aSchemaResponse = {
type: "float32" type: "float32"
}, },
annotations: { annotations: {
obs: [ obs: {
{ name: "name", type: "string" }, index: "name",
{ name: "field1", type: "int32" }, columns: [
{ name: "field2", type: "float32" }, { name: "name", type: "string" },
{ name: "field3", type: "boolean" }, { name: "field1", type: "int32" },
{ { name: "field2", type: "float32" },
name: "field4", { name: "field3", type: "boolean" },
type: "categorical", {
categories: field4Categories name: "field4",
} type: "categorical",
], categories: field4Categories
var: [ }
{ name: "name", type: "string" }, ]
{ name: "fieldA", type: "int32" }, },
{ name: "fieldB", type: "float32" }, var: {
{ name: "fieldC", type: "boolean" }, index: "name",
{ columns: [
name: "fieldD", { name: "name", type: "string" },
type: "categorical", { name: "fieldA", type: "int32" },
categories: fieldDCategories { name: "fieldB", type: "float32" },
} { name: "fieldC", type: "boolean" },
] {
name: "fieldD",
type: "categorical",
categories: fieldDCategories
}
]
}
}, },
layout: { layout: {
obs: [{ name: "umap", type: "float32", dims: ["umap_0", "umap_1"] }], obs: [{ name: "umap", type: "float32", dims: ["umap_0", "umap_1"] }],
@@ -53,7 +53,7 @@ describe("createUniverseFromResponse", () => {
expect(universe.obsAnnotations.dims).toEqual([ expect(universe.obsAnnotations.dims).toEqual([
nObs, nObs,
REST.schema.schema.annotations.obs.length REST.schema.schema.annotations.obs.columns.length
]); ]);
expect(universe.obsLayout.dims).toEqual([nObs, 2]); expect(universe.obsLayout.dims).toEqual([nObs, 2]);
expect(universe.obsLayout.colIndex.keys()).toEqual( expect(universe.obsLayout.colIndex.keys()).toEqual(
@@ -61,7 +61,7 @@ describe("createUniverseFromResponse", () => {
); );
expect(universe.varAnnotations.dims).toEqual([ expect(universe.varAnnotations.dims).toEqual([
nVar, nVar,
REST.schema.schema.annotations.var.length REST.schema.schema.annotations.var.columns.length
]); ]);
expect(universe.varData.isEmpty()).toBeTruthy(); expect(universe.varData.isEmpty()).toBeTruthy();
}); });
@@ -155,14 +155,18 @@ describe("createObsDimensionMap", () => {
const { crossfilter } = defaultBigBang(); const { crossfilter } = defaultBigBang();
const annotationNames = _.map( const annotationNames = _.map(
REST.schema.schema.annotations.obs, REST.schema.schema.annotations.obs.columns,
c => c.name c => c.name
); );
const schemaByObsName = _.keyBy(REST.schema.schema.annotations.obs, "name"); const obsIndexColName = REST.schema.schema.annotations.obs.index;
const schemaByObsName = _.keyBy(
REST.schema.schema.annotations.obs.columns,
"name"
);
expect(crossfilter).toBeDefined(); expect(crossfilter).toBeDefined();
annotationNames.forEach(name => { annotationNames.forEach(name => {
const dim = crossfilter.dimensions[obsAnnoDimensionName(name)]; const dim = crossfilter.dimensions[obsAnnoDimensionName(name)];
if (name === "name") { if (name === obsIndexColName) {
expect(dim).toBeUndefined(); expect(dim).toBeUndefined();
} else { } else {
const { type } = schemaByObsName[name]; const { type } = schemaByObsName[name];
+31 -20
View File
@@ -21,24 +21,32 @@ const doInitialDataLoad = () =>
dispatch({ type: "initial data load start" }); dispatch({ type: "initial data load start" });
try { try {
const requestJson = _(["config", "schema"]) /*
Step 1 - config & schema, all JSON
*/
const requestJson = ["config", "schema"]
.map(r => `${globals.API.prefix}${globals.API.version}${r}`) .map(r => `${globals.API.prefix}${globals.API.version}${r}`)
.map(url => doJsonRequest(url)) .map(url => doJsonRequest(url));
.value(); const stepOneResults = await Promise.all(requestJson);
const requestBinary = _([
"annotations/obs",
"annotations/var?annotation-name=name",
"layout/obs"
])
.map(r => `${globals.API.prefix}${globals.API.version}${r}`)
.map(url => doBinaryRequest(url))
.value();
const results = await Promise.all(_.concat(requestJson, requestBinary));
/* set config defaults */ /* set config defaults */
const config = { ...globals.configDefaults, ...results[0].config }; const config = { ...globals.configDefaults, ...stepOneResults[0].config };
const [, schema, obsAnno, varAnno, obsLayout] = [...results]; const schema = stepOneResults[1];
/*
Step 2 - dataframes, all binary. NOTE: uses results of step 1.
*/
/* only load names for var annotations, if possible*/
const varIndexName = schema?.schema?.annotations?.var?.index;
const varAnnotationsQuery = varIndexName
? `?annotation-name=${varIndexName}`
: "";
const varAnnotationsURL = `annotations/var${varAnnotationsQuery}`;
const requestBinary = ["annotations/obs", varAnnotationsURL, "layout/obs"]
.map(r => `${globals.API.prefix}${globals.API.version}${r}`)
.map(url => doBinaryRequest(url));
const stepTwoResults = await Promise.all(requestBinary);
const [obsAnno, varAnno, obsLayout] = [...stepTwoResults];
const universe = Universe.createUniverseFromResponse( const universe = Universe.createUniverseFromResponse(
config, config,
schema, schema,
@@ -91,6 +99,10 @@ needs expression data.
Transparently utilizes cached data if it is already present. Transparently utilizes cached data if it is already present.
*/ */
async function _doRequestExpressionData(dispatch, getState, genes) { async function _doRequestExpressionData(dispatch, getState, genes) {
const state = getState();
const { universe } = state;
const varIndexName = universe.schema.annotations.var.index;
/* helper for this function only */ /* helper for this function only */
const fetchData = async geneNames => { const fetchData = async geneNames => {
const res = await fetch( const res = await fetch(
@@ -100,7 +112,7 @@ async function _doRequestExpressionData(dispatch, getState, genes) {
body: JSON.stringify({ body: JSON.stringify({
filter: { filter: {
var: { var: {
annotation_value: [{ name: "name", values: geneNames }] annotation_value: [{ name: varIndexName, values: geneNames }]
} }
} }
}), }),
@@ -123,8 +135,6 @@ async function _doRequestExpressionData(dispatch, getState, genes) {
return Universe.convertDataFBStoObject(universe, data); return Universe.convertDataFBStoObject(universe, data);
}; };
const state = getState();
const { universe } = state;
/* preload data already in cache */ /* preload data already in cache */
let expressionData = _.transform( let expressionData = _.transform(
genes, genes,
@@ -241,6 +251,7 @@ const requestDifferentialExpression = (set1, set2, num_genes = 10) => async (
*/ */
const state = getState(); const state = getState();
const { universe } = state; const { universe } = state;
const varIndexName = universe.schema.annotations.var.index;
// Legal values are null, Array or TypedArray. Null is initial state. // Legal values are null, Array or TypedArray. Null is initial state.
if (!set1) set1 = []; if (!set1) set1 = [];
@@ -277,7 +288,7 @@ const requestDifferentialExpression = (set1, set2, num_genes = 10) => async (
const data = await res.json(); const data = await res.json();
// result is [ [varIdx, ...], ... ] // result is [ [varIdx, ...], ... ]
const topNGenes = _.map(data, r => const topNGenes = _.map(data, r =>
universe.varAnnotations.at(r[0], "name") universe.varAnnotations.at(r[0], varIndexName)
); );
/* /*
@@ -66,7 +66,8 @@ class Continuous extends React.Component {
? _.map(obsAnnotations.colIndex.keys(), key => { ? _.map(obsAnnotations.colIndex.keys(), key => {
const isColorField = const isColorField =
key.includes("color") || key.includes("Color"); key.includes("color") || key.includes("Color");
if (key === "name" || isColorField) return null; if (key === schema.annotations.obs.index || isColorField)
return null;
const summary = obsAnnotations.col(key).summarize(); const summary = obsAnnotations.col(key).summarize();
const nonFiniteExtent = const nonFiniteExtent =
+13 -6
View File
@@ -85,7 +85,8 @@ class GeneExpression extends React.Component {
*/ */
const { world } = this.props; const { world } = this.props;
const { varAnnotations } = world; const { varAnnotations } = world;
const geneNames = varAnnotations.col("name").asArray(); const varIndexName = world.schema.annotations.var.index;
const geneNames = varAnnotations.col(varIndexName).asArray();
if (geneNames.length > 0) { if (geneNames.length > 0) {
const placeholder = []; const placeholder = [];
let len = geneNames.length; let len = geneNames.length;
@@ -107,6 +108,7 @@ class GeneExpression extends React.Component {
handleClick(g) { handleClick(g) {
const { world, dispatch, userDefinedGenes } = this.props; const { world, dispatch, userDefinedGenes } = this.props;
const varIndexName = world.schema.annotations.var.index;
const gene = g.target; const gene = g.target;
if (userDefinedGenes.indexOf(gene) !== -1) { if (userDefinedGenes.indexOf(gene) !== -1) {
postUserErrorToast("That gene already exists"); postUserErrorToast("That gene already exists");
@@ -114,7 +116,9 @@ class GeneExpression extends React.Component {
postUserErrorToast( postUserErrorToast(
"That's too many genes, you can have at most 15 user defined genes" "That's too many genes, you can have at most 15 user defined genes"
); );
} else if (world.varAnnotations.col("name").indexOf(gene) === undefined) { } else if (
world.varAnnotations.col(varIndexName).indexOf(gene) === undefined
) {
postUserErrorToast("That doesn't appear to be a valid gene name."); postUserErrorToast("That doesn't appear to be a valid gene name.");
} else { } else {
dispatch({ type: "single user defined gene start" }); dispatch({ type: "single user defined gene start" });
@@ -127,6 +131,7 @@ class GeneExpression extends React.Component {
handleBulkAddClick() { handleBulkAddClick() {
const { world, dispatch, userDefinedGenes } = this.props; const { world, dispatch, userDefinedGenes } = this.props;
const varIndexName = world.schema.annotations.var.index;
const { bulkAdd } = this.state; const { bulkAdd } = this.state;
/* /*
@@ -145,7 +150,9 @@ class GeneExpression extends React.Component {
if (userDefinedGenes.indexOf(gene) !== -1) { if (userDefinedGenes.indexOf(gene) !== -1) {
return keepAroundErrorToast("That gene already exists"); return keepAroundErrorToast("That gene already exists");
} }
if (world.varAnnotations.col("name").indexOf(gene) === undefined) { if (
world.varAnnotations.col(varIndexName).indexOf(gene) === undefined
) {
return keepAroundErrorToast( return keepAroundErrorToast(
`${gene} doesn't appear to be a valid gene name.` `${gene} doesn't appear to be a valid gene name.`
); );
@@ -168,7 +175,7 @@ class GeneExpression extends React.Component {
userDefinedGenesLoading, userDefinedGenesLoading,
differential differential
} = this.props; } = this.props;
const varIndexName = world?.schema?.annotations?.var?.index;
const { tab, bulkAdd } = this.state; const { tab, bulkAdd } = this.state;
return ( return (
@@ -243,7 +250,7 @@ class GeneExpression extends React.Component {
itemRenderer={renderGene.bind(this)} itemRenderer={renderGene.bind(this)}
items={ items={
world && world.varAnnotations world && world.varAnnotations
? world.varAnnotations.col("name").asArray() ? world.varAnnotations.col(varIndexName).asArray()
: ["No genes"] : ["No genes"]
} }
popoverProps={{ minimal: true }} popoverProps={{ minimal: true }}
@@ -322,7 +329,7 @@ class GeneExpression extends React.Component {
<ExpressionButtons /> <ExpressionButtons />
{differential.diffExp {differential.diffExp
? _.map(differential.diffExp, (value, index) => { ? _.map(differential.diffExp, (value, index) => {
const name = world.varAnnotations.at(value[0], "name"); const name = world.varAnnotations.at(value[0], varIndexName);
const values = world.varData.col(name); const values = world.varData.col(name);
if (!values) { if (!values) {
return null; return null;
+2 -1
View File
@@ -93,9 +93,10 @@ const Controls = (
} }
case "request differential expression success": { case "request differential expression success": {
const { world } = prevSharedState; const { world } = prevSharedState;
const varIndexName = world.schema.annotations.var.index;
const _diffexpGenes = []; const _diffexpGenes = [];
action.data.forEach(d => { action.data.forEach(d => {
_diffexpGenes.push(world.varAnnotations.at(d[0], "name")); _diffexpGenes.push(world.varAnnotations.at(d[0], varIndexName));
}); });
return { return {
...state, ...state,
+4 -2
View File
@@ -90,8 +90,9 @@ const CrossfilterReducer = (
case "request differential expression success": { case "request differential expression success": {
const { world } = prevSharedState; const { world } = prevSharedState;
const varIndexName = world.schema.annotations.var.index;
const genes = _.map(action.data, d => const genes = _.map(action.data, d =>
world.varAnnotations.at(d[0], "name") world.varAnnotations.at(d[0], varIndexName)
); );
const crossfilter = _.reduce( const crossfilter = _.reduce(
genes, genes,
@@ -109,10 +110,11 @@ const CrossfilterReducer = (
case "clear differential expression": { case "clear differential expression": {
const { world } = prevSharedState; const { world } = prevSharedState;
const varIndexName = world.schema.annotations.var.index;
const crossfilter = _.reduce( const crossfilter = _.reduce(
action.diffExp, action.diffExp,
(xfltr, values) => { (xfltr, values) => {
const name = world.varAnnotations.at(values[0], "name"); const name = world.varAnnotations.at(values[0], varIndexName);
return xfltr.delDimension(diffexpDimensionName(name)); return xfltr.delDimension(diffexpDimensionName(name));
}, },
state state
@@ -56,13 +56,14 @@ function topNCategories(summary) {
export function createCategoricalSelection(maxCategoryItems, world) { export function createCategoricalSelection(maxCategoryItems, world) {
const res = {}; const res = {};
const obsIndexName = world.schema.annotations.obs.index;
_.forEach(world.obsAnnotations.colIndex.keys(), key => { _.forEach(world.obsAnnotations.colIndex.keys(), key => {
const summary = world.obsAnnotations.col(key).summarize(); const summary = world.obsAnnotations.col(key).summarize();
if (summary.categories) { if (summary.categories) {
const isColorField = key.includes("color") || key.includes("Color"); const isColorField = key.includes("color") || key.includes("Color");
const isSelectableCategory = const isSelectableCategory =
!isColorField && !isColorField &&
key !== "name" && key !== obsIndexName &&
summary.categories.length < maxCategoryItems; summary.categories.length < maxCategoryItems;
if (isSelectableCategory) { if (isSelectableCategory) {
const [categoryValues, categoryValueCounts] = topNCategories(summary); const [categoryValues, categoryValueCounts] = topNCategories(summary);
+5 -4
View File
@@ -124,7 +124,7 @@ function reconcileSchemaCategoriesWithSummary(universe) {
cases, add a 'categories' field to the schema so it is accessible. cases, add a 'categories' field to the schema so it is accessible.
*/ */
universe.schema.annotations.obs.forEach(s => { universe.schema.annotations.obs.columns.forEach(s => {
if ( if (
s.type === "string" || s.type === "string" ||
s.type === "boolean" || s.type === "boolean" ||
@@ -179,10 +179,10 @@ export function createUniverseFromResponse(
/* Index schema for ease of use */ /* Index schema for ease of use */
universe.schema.annotations.obsByName = fromEntries( universe.schema.annotations.obsByName = fromEntries(
universe.schema.annotations.obs.map(v => [v.name, v]) universe.schema.annotations.obs.columns.map(v => [v.name, v])
); );
universe.schema.annotations.varByName = fromEntries( universe.schema.annotations.varByName = fromEntries(
universe.schema.annotations.var.map(v => [v.name, v]) universe.schema.annotations.var.columns.map(v => [v.name, v])
); );
universe.schema.layout.obsByName = fromEntries( universe.schema.layout.obsByName = fromEntries(
universe.schema.layout.obs.map(v => [v.name, v]) universe.schema.layout.obs.map(v => [v.name, v])
@@ -213,8 +213,9 @@ export function convertDataFBStoObject(universe, arrayBuffer) {
throw new Error("Unexpected non-floating point response from server."); throw new Error("Unexpected non-floating point response from server.");
} }
const varIndexName = universe.schema.annotations.var.index;
for (let c = 0; c < colIdx.length; c += 1) { for (let c = 0; c < colIdx.length; c += 1) {
const varName = universe.varAnnotations.at(colIdx[c], "name"); const varName = universe.varAnnotations.at(colIdx[c], varIndexName);
result[varName] = columns[c]; result[varName] = columns[c];
} }
return result; return result;
+6 -2
View File
@@ -263,10 +263,14 @@ function deduceDimensionType(attributes, fieldName) {
export function createObsDimensions(crossfilter, world, XYdimNames) { export function createObsDimensions(crossfilter, world, XYdimNames) {
/* /*
create and return a crossfilter with a dimension for every obs annotation create and return a crossfilter with a dimension for every obs annotation
for which we have a supported type, *except* 'name' for which we have a supported type, *except* for the index column, indicated
by schema.annotations.obs.index.
*/ */
const { schema, obsLayout, obsAnnotations } = world; const { schema, obsLayout, obsAnnotations } = world;
const annoList = schema.annotations.obs.filter(anno => anno.name !== "name"); const indexName = schema.annotations.obs.index;
const annoList = schema.annotations.obs.columns.filter(
anno => anno.name !== indexName
);
crossfilter = annoList.reduce((xfltr, anno) => { crossfilter = annoList.reduce((xfltr, anno) => {
const dimType = deduceDimensionType(anno, anno.name); const dimType = deduceDimensionType(anno, anno.name);
const colData = obsAnnotations.col(anno.name).asArray(); const colData = obsAnnotations.col(anno.name).asArray();
+65 -33
View File
@@ -50,41 +50,61 @@ class ScanpyEngine(CXGDriver):
"diffexp_lfc_cutoff": 0.01, "diffexp_lfc_cutoff": 0.01,
} }
def _alias_annotation_names(self, axis, name): @staticmethod
""" def _create_unique_column_name(df, col_name_prefix):
Do all user-specified annotation aliasing. """ given the columns of a dataframe, and a name prefix, return a column name which
does not exist in the dataframe, AND which is prefixed by `prefix`
As a *critical* side-effect, ensure the indices are simple number ranges The approach is to append a numeric suffix, starting at zero and increasing by
(accomplished by calling pandas.DataFrame.reset_index()) one, until an unused name is found (eg, prefix_0, prefix_1, ...).
""" """
if name == "name": suffix = 0
# a noop, so skip it while f"{col_name_prefix}{suffix}" in df:
return suffix += 1
return f"{col_name_prefix}{suffix}"
ax_name = str(axis) def _alias_annotation_names(self):
df_axis = getattr(self.data, ax_name) """
if name is None: The front-end relies on the existance of a unique, human-readable
# reset index to simple range; alias "name" to point at the index for obs & var (eg, var is typically gene name, obs the cell name).
# previously specified index. The user can specify these via the --obs-names and --var-names config.
df_axis.reset_index(inplace=True) If they are not specified, use the existing index to create them, giving
df_axis.rename(inplace=True, columns={"index": "name"}) the resulting column a unique name (eg, "name").
elif name in df_axis.columns:
if name not in df_axis.columns: In both cases, enforce that the result is unique, and communicate the
index column name to the front-end via the obs_names and var_names config
(which is incorporated into the schema).
"""
for (ax_name, config_name) in ((Axis.OBS, "obs_names"), (Axis.VAR, "var_names")):
name = self.config[config_name]
df_axis = getattr(self.data, str(ax_name))
if name is None:
# Default: create unique names from index
if not df_axis.index.is_unique:
raise KeyError(
f"Values in {ax_name}.index must be unique. "
"Please prepare data to contain unique index values, or specify an "
"alternative with --{ax_name}-name."
)
name = self._create_unique_column_name(df_axis.columns, "name_")
self.config[config_name] = name
# reset index to simple range; alias name to point at the
# previously specified index.
df_axis.rename_axis(name, inplace=True)
df_axis.reset_index(inplace=True)
elif name in df_axis.columns:
# User has specified alternative column for unique names, and it exists
if not df_axis[name].is_unique:
raise KeyError(
f"Values in {ax_name}.{name} must be unique. "
"Please prepare data to contain unique values."
)
df_axis.reset_index(drop=True, inplace=True)
else:
# user specified a non-existent column name
raise KeyError( raise KeyError(
f"Annotation name {name}, specified in --{ax_name}-name does not exist." f"Annotation name {name}, specified in --{ax_name}-name does not exist."
) )
if not df_axis[name].is_unique:
raise KeyError(
f"Values in -{ax_name}-name must be unique. "
"Please prepare data to contain unique values."
)
# reset index to simple range; alias user-specified annotation to "name"
df_axis.reset_index(drop=True, inplace=True)
df_axis.rename(inplace=True, columns={name: "name"})
else:
raise KeyError(
f"Annotation name {name}, specified in --{ax_name}_name does not exist."
)
@staticmethod @staticmethod
def _can_cast_to_float32(ann): def _can_cast_to_float32(ann):
@@ -114,7 +134,16 @@ class ScanpyEngine(CXGDriver):
"nVar": self.gene_count, "nVar": self.gene_count,
"type": str(self.data.X.dtype), "type": str(self.data.X.dtype),
}, },
"annotations": {"obs": [], "var": []}, "annotations": {
"obs": {
"index": self.config["obs_names"],
"columns": []
},
"var": {
"index": self.config["var_names"],
"columns": []
}
},
"layout": {"obs": []} "layout": {"obs": []}
} }
for ax in Axis: for ax in Axis:
@@ -139,7 +168,7 @@ class ScanpyEngine(CXGDriver):
raise TypeError( raise TypeError(
f"Annotations of type {curr_axis[ann].dtype} are unsupported by cellxgene." f"Annotations of type {curr_axis[ann].dtype} are unsupported by cellxgene."
) )
self.schema["annotations"][ax].append(ann_schema) self.schema["annotations"][ax]["columns"].append(ann_schema)
for layout in self.config['layout']: for layout in self.config['layout']:
layout_schema = { layout_schema = {
@@ -173,8 +202,11 @@ class ScanpyEngine(CXGDriver):
@requires_data @requires_data
def _validate_and_initialize(self): def _validate_and_initialize(self):
self._alias_annotation_names(Axis.OBS, self.config["obs_names"]) # var and obs column names must be unique
self._alias_annotation_names(Axis.VAR, self.config["var_names"]) if not self.data.obs.columns.is_unique or not self.data.var.columns.is_unique:
raise KeyError(f"All annotation column names must be unique.")
self._alias_annotation_names()
self._validate_data_types() self._validate_data_types()
self.cell_count = self.data.shape[0] self.cell_count = self.data.shape[0]
self.gene_count = self.data.shape[1] self.gene_count = self.data.shape[1]
+48 -42
View File
@@ -5,48 +5,54 @@
"type": "float32" "type": "float32"
}, },
"annotations": { "annotations": {
"obs": [ "obs": {
{ "index": "name_0",
"name": "name", "columns": [
"type": "string" {
}, "name": "name_0",
{ "type": "string"
"name": "n_genes", },
"type": "int32" {
}, "name": "n_genes",
{ "type": "int32"
"name": "percent_mito", },
"type": "float32" {
}, "name": "percent_mito",
{ "type": "float32"
"name": "n_counts", },
"type": "float32" {
}, "name": "n_counts",
{ "type": "float32"
"name": "louvain", },
"type": "categorical", {
"categories": [ "name": "louvain",
"CD4 T cells", "type": "categorical",
"CD14+ Monocytes", "categories": [
"B cells", "CD4 T cells",
"CD8 T cells", "CD14+ Monocytes",
"NK cells", "B cells",
"FCGR3A+ Monocytes", "CD8 T cells",
"Dendritic cells", "NK cells",
"Megakaryocytes" "FCGR3A+ Monocytes",
] "Dendritic cells",
} "Megakaryocytes"
], ]
"var": [ }
{ ]
"name": "name", },
"type": "string" "var": {
}, "index": "name_0",
{ "columns": [
"name": "n_cells", {
"type": "int32" "name": "name_0",
} "type": "string"
] },
{
"name": "n_cells",
"type": "int32"
}
]
}
}, },
"layout": { "layout": {
"obs": [ "obs": [
+10 -5
View File
@@ -23,7 +23,8 @@ class EndPoints(unittest.TestCase):
session = requests.Session() session = requests.Session()
for i in range(90): for i in range(90):
try: try:
session.get(f"{URL_BASE}schema") result = session.get(f"{URL_BASE}schema")
cls.schema = result.json()
except requests.exceptions.ConnectionError: except requests.exceptions.ConnectionError:
time.sleep(1) time.sleep(1)
@@ -45,7 +46,8 @@ class EndPoints(unittest.TestCase):
self.assertEqual(result.headers["Content-Type"], "application/json") self.assertEqual(result.headers["Content-Type"], "application/json")
result_data = result.json() result_data = result.json()
self.assertEqual(result_data["schema"]["dataframe"]["nObs"], 2638) self.assertEqual(result_data["schema"]["dataframe"]["nObs"], 2638)
self.assertEqual(len(result_data["schema"]["annotations"]["obs"]), 5) self.assertEqual(len(result_data["schema"]["annotations"]["obs"]), 2)
self.assertEqual(len(result_data["schema"]["annotations"]["obs"]["columns"]), 5)
def test_config(self): def test_config(self):
endpoint = "config" endpoint = "config"
@@ -95,7 +97,8 @@ class EndPoints(unittest.TestCase):
self.assertIsNotNone(df['col_idx']) self.assertIsNotNone(df['col_idx'])
self.assertIsNone(df['row_idx']) self.assertIsNone(df['row_idx'])
self.assertEqual(len(df['columns']), df['n_cols']) self.assertEqual(len(df['columns']), df['n_cols'])
self.assertListEqual(df['col_idx'], ['name', 'n_genes', 'percent_mito', 'n_counts', 'louvain']) obs_index_col_name = self.schema["schema"]["annotations"]["obs"]["index"]
self.assertListEqual(df['col_idx'], [obs_index_col_name, 'n_genes', 'percent_mito', 'n_counts', 'louvain'])
def test_get_annotations_obs_keys_fbs(self): def test_get_annotations_obs_keys_fbs(self):
endpoint = "annotations/obs" endpoint = "annotations/obs"
@@ -165,7 +168,8 @@ class EndPoints(unittest.TestCase):
self.assertIsNotNone(df['col_idx']) self.assertIsNotNone(df['col_idx'])
self.assertIsNone(df['row_idx']) self.assertIsNone(df['row_idx'])
self.assertEqual(len(df['columns']), df['n_cols']) self.assertEqual(len(df['columns']), df['n_cols'])
self.assertListEqual(df['col_idx'], ['name', 'n_cells']) var_index_col_name = self.schema["schema"]["annotations"]["var"]["index"]
self.assertListEqual(df['col_idx'], [var_index_col_name, 'n_cells'])
def test_get_annotations_var_keys_fbs(self): def test_get_annotations_var_keys_fbs(self):
endpoint = "annotations/var" endpoint = "annotations/var"
@@ -247,7 +251,8 @@ class EndPoints(unittest.TestCase):
endpoint = f"data/var" endpoint = f"data/var"
url = f"{URL_BASE}{endpoint}" url = f"{URL_BASE}{endpoint}"
header = {"Accept": "application/octet-stream"} header = {"Accept": "application/octet-stream"}
var_filter = {"filter": {"var": {"annotation_value": [{"name": "name", "values": ["RER1"]}]}}} index_col_name = self.schema["schema"]["annotations"]["var"]["index"]
var_filter = {"filter": {"var": {"annotation_value": [{"name": index_col_name, "values": ["RER1"]}]}}}
result = self.session.put(url, headers=header, json=var_filter) result = self.session.put(url, headers=header, json=var_filter)
self.assertEqual(result.status_code, HTTPStatus.OK) self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.headers["Content-Type"], "application/octet-stream") self.assertEqual(result.headers["Content-Type"], "application/octet-stream")
+4 -2
View File
@@ -58,14 +58,16 @@ class NaNTest(unittest.TestCase):
def test_annotation(self): def test_annotation(self):
annotations = decode_fbs.decode_matrix_FBS(self.data.annotation_to_fbs_matrix("obs")) annotations = decode_fbs.decode_matrix_FBS(self.data.annotation_to_fbs_matrix("obs"))
obs_index_col_name = self.data.schema["annotations"]["obs"]["index"]
self.assertEqual( self.assertEqual(
annotations["col_idx"], annotations["col_idx"],
["name", "n_genes", "percent_mito", "n_counts", "louvain"] [obs_index_col_name, "n_genes", "percent_mito", "n_counts", "louvain"]
) )
self.assertEqual(annotations["n_rows"], 100) self.assertEqual(annotations["n_rows"], 100)
self.assertTrue(math.isnan(annotations["columns"][2][0])) self.assertTrue(math.isnan(annotations["columns"][2][0]))
annotations = decode_fbs.decode_matrix_FBS(self.data.annotation_to_fbs_matrix("var")) annotations = decode_fbs.decode_matrix_FBS(self.data.annotation_to_fbs_matrix("var"))
self.assertEqual(annotations["col_idx"], ["name", "n_cells", "var_with_nans"]) var_index_col_name = self.data.schema["annotations"]["var"]["index"]
self.assertEqual(annotations["col_idx"], [var_index_col_name, "n_cells", "var_with_nans"])
self.assertEqual(annotations["n_rows"], 100) self.assertEqual(annotations["n_rows"], 100)
self.assertTrue(math.isnan(annotations["columns"][2][0])) self.assertTrue(math.isnan(annotations["columns"][2][0]))
+17 -9
View File
@@ -31,9 +31,11 @@ class EngineTest(unittest.TestCase):
self.assertTrue(self.data.data.X[0, 0] - -0.171_469_51 < epsilon) self.assertTrue(self.data.data.X[0, 0] - -0.171_469_51 < epsilon)
def test_mandatory_annotations(self): def test_mandatory_annotations(self):
self.assertIn("name", self.data.data.obs) obs_index_col_name = self.data.schema["annotations"]["obs"]["index"]
self.assertIn(obs_index_col_name, self.data.data.obs)
self.assertEqual(list(self.data.data.obs.index), list(range(2638))) self.assertEqual(list(self.data.data.obs.index), list(range(2638)))
self.assertIn("name", self.data.data.var) var_index_col_name = self.data.schema["annotations"]["var"]["index"]
self.assertIn(var_index_col_name, self.data.data.var)
self.assertEqual(list(self.data.data.var.index), list(range(1838))) self.assertEqual(list(self.data.data.var.index), list(range(1838)))
@pytest.mark.filterwarnings("ignore:Scanpy data matrix") @pytest.mark.filterwarnings("ignore:Scanpy data matrix")
@@ -70,12 +72,14 @@ class EngineTest(unittest.TestCase):
self.assertEqual(data["n_cols"], 91) self.assertEqual(data["n_cols"], 91)
def test_obs_and_var_names(self): def test_obs_and_var_names(self):
self.assertEqual(np.sum(self.data.data.var["name"].isna()), 0) self.assertEqual(np.sum(self.data.data.var[self.data.schema["annotations"]["var"]["index"]].isna()), 0)
self.assertEqual(np.sum(self.data.data.obs["name"].isna()), 0) self.assertEqual(np.sum(self.data.data.obs[self.data.schema["annotations"]["obs"]["index"]].isna()), 0)
def test_schema(self): def test_schema(self):
with open(path.join(path.dirname(__file__), "schema.json")) as fh: with open(path.join(path.dirname(__file__), "schema.json")) as fh:
schema = json.load(fh) schema = json.load(fh)
print(schema)
print(self.data.schema)
self.assertEqual(self.data.schema, schema) self.assertEqual(self.data.schema, schema)
def test_schema_produces_error(self): def test_schema_produces_error(self):
@@ -108,16 +112,18 @@ class EngineTest(unittest.TestCase):
annotations = decode_fbs.decode_matrix_FBS(fbs) annotations = decode_fbs.decode_matrix_FBS(fbs)
self.assertEqual(annotations["n_rows"], 2638) self.assertEqual(annotations["n_rows"], 2638)
self.assertEqual(annotations["n_cols"], 5) self.assertEqual(annotations["n_cols"], 5)
obs_index_col_name = self.data.schema["annotations"]["obs"]["index"]
self.assertEqual( self.assertEqual(
annotations["col_idx"], annotations["col_idx"],
["name", "n_genes", "percent_mito", "n_counts", "louvain"], [obs_index_col_name, "n_genes", "percent_mito", "n_counts", "louvain"],
) )
fbs = self.data.annotation_to_fbs_matrix("var") fbs = self.data.annotation_to_fbs_matrix("var")
annotations = decode_fbs.decode_matrix_FBS(fbs) annotations = decode_fbs.decode_matrix_FBS(fbs)
self.assertEqual(annotations['n_rows'], 1838) self.assertEqual(annotations['n_rows'], 1838)
self.assertEqual(annotations['n_cols'], 2) self.assertEqual(annotations['n_cols'], 2)
self.assertEqual(annotations["col_idx"], ["name", "n_cells"]) var_index_col_name = self.data.schema["annotations"]["var"]["index"]
self.assertEqual(annotations["col_idx"], [var_index_col_name, "n_cells"])
def test_annotation_fields(self): def test_annotation_fields(self):
fbs = self.data.annotation_to_fbs_matrix("obs", ["n_genes", "n_counts"]) fbs = self.data.annotation_to_fbs_matrix("obs", ["n_genes", "n_counts"])
@@ -125,7 +131,8 @@ class EngineTest(unittest.TestCase):
self.assertEqual(annotations["n_rows"], 2638) self.assertEqual(annotations["n_rows"], 2638)
self.assertEqual(annotations['n_cols'], 2) self.assertEqual(annotations['n_cols'], 2)
fbs = self.data.annotation_to_fbs_matrix("var", ["name"]) var_index_col_name = self.data.schema["annotations"]["var"]["index"]
fbs = self.data.annotation_to_fbs_matrix("var", [var_index_col_name])
annotations = decode_fbs.decode_matrix_FBS(fbs) annotations = decode_fbs.decode_matrix_FBS(fbs)
self.assertEqual(annotations['n_rows'], 1838) self.assertEqual(annotations['n_rows'], 1838)
self.assertEqual(annotations['n_cols'], 1) self.assertEqual(annotations['n_cols'], 1)
@@ -163,9 +170,10 @@ class EngineTest(unittest.TestCase):
self.data.data_frame_to_fbs_matrix(filter_["filter"], "var") self.data.data_frame_to_fbs_matrix(filter_["filter"], "var")
def test_data_named_gene(self): def test_data_named_gene(self):
var_index_col_name = self.data.schema["annotations"]["var"]["index"]
filter_ = { filter_ = {
"filter": { "filter": {
"var": {"annotation_value": [{"name": "name", "values": ["RER1"]}]} "var": {"annotation_value": [{"name": var_index_col_name, "values": ["RER1"]}]}
} }
} }
fbs = self.data.data_frame_to_fbs_matrix(filter_["filter"], "var") fbs = self.data.data_frame_to_fbs_matrix(filter_["filter"], "var")
@@ -176,7 +184,7 @@ class EngineTest(unittest.TestCase):
filter_ = { filter_ = {
"filter": { "filter": {
"var": {"annotation_value": [{"name": "name", "values": ["SPEN", "TYMP", "PRMT2"]}]} "var": {"annotation_value": [{"name": var_index_col_name, "values": ["SPEN", "TYMP", "PRMT2"]}]}
} }
} }
fbs = self.data.data_frame_to_fbs_matrix(filter_["filter"], "var") fbs = self.data.data_frame_to_fbs_matrix(filter_["filter"], "var")