mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-09 07:18:11 +08:00
categorical type handling fix (#1342)
* fix numeric category conversion bug * correctly compute categorical summaries * lint * remove debugging print
This commit is contained in:
@@ -7,7 +7,10 @@ import {
|
|||||||
callOnceLazy,
|
callOnceLazy,
|
||||||
memoize
|
memoize
|
||||||
} from "./util";
|
} from "./util";
|
||||||
import { summarizeContinuous, summarizeCategorical } from "./summarize";
|
import {
|
||||||
|
summarizeContinuous,
|
||||||
|
summarizeCategorical as _summarizeCategorical
|
||||||
|
} from "./summarize";
|
||||||
import {
|
import {
|
||||||
histogramCategorical,
|
histogramCategorical,
|
||||||
hashCategorical,
|
hashCategorical,
|
||||||
@@ -243,8 +246,11 @@ class Dataframe {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/*
|
/*
|
||||||
Summarize the column data. Lazy eval;
|
Summarize the column data. Lazy eval, memoized
|
||||||
*/
|
*/
|
||||||
|
const summarizeCategorical = callOnceLazy(() =>
|
||||||
|
_summarizeCategorical(column)
|
||||||
|
);
|
||||||
const summarize = callOnceLazy(() =>
|
const summarize = callOnceLazy(() =>
|
||||||
isTypedArray(column)
|
isTypedArray(column)
|
||||||
? summarizeContinuous(column)
|
? summarizeContinuous(column)
|
||||||
@@ -263,6 +269,7 @@ class Dataframe {
|
|||||||
}
|
}
|
||||||
|
|
||||||
get.summarize = summarize;
|
get.summarize = summarize;
|
||||||
|
get.summarizeCategorical = summarizeCategorical;
|
||||||
get.asArray = asArray;
|
get.asArray = asArray;
|
||||||
get.has = has;
|
get.has = has;
|
||||||
get.ihas = ihas;
|
get.ihas = ihas;
|
||||||
|
|||||||
@@ -103,7 +103,7 @@ export function createCategoricalSelection(world, names) {
|
|||||||
in sorted order, and must include all category values even if they are not
|
in sorted order, and must include all category values even if they are not
|
||||||
actively used in the current world.
|
actively used in the current world.
|
||||||
*/
|
*/
|
||||||
const summary = obsAnnotations.col(name).summarize();
|
const summary = obsAnnotations.col(name).summarizeCategorical();
|
||||||
const [categoryValues, categoryValueCounts] = topNCategories(
|
const [categoryValues, categoryValueCounts] = topNCategories(
|
||||||
colSchema,
|
colSchema,
|
||||||
summary,
|
summary,
|
||||||
@@ -203,7 +203,9 @@ export function pruneVarDataCache(varData, needed) {
|
|||||||
|
|
||||||
export function subsetAndResetGeneLists(state) {
|
export function subsetAndResetGeneLists(state) {
|
||||||
const { userDefinedGenes, diffexpGenes } = state;
|
const { userDefinedGenes, diffexpGenes } = state;
|
||||||
const newUserDefinedGenes = [].concat(userDefinedGenes, diffexpGenes).slice(0, globals.maxGenes);
|
const newUserDefinedGenes = []
|
||||||
|
.concat(userDefinedGenes, diffexpGenes)
|
||||||
|
.slice(0, globals.maxGenes);
|
||||||
const newDiffExpGenes = [];
|
const newDiffExpGenes = [];
|
||||||
return [newUserDefinedGenes, newDiffExpGenes];
|
return [newUserDefinedGenes, newDiffExpGenes];
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -52,7 +52,6 @@ import anndata
|
|||||||
import tiledb
|
import tiledb
|
||||||
import argparse
|
import argparse
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import pandas as pd
|
|
||||||
from os.path import splitext, basename
|
from os.path import splitext, basename
|
||||||
import json
|
import json
|
||||||
|
|
||||||
@@ -152,45 +151,64 @@ def write_cxg(adata, container, title, var_names=None, obs_names=None, about=Non
|
|||||||
log(1, "\t...X created")
|
log(1, "\t...X created")
|
||||||
|
|
||||||
|
|
||||||
def _can_cast_to_float32(col):
|
"""
|
||||||
if col.dtype.kind == "f":
|
TODO: the code used to handle type inferencing should not be duplicated between
|
||||||
|
this tool and the server/common/utils code. When this tool is merged into
|
||||||
|
the cellxgene CLI, consolidate.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def dtype_to_schema(dtype):
|
||||||
|
if dtype == np.float32:
|
||||||
|
return (np.float32, {})
|
||||||
|
elif dtype == np.int32:
|
||||||
|
return (np.int32, {})
|
||||||
|
elif dtype == np.bool_:
|
||||||
|
return (np.uint8, {type: "boolean"})
|
||||||
|
elif dtype == np.str:
|
||||||
|
return (np.unicode, {"type": "string"})
|
||||||
|
elif dtype == "category":
|
||||||
|
typ, hint = cxg_type(dtype.categories)
|
||||||
|
return (typ, {"type": "categorical", "categories": dtype.categories.tolist()})
|
||||||
|
else:
|
||||||
|
raise TypeError(f"Annotations of type {dtype} are unsupported.")
|
||||||
|
|
||||||
|
|
||||||
|
def _can_cast_to_float32(array):
|
||||||
|
if array.dtype.kind == "f":
|
||||||
# force downcast for all floats
|
# force downcast for all floats
|
||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
def _can_cast_to_int32(col):
|
def _can_cast_to_int32(array):
|
||||||
if col.dtype.kind in ["i", "u"]:
|
if array.dtype.kind in ["i", "u"]:
|
||||||
if np.can_cast(col.dtype, np.int32):
|
if np.can_cast(array.dtype, np.int32):
|
||||||
return True
|
return True
|
||||||
ii32 = np.iinfo(np.int32)
|
ii32 = np.iinfo(np.int32)
|
||||||
if col.min() >= ii32.min and col.max() <= ii32.max:
|
if array.min() >= ii32.min and array.max() <= ii32.max:
|
||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
def cxg_type(col):
|
def cxg_type(array):
|
||||||
""" given a dtype, return an encoding dtype and any schema hints """
|
try:
|
||||||
dtype = col.dtype
|
return dtype_to_schema(array.dtype)
|
||||||
kind = dtype.kind
|
except TypeError:
|
||||||
if _can_cast_to_float32(col):
|
dtype = array.dtype
|
||||||
return (np.float32, {})
|
data_kind = dtype.kind
|
||||||
elif _can_cast_to_int32(col):
|
if _can_cast_to_float32(array):
|
||||||
return (np.int32, {})
|
return (np.float32, {})
|
||||||
elif dtype == np.bool_ or dtype == np.bool:
|
elif _can_cast_to_int32(array):
|
||||||
return (np.uint8, {type: "boolean"})
|
return (np.int32, {})
|
||||||
elif kind == "O" and isinstance(dtype, pd.CategoricalDtype):
|
elif data_kind == "O" and dtype == "object":
|
||||||
typ, hint = cxg_type(dtype.categories)
|
return (np.unicode, {"type": "string"})
|
||||||
hint["categories"] = dtype.categories.tolist()
|
else:
|
||||||
return (typ, hint)
|
raise TypeError(f"Annotations of type {dtype} are unsupported.")
|
||||||
elif kind == "O":
|
|
||||||
return (np.unicode, {"type": "string"})
|
|
||||||
else:
|
|
||||||
raise TypeError(f"Annotations of type {dtype} are unsupported by cellxgene.")
|
|
||||||
|
|
||||||
|
|
||||||
def cxg_dtype(col):
|
def cxg_dtype(array):
|
||||||
return cxg_type(col)[0]
|
return cxg_type(array)[0]
|
||||||
|
|
||||||
|
|
||||||
def create_dataframe(name, df, ctx):
|
def create_dataframe(name, df, ctx):
|
||||||
@@ -407,9 +425,8 @@ def sanitize_keys(keys):
|
|||||||
|
|
||||||
Masking out [~/.] and anything outside the ASCII range.
|
Masking out [~/.] and anything outside the ASCII range.
|
||||||
"""
|
"""
|
||||||
# p = re.compile(r"[^a-zA-Z0-9!\-_\.\*'\(\)&$@=;:\+ ,\?]")
|
|
||||||
p = re.compile(r"[^ -\.0-\[\]-\}]")
|
p = re.compile(r"[^ -\.0-\[\]-\}]")
|
||||||
clean_keys = {k: p.sub('_', k) for k in keys}
|
clean_keys = {k: p.sub("_", k) for k in keys}
|
||||||
|
|
||||||
used_keys = set()
|
used_keys = set()
|
||||||
clean_unique_keys = {}
|
clean_unique_keys = {}
|
||||||
@@ -422,7 +439,7 @@ def sanitize_keys(keys):
|
|||||||
# else, needs deduping.
|
# else, needs deduping.
|
||||||
counter = 1
|
counter = 1
|
||||||
while True:
|
while True:
|
||||||
candidate_name = v + '-' + str(counter)
|
candidate_name = v + "-" + str(counter)
|
||||||
if candidate_name not in used_keys:
|
if candidate_name not in used_keys:
|
||||||
used_keys.add(candidate_name)
|
used_keys.add(candidate_name)
|
||||||
clean_unique_keys[k] = candidate_name
|
clean_unique_keys[k] = candidate_name
|
||||||
|
|||||||
Reference in New Issue
Block a user