issue #480 workaround (#484)

* only load annotation var names

* remove incorrect usage of var annotation data

* temporary workaround for issue #480

* lint

* issue warnings only once per item
This commit is contained in:
Bruce Martin
2018-11-29 16:33:21 -08:00
committed by GitHub
parent a83ec60308
commit af0d1f6fb2
3 changed files with 55 additions and 2 deletions
+1 -1
View File
@@ -24,7 +24,7 @@ const doInitialDataLoad = () =>
"config",
"schema",
"annotations/obs",
"annotations/var",
"annotations/var?annotation-name=name",
"layout/obs"
])
.map(r => `${globals.API.prefix}${globals.API.version}${r}`)
@@ -27,7 +27,9 @@ const renderGene = (fuzzySortResult, { handleClick, modifiers, query }) => {
<MenuItem
active={modifiers.active}
disabled={modifiers.disabled}
label={gene.n_counts}
// Use of annotations in this way is incorrect and dataset specific.
// See https://github.com/chanzuckerberg/cellxgene/issues/483
// label={gene.n_counts}
key={gene.name}
onClick={g => {
/* this fires when user clicks a menu item */
+51
View File
@@ -35,6 +35,10 @@ class ScanpyEngine(CXGDriver):
self.diffexp_options = ["ttest"]
self._create_schema()
# TODO: temporary work-arounds
self._IEEE754_X_warning_issued = False
self._IEEE754_special_values_workaround_annotations()
def _alias_annotation_names(self, axis, name):
"""
Do all user-specified annotation aliasing.
@@ -172,6 +176,52 @@ class ScanpyEngine(CXGDriver):
f"`cellxgene prepare --layout {self.layout_method} <datafile>` "
f"to solve this problem. ")
def _IEEE754_special_values_workaround_annotations(self):
"""
TODO: temporary workaround
Because all floating point data is serialized to JSON, and JSON has no means of representing
non-finite, floating point special values (NaN, +/-Infinity, etc), we include this temporary
work-around.
This will likely be removed in the future, contingent upon improved marshalling.
Where non-finite floating point is present in obs, var or X:
* issue a warning to the user that these values will be treated as zeros.
* set the value to zero within the in-memory data (self.data)
"""
for ax in Axis:
curr_axis = getattr(self.data, str(ax))
for ann in curr_axis:
dtype = curr_axis[ann].dtype
if dtype.kind == 'f':
not_finite = np.isfinite(curr_axis[ann]) == False # noqa: E712
if np.count_nonzero(not_finite) > 0:
warnings.warn(
f"{str(ax).title()} annotation '{ann}' contains floating point NaN or Infinities. "
f"These values will be treated as zero."
)
curr_axis[ann][not_finite] = 0
def _IEEE754_special_values_workaround_X(self, _X):
"""
TODO: temporary workaround
See comments in _IEEE754_special_values_workaround_annotations
"""
not_finite = np.isfinite(_X) == False # noqa: E712
if np.count_nonzero(not_finite) > 0:
_X[not_finite] = 0
if not self._IEEE754_X_warning_issued:
# only want to issue this warning once.
warnings.warn(
"Dataframe X contains floating point NaN or Infinities. "
"These values will be treated as zero."
)
self._IEEE754_X_warning_issued = True
return _X
def filter_dataframe(self, filter):
"""
Filter cells from data and return a subset of the data. They can operate on both obs and var dimension with
@@ -323,6 +373,7 @@ class ScanpyEngine(CXGDriver):
_X = _X.toarray()
var_index_sliced = self.data.var.index[var_selector]
obs_index_sliced = self.data.obs.index[obs_selector]
_X = self._IEEE754_special_values_workaround_X(_X)
if axis == Axis.OBS:
result = {
"var": var_index_sliced.tolist(),