Experimental - manual annotations (#837)

* icons, partway

* redux for values

* onChange

* cancel

* annotations lifecycle for category names

* copy categorical

* edit category

* add Dataframe.withColsFrom

* render user annotations; default add/delete annotation category

* add label name to actions

* category name edit

* error checking improvements

* change schema field isUserAnnotation to writable

* always have an unassigned label; implement delete label

* implement add new label and edit label name

* label current cell selection

* fix select exact bug in crossfilter

* clean up categorical reducer

* fix tests

* remove debugging printf

* implement subset/reset for user annotations

* undo redo support for user annotations

* remove duplicate button from categories

* add modal

* remove obsolete duplicate annotation reducers

* remove old debugging printf

* connect modal to annotation create and dup

* initial full-stack wiring

* finish up end-to-end wiring

* fix existing unit tests

* fix pytests to match new schema API

* remove debugging printfs

* add label file rotation

* remove obsolete comment

* add fbs encode/decode tests

* add tests for writable annotations

* simplify code

* fix hashing bug with FBS encoding

* lint

* fix smoke tests

* improve error checking in Dataframe.withColsFrom

* add unit test for Dataframe.withColsFrom

* add unit test for Dataframe.columns and Dataframe.renameCol

* fix bug in FBS encode, add better error checks, refactor

* add FBS encode/decode test

* add clarifying comment

* clean up action type names; fix state inconsistency in crossfilter update

* change autosave timer to 2.5sec

* sort categorical metadata render order so it remains consistent

* add temporary autogenerated label for add-new-label operation

* fix hover-over label menu interference with cell highlighting

* remove debugging code

* add missing reducer cases & fix typo

* make dataframe memoize more general purpose

* add dev mode for annos

* fix error on select duplicate

* handle zero occupancy categories

* correctly maintain unclipped AND clipped world

* correctly handle zero length FBS matrix and label files

* ensure all writable categorical schema contains an unassigned category

* handle case where building occupancy stack for category with no members

* dialog for creating label, disable button if duplicate or empty

* visually separate writeable

* edit category

* fix edit category name

* remove debugging code

* fix edit annotation label

* visually define unassigned, change options

* Pull in requirements.txt from `master`

* label currently selected cells

* duplicate label

* lint

* fix pytest merge issues

* rename --label-file to --experimental-label-file

* remove debugging console log

* spelling error fix; fix bug found in PR review.

* lint
This commit is contained in:
Bruce Martin
2019-09-18 07:33:41 -04:00
committed by Colin Megill
parent ab2c423006
commit 3660a6cc27
51 changed files with 2823 additions and 337 deletions
+49
View File
@@ -0,0 +1,49 @@
"""
Helpers for user annotations / label_file parameter
"""
from os.path import exists, splitext, getsize
from os import remove, rename
import pandas as pd
def read_labels(fname):
if exists(fname) and getsize(fname) > 0:
return pd.read_csv(fname, dtype='category')
else:
return pd.DataFrame()
def write_labels(fname, df):
rotate_fname(fname)
if not df.empty:
df.to_csv(fname, index=False)
else:
open(fname, 'a').close()
def rotate_fname(fname):
"""
save N backups of file.
fname -> fname-0
fname-0 -> fname->1
...
fname-(N-1) -> fname-N
"""
def rotate(src, dst):
if exists(src):
if exists(dst):
remove(dst)
rename(src, dst)
rotation_size = 9 # rotation size
name, ext = splitext(fname)
# rotate existing files
for i in range(rotation_size - 1, 0, -1):
src = f"{name}-{i}{ext}"
tgt = f"{name}-{i+1}{ext}"
rotate(src, tgt)
tgt = f"{name}-1{ext}"
rotate(fname, tgt)
+111 -21
View File
@@ -1,4 +1,6 @@
import warnings
import copy
import threading
import numpy as np
import pandas
@@ -13,10 +15,12 @@ from server.app.util.errors import (
JSONEncodingValueError,
PrepareError,
ScanpyFileError,
DisabledFeatureError,
)
from server.app.util.utils import jsonify_scanpy, requires_data
from server.app.scanpy_engine.diffexp import diffexp_ttest
from server.app.util.fbs.matrix import encode_matrix_fbs
from server.app.util.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
from server.app.scanpy_engine.labels import read_labels, write_labels
"""
Sort order for methods
@@ -31,6 +35,8 @@ Sort order for methods
class ScanpyEngine(CXGDriver):
def __init__(self, data=None, args={}):
super().__init__(data, args)
# lock used to protect label file write ops
self.label_lock = threading.Lock()
if self.data:
self._validate_and_initialize()
@@ -47,6 +53,7 @@ class ScanpyEngine(CXGDriver):
"obs_names": None,
"var_names": None,
"diffexp_lfc_cutoff": 0.01,
"label_file": None,
}
@staticmethod
@@ -125,6 +132,29 @@ class ScanpyEngine(CXGDriver):
return True
return False
@staticmethod
def _get_col_type(col):
dtype = col.dtype
data_kind = dtype.kind
schema = {}
if ScanpyEngine._can_cast_to_float32(col):
schema["type"] = "float32"
elif ScanpyEngine._can_cast_to_int32(col):
schema["type"] = "int32"
elif dtype == np.bool_:
schema["type"] = "boolean"
elif data_kind == "O" and dtype == "object":
schema["type"] = "string"
elif data_kind == "O" and dtype == "category":
schema["type"] = "categorical"
schema["categories"] = dtype.categories.tolist()
else:
raise TypeError(
f"Annotations of type {dtype} are unsupported by cellxgene."
)
return schema
@requires_data
def _create_schema(self):
self.schema = {
@@ -148,25 +178,8 @@ class ScanpyEngine(CXGDriver):
for ax in Axis:
curr_axis = getattr(self.data, str(ax))
for ann in curr_axis:
ann_schema = {"name": ann}
dtype = curr_axis[ann].dtype
data_kind = dtype.kind
if self._can_cast_to_float32(curr_axis[ann]):
ann_schema["type"] = "float32"
elif self._can_cast_to_int32(curr_axis[ann]):
ann_schema["type"] = "int32"
elif dtype == np.bool_:
ann_schema["type"] = "boolean"
elif data_kind == "O" and dtype == "object":
ann_schema["type"] = "string"
elif data_kind == "O" and dtype == "category":
ann_schema["type"] = "categorical"
ann_schema["categories"] = curr_axis[ann].dtype.categories.tolist()
else:
raise TypeError(
f"Annotations of type {curr_axis[ann].dtype} are unsupported by cellxgene."
)
ann_schema = {"name": ann, "writable": False}
ann_schema.update(self._get_col_type(curr_axis[ann]))
self.schema["annotations"][ax]["columns"].append(ann_schema)
for layout in self.config['layout']:
@@ -177,7 +190,24 @@ class ScanpyEngine(CXGDriver):
}
self.schema["layout"]["obs"].append(layout_schema)
@requires_data
def get_schema(self):
schema = self.schema # base schema
# add label obs annotations as needed
if self.labels is not None:
schema = copy.deepcopy(schema)
for col in self.labels.columns:
col_schema = {
"name": col,
"writable": True,
}
col_schema.update(self._get_col_type(self.labels[col]))
schema["annotations"]["obs"]["columns"].append(col_schema)
return schema
def _load_data(self, data_locator):
# as of AnnData 0.6.19, backed mode performs initial load fast, but at the
# cost of significantly slower access to X data.
try:
# there is no guarantee data_locator indicates a local file. The AnnData
# API will only consume local file objects. If we get a non-local object,
@@ -203,6 +233,17 @@ class ScanpyEngine(CXGDriver):
f"Please check your input and try again."
)
if self.config["label_file"]:
try:
self.labels = read_labels(self.config["label_file"])
except Exception as e:
raise ScanpyFileError(
f"Error while loading label file: {e}, File must be in the .csv format, please check "
f"your input and try again."
)
else:
self.labels = None
@requires_data
def _validate_and_initialize(self):
# var and obs column names must be unique
@@ -214,6 +255,7 @@ class ScanpyEngine(CXGDriver):
self.cell_count = self.data.shape[0]
self.gene_count = self.data.shape[1]
self._default_and_validate_layouts()
self._validate_label_file()
self._create_schema()
@requires_data
@@ -297,6 +339,26 @@ class ScanpyEngine(CXGDriver):
f"annotations with more than 500 categories in the UI"
)
@requires_data
def _validate_label_file(self):
"""
labels is None if disabled, empty if enabled by no data
"""
if self.labels is None or self.labels.empty:
return
# all lables must have a name, which must be unique and not used in obs column names
if not self.labels.columns.is_unique:
raise KeyError(f"All column names specified in {self.config['label_file']} must be unique.")
duplicate_columns = list(set(self.labels.columns) & set(self.data.obs.columns))
if len(duplicate_columns) > 0:
raise KeyError(f"Labels file may not contain column names which overlap "
f"with h5ad obs columns {duplicate_columns}")
# labels must have same count as obs annotations
if self.labels.shape[0] != self.data.obs.shape[0]:
raise ValueError("Labels file must have same number of rows as h5ad file.")
@staticmethod
def _annotation_filter_to_mask(filter, d_axis, count):
mask = np.ones((count,), dtype=bool)
@@ -364,13 +426,41 @@ class ScanpyEngine(CXGDriver):
@requires_data
def annotation_to_fbs_matrix(self, axis, fields=None):
if axis == Axis.OBS:
df = self.data.obs
if self.labels is not None and not self.labels.empty:
df = pandas.concat([self.data.obs, self.labels], axis=1, join_axes=[self.data.obs.index], copy=False)
else:
df = self.data.obs
else:
df = self.data.var
if fields is not None and len(fields) > 0:
df = df[fields]
return encode_matrix_fbs(df, col_idx=df.columns)
@requires_data
def annotation_put_fbs(self, axis, fbs):
fname = self.config["label_file"]
if not fname or self.labels is None:
raise DisabledFeatureError("Writable annotations are not enabled")
if axis != Axis.OBS:
raise ValueError("Only OBS dimension access is supported")
new_label_df = decode_matrix_fbs(fbs)
# if any of the new column labels overlap with our existing labels, raise error
duplicate_columns = list(set(new_label_df.columns) & set(self.data.obs.columns))
if not new_label_df.columns.is_unique or len(duplicate_columns) > 0:
raise KeyError(f"Labels file may not contain column names which overlap "
f"with h5ad obs columns {duplicate_columns}")
# update our internal state and save it. Multi-threading often enabled,
# so treat this as a critical section critical section.
with self.label_lock:
self.labels = new_label_df
write_labels(fname, self.labels)
return jsonify_scanpy({"status": "OK"})
@staticmethod
def slice_columns(X, var_mask):
"""