Original /initialize working

This commit is contained in:
Charlotte Weaver
2018-06-22 11:00:22 -07:00
parent 1621d20bc4
commit 9b4bcb7f7b
3 changed files with 24 additions and 32 deletions
+12 -8
View File
@@ -36,18 +36,18 @@ class CXGDriver(metaclass=ABCMeta):
""" """
Filter cells from data and return a subset of the data Filter cells from data and return a subset of the data
:param filter: :param filter:
:return: iterator through cell ids :return: filtered dataframe
""" """
pass pass
# Should this return the order of metadata fields as the first value? # Should this return the order of metadata fields as the first value?
@abstractmethod @abstractmethod
def metadata(self, cells_iterator, fields=None): def metadata(self, df, fields=None):
""" """
Generator for metadata. Gets the metadata values cell by cell and returns all value Generator for metadata. Gets the metadata values cell by cell and returns all value
or only certain values if names is not None or only certain values if names is not None
:param cells_iterator: from filter cells, iterator for cellids :param df: from filter_cells, dataframe
:param fields: list of keys for metadata to return, returns all metadata values if not set. :param fields: list of keys for metadata to return, returns all metadata values if not set.
:return: Iterator for cellid + list of cells metadata values ex. [cell-id, val1, val2, val3] :return: Iterator for cellid + list of cells metadata values ex. [cell-id, val1, val2, val3]
""" """
@@ -56,22 +56,26 @@ class CXGDriver(metaclass=ABCMeta):
@abstractmethod @abstractmethod
def create_graph(self, cells_iterator): def create_graph(self, df):
""" """
Computes a n-d layout for cells through dimensionality reduction. Computes a n-d layout for cells through dimensionality reduction.
:param cells_iterator: from filter cells, iterator for cellids :param df: from filter_cells, dataframe
:return: Iterator for [cellid-1, pos1, pos2], [cellid-2, pos1, pos2] :return: Iterator for [cellid-1, pos1, pos2], [cellid-2, pos1, pos2]
""" """
pass pass
@abstractmethod @abstractmethod
def diffexp(self, cells_iterator_1, cells_iterator_2): def diffexp(self, df1, df2):
""" """
Computes the top differentially expressed genes between two clusters Computes the top differentially expressed genes between two clusters
:param cells_iterator_1: First set of cell ids :param df1: First set of cells
:param cells_iterator_2: Second set of cell ids :param df2: Second set of cells
:return: Up in the air: I recommend [gene name, mean_expression_cells1, mean_expression_cells2, average_difference, statistic_value] :return: Up in the air: I recommend [gene name, mean_expression_cells1, mean_expression_cells2, average_difference, statistic_value]
""" """
pass pass
@abstractmethod
def expression(self, df):
pass
+2 -2
View File
@@ -201,9 +201,9 @@ class CellsAPI(Resource):
} }
# get query params # get query params
filter = parse_filter(request.args, data.schema) filter = parse_filter(request.args, data.schema)
filtered_data = list(data.filter_cells(filter)) filtered_data = data.filter_cells(filter)
payload["metadata"] = list(data.metadata(filtered_data)) payload["metadata"] = list(data.metadata(filtered_data))
payload["ranges"] = list(data.metadata_ranges(filtered_data)) payload["ranges"] = data.metadata_ranges(filtered_data)
payload["cellids"] = filtered_data payload["cellids"] = filtered_data
payload["cellcount"] = len(payload["cellids"]) payload["cellcount"] = len(payload["cellids"])
payload["graph"] = list(data.create_graph(filtered_data)) payload["graph"] = list(data.create_graph(filtered_data))
+10 -22
View File
@@ -56,8 +56,7 @@ class ScanpyEngine(CXGDriver):
:param filter: :param filter:
:return: iterator through cell ids :return: iterator through cell ids
""" """
cell_idx = np.ones((self.cell_count,), dtype=bool) cell_idx = np.ones((self.cell_count(),), dtype=bool)
# TODO does this need to be a generator too?
for key, value in filter.items(): for key, value in filter.items():
if value["variable_type"] == "categorical": if value["variable_type"] == "categorical":
key_idx = np.in1d(getattr(self.data.obs, key), value["query"]) key_idx = np.in1d(getattr(self.data.obs, key), value["query"])
@@ -71,34 +70,27 @@ class ScanpyEngine(CXGDriver):
if max_: if max_:
key_idx = np.array((getattr(self.data.obs, key) <= min_).data) key_idx = np.array((getattr(self.data.obs, key) <= min_).data)
cell_idx = np.logical_and(cell_idx, key_idx) cell_idx = np.logical_and(cell_idx, key_idx)
# If this is slow, could vectorize with logical array and then loop through that return self.data[cell_idx, :]
for idx in range(self.cell_count):
if cell_idx[idx]:
yield self.data.obs.index[idx]
def metadata_ranges(self, df=None):
def metadata_ranges(self, cells_iterator=None):
metadata_ranges = {} metadata_ranges = {}
if cells_iterator: if not df:
data = self.data.obs.iloc[[i for i in cells_iterator], :] df = self.data
else:
data = self.data.obs
for field in self.schema: for field in self.schema:
if self.schema[field]["variabletype"] == "categorical": if self.schema[field]["variabletype"] == "categorical":
group_by = field group_by = field
if group_by == "CellName": if group_by == "CellName":
group_by = 'cell_name' group_by = 'cell_name'
metadata_ranges[field] = {"options": data.groupby(group_by).size().to_dict()} metadata_ranges[field] = {"options": df.obs.groupby(group_by).size().to_dict()}
else: else:
metadata_ranges[field] = { metadata_ranges[field] = {
"range": { "range": {
"min": data[field].min(), "min": df.obs[field].min(),
"max": data[field].max() "max": df.obs[field].max()
} }
} }
return metadata_ranges return metadata_ranges
# Should this return the order of metadata fields as the first value?
def metadata(self, cells_iterator, fields=None): def metadata(self, cells_iterator, fields=None):
""" """
Generator for metadata. Gets the metadata values cell by cell and returns all value Generator for metadata. Gets the metadata values cell by cell and returns all value
@@ -129,13 +121,9 @@ class ScanpyEngine(CXGDriver):
def diffexp(self, cells_iterator_1, cells_iterator_2): def diffexp(self, cells_iterator_1, cells_iterator_2):
""" pass
Computes the top differentially expressed genes between two clusters
:param cells_iterator_1: First set of cell ids def expression(self, ):
:param cells_iterator_2: Second set of cell ids
:return: Up in the air: I recommend [gene name, mean_expression_cells1, mean_expression_cells2, average_difference, statistic_value]
"""
pass pass