mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-07 18:18:12 +08:00
Original /initialize working
This commit is contained in:
@@ -36,18 +36,18 @@ class CXGDriver(metaclass=ABCMeta):
|
|||||||
"""
|
"""
|
||||||
Filter cells from data and return a subset of the data
|
Filter cells from data and return a subset of the data
|
||||||
:param filter:
|
:param filter:
|
||||||
:return: iterator through cell ids
|
:return: filtered dataframe
|
||||||
"""
|
"""
|
||||||
pass
|
pass
|
||||||
|
|
||||||
# Should this return the order of metadata fields as the first value?
|
# Should this return the order of metadata fields as the first value?
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
def metadata(self, cells_iterator, fields=None):
|
def metadata(self, df, fields=None):
|
||||||
"""
|
"""
|
||||||
Generator for metadata. Gets the metadata values cell by cell and returns all value
|
Generator for metadata. Gets the metadata values cell by cell and returns all value
|
||||||
or only certain values if names is not None
|
or only certain values if names is not None
|
||||||
|
|
||||||
:param cells_iterator: from filter cells, iterator for cellids
|
:param df: from filter_cells, dataframe
|
||||||
:param fields: list of keys for metadata to return, returns all metadata values if not set.
|
:param fields: list of keys for metadata to return, returns all metadata values if not set.
|
||||||
:return: Iterator for cellid + list of cells metadata values ex. [cell-id, val1, val2, val3]
|
:return: Iterator for cellid + list of cells metadata values ex. [cell-id, val1, val2, val3]
|
||||||
"""
|
"""
|
||||||
@@ -56,22 +56,26 @@ class CXGDriver(metaclass=ABCMeta):
|
|||||||
|
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
def create_graph(self, cells_iterator):
|
def create_graph(self, df):
|
||||||
"""
|
"""
|
||||||
Computes a n-d layout for cells through dimensionality reduction.
|
Computes a n-d layout for cells through dimensionality reduction.
|
||||||
:param cells_iterator: from filter cells, iterator for cellids
|
:param df: from filter_cells, dataframe
|
||||||
:return: Iterator for [cellid-1, pos1, pos2], [cellid-2, pos1, pos2]
|
:return: Iterator for [cellid-1, pos1, pos2], [cellid-2, pos1, pos2]
|
||||||
"""
|
"""
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
def diffexp(self, cells_iterator_1, cells_iterator_2):
|
def diffexp(self, df1, df2):
|
||||||
"""
|
"""
|
||||||
Computes the top differentially expressed genes between two clusters
|
Computes the top differentially expressed genes between two clusters
|
||||||
|
|
||||||
:param cells_iterator_1: First set of cell ids
|
:param df1: First set of cells
|
||||||
:param cells_iterator_2: Second set of cell ids
|
:param df2: Second set of cells
|
||||||
:return: Up in the air: I recommend [gene name, mean_expression_cells1, mean_expression_cells2, average_difference, statistic_value]
|
:return: Up in the air: I recommend [gene name, mean_expression_cells1, mean_expression_cells2, average_difference, statistic_value]
|
||||||
"""
|
"""
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
@abstractmethod
|
||||||
|
def expression(self, df):
|
||||||
|
pass
|
||||||
|
|||||||
@@ -201,9 +201,9 @@ class CellsAPI(Resource):
|
|||||||
}
|
}
|
||||||
# get query params
|
# get query params
|
||||||
filter = parse_filter(request.args, data.schema)
|
filter = parse_filter(request.args, data.schema)
|
||||||
filtered_data = list(data.filter_cells(filter))
|
filtered_data = data.filter_cells(filter)
|
||||||
payload["metadata"] = list(data.metadata(filtered_data))
|
payload["metadata"] = list(data.metadata(filtered_data))
|
||||||
payload["ranges"] = list(data.metadata_ranges(filtered_data))
|
payload["ranges"] = data.metadata_ranges(filtered_data)
|
||||||
payload["cellids"] = filtered_data
|
payload["cellids"] = filtered_data
|
||||||
payload["cellcount"] = len(payload["cellids"])
|
payload["cellcount"] = len(payload["cellids"])
|
||||||
payload["graph"] = list(data.create_graph(filtered_data))
|
payload["graph"] = list(data.create_graph(filtered_data))
|
||||||
|
|||||||
@@ -56,8 +56,7 @@ class ScanpyEngine(CXGDriver):
|
|||||||
:param filter:
|
:param filter:
|
||||||
:return: iterator through cell ids
|
:return: iterator through cell ids
|
||||||
"""
|
"""
|
||||||
cell_idx = np.ones((self.cell_count,), dtype=bool)
|
cell_idx = np.ones((self.cell_count(),), dtype=bool)
|
||||||
# TODO does this need to be a generator too?
|
|
||||||
for key, value in filter.items():
|
for key, value in filter.items():
|
||||||
if value["variable_type"] == "categorical":
|
if value["variable_type"] == "categorical":
|
||||||
key_idx = np.in1d(getattr(self.data.obs, key), value["query"])
|
key_idx = np.in1d(getattr(self.data.obs, key), value["query"])
|
||||||
@@ -71,34 +70,27 @@ class ScanpyEngine(CXGDriver):
|
|||||||
if max_:
|
if max_:
|
||||||
key_idx = np.array((getattr(self.data.obs, key) <= min_).data)
|
key_idx = np.array((getattr(self.data.obs, key) <= min_).data)
|
||||||
cell_idx = np.logical_and(cell_idx, key_idx)
|
cell_idx = np.logical_and(cell_idx, key_idx)
|
||||||
# If this is slow, could vectorize with logical array and then loop through that
|
return self.data[cell_idx, :]
|
||||||
for idx in range(self.cell_count):
|
|
||||||
if cell_idx[idx]:
|
|
||||||
yield self.data.obs.index[idx]
|
|
||||||
|
|
||||||
|
def metadata_ranges(self, df=None):
|
||||||
def metadata_ranges(self, cells_iterator=None):
|
|
||||||
metadata_ranges = {}
|
metadata_ranges = {}
|
||||||
if cells_iterator:
|
if not df:
|
||||||
data = self.data.obs.iloc[[i for i in cells_iterator], :]
|
df = self.data
|
||||||
else:
|
|
||||||
data = self.data.obs
|
|
||||||
for field in self.schema:
|
for field in self.schema:
|
||||||
if self.schema[field]["variabletype"] == "categorical":
|
if self.schema[field]["variabletype"] == "categorical":
|
||||||
group_by = field
|
group_by = field
|
||||||
if group_by == "CellName":
|
if group_by == "CellName":
|
||||||
group_by = 'cell_name'
|
group_by = 'cell_name'
|
||||||
metadata_ranges[field] = {"options": data.groupby(group_by).size().to_dict()}
|
metadata_ranges[field] = {"options": df.obs.groupby(group_by).size().to_dict()}
|
||||||
else:
|
else:
|
||||||
metadata_ranges[field] = {
|
metadata_ranges[field] = {
|
||||||
"range": {
|
"range": {
|
||||||
"min": data[field].min(),
|
"min": df.obs[field].min(),
|
||||||
"max": data[field].max()
|
"max": df.obs[field].max()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return metadata_ranges
|
return metadata_ranges
|
||||||
|
|
||||||
# Should this return the order of metadata fields as the first value?
|
|
||||||
def metadata(self, cells_iterator, fields=None):
|
def metadata(self, cells_iterator, fields=None):
|
||||||
"""
|
"""
|
||||||
Generator for metadata. Gets the metadata values cell by cell and returns all value
|
Generator for metadata. Gets the metadata values cell by cell and returns all value
|
||||||
@@ -129,13 +121,9 @@ class ScanpyEngine(CXGDriver):
|
|||||||
|
|
||||||
|
|
||||||
def diffexp(self, cells_iterator_1, cells_iterator_2):
|
def diffexp(self, cells_iterator_1, cells_iterator_2):
|
||||||
"""
|
pass
|
||||||
Computes the top differentially expressed genes between two clusters
|
|
||||||
|
|
||||||
:param cells_iterator_1: First set of cell ids
|
def expression(self, ):
|
||||||
:param cells_iterator_2: Second set of cell ids
|
|
||||||
:return: Up in the air: I recommend [gene name, mean_expression_cells1, mean_expression_cells2, average_difference, statistic_value]
|
|
||||||
"""
|
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user