Reorganize order of methods in engine and rest (#305)

This commit is contained in:
Charlotte Weaver
2018-10-10 16:44:05 -07:00
committed by GitHub
parent dce778f658
commit eb3b3fc59a
3 changed files with 287 additions and 258 deletions
+40 -30
View File
@@ -1,5 +1,14 @@
from abc import ABCMeta, abstractmethod from abc import ABCMeta, abstractmethod
"""
Sort order for methods
1. Initialize
2. Helper
3. Filter
4. Data & Metadata
5. Computation
"""
class CXGDriver(metaclass=ABCMeta): class CXGDriver(metaclass=ABCMeta):
@@ -50,7 +59,7 @@ class CXGDriver(metaclass=ABCMeta):
See REST specs for info on filter format: See REST specs for info on filter format:
https://docs.google.com/document/d/1Fxjp1SKtCk7l8QP9-7KAjGXL0eldi_qEnNT0NmlGzXI/edit#heading=h.8qc9q57amldx https://docs.google.com/document/d/1Fxjp1SKtCk7l8QP9-7KAjGXL0eldi_qEnNT0NmlGzXI/edit#heading=h.8qc9q57amldx
:param filter: dictionary with filter parames :param filter: dictionary with filter params
:return: View into scanpy object with cells/genes filtered :return: View into scanpy object with cells/genes filtered
""" """
pass pass
@@ -59,11 +68,37 @@ class CXGDriver(metaclass=ABCMeta):
def annotation(self, df, axis, fields=None): def annotation(self, df, axis, fields=None):
""" """
Gets annotation value for each observation Gets annotation value for each observation
:param axis:
:param df: from filter_cells, dataframe :param df: from filter_cells, dataframe
:param axis: string obs or var
:param fields: list of keys for annotation to return, returns all annotation values if not set. :param fields: list of keys for annotation to return, returns all annotation values if not set.
:return: dict: names - list of fields in order, data - list of lists or metadata [idx, val1, val2...] :return: dict: names - list of fields in order, data - list of lists or metadata
[observation ids, val1, val2...]
"""
pass
@abstractmethod
def data_frame(self, df, axis):
"""
Retrieves data for each variable for observations in data frame
:param df: from filter_cells, dataframe
:param axis: string obs or var
:return: {
"var": list of variable ids,
"obs": [cellid, var1 expression, var2 expression, ...],
}
"""
pass
@abstractmethod
def diffexp(self, df1, df2, top_n):
"""
Computes the top differentially expressed variables between two observation sets. If dataframes
contain a subset of variables, then statistics for all variables will be returned, otherwise
only the top N vars will be returned.
:param df1: from filter_cells, dataframe containing first set of observations
:param df2: from filter_cells, dataframe containing second set of observations
:param top_n: Limit results to top N (Top var mode only)
:return: top genes, stats and expression values for variables
""" """
pass pass
@@ -72,31 +107,6 @@ class CXGDriver(metaclass=ABCMeta):
""" """
Computes a n-d layout for cells through dimensionality reduction. Computes a n-d layout for cells through dimensionality reduction.
:param df: from filter_cells, dataframe :param df: from filter_cells, dataframe
:return: [cellid, x, y] :return: [cellid, x, y, ...]
"""
pass
@abstractmethod
def diffexp(self, df1, df2, genes):
"""
Computes the top differentially expressed variables between two observation sets. If dataframes
contain a subset of variables, then statistics for all variables will be returned, otherwise
only the top N vars will be returned.
:param df1: from filter_cells, dataframe containing first set of observations
:param df2: from filter_cells, dataframe containing second set of observations
:param topN: Limit results to top N (Top var mode only)
:return: top genes, stats and expression values for variables
"""
pass
@abstractmethod
def data_frame(self, df):
"""
Retrieves expression for each gene for cells in data frame
:param df: from filter_cells, dataframe
:return: {
"var": list of variable ids,
"obs": [cellid, var1 expression, var2 expression, ...],
}
""" """
pass pass
+187 -177
View File
@@ -12,6 +12,13 @@ from server.app.util.filter import parse_filter, QueryStringError
from server.app.util.models import FilterModel from server.app.util.models import FilterModel
from server.app.util.utils import MimeTypeError, get_mime_type from server.app.util.utils import MimeTypeError, get_mime_type
"""
Sort order for routes
1. Initialize
2. Data & Metadata
3. Computation
"""
class SchemaAPI(Resource): class SchemaAPI(Resource):
@swagger.doc({ @swagger.doc({
@@ -109,75 +116,6 @@ class ConfigAPI(Resource):
return make_response(jsonify(config), HTTPStatus.OK) return make_response(jsonify(config), HTTPStatus.OK)
class LayoutObsAPI(Resource):
@swagger.doc({
"summary": "Get the default layout for all observations.",
"tags": ["layout"],
"parameters": [],
"responses": {
"200": {
"description": "layout",
"examples": {
"application/json": {
"layout": {
"ndims": 2,
"coordinates": [
[0, 0.284483, 0.983744],
[1, 0.038844, 0.739444]
]
}
}
}
}
}
})
def get(self):
return make_response((jsonify({"layout": current_app.data.layout(current_app.data.data)})), HTTPStatus.OK)
@swagger.doc({
"summary": "Observation layout for filtered subset.",
"tags": ["layout"],
"parameters": [
{
"name": "filter",
"description": "Complex Filter",
"in": "body",
"schema": FilterModel
}
],
"responses": {
"200": {
"description": "layout",
"examples": {
"application/json": {
"layout": {
"ndims": 2,
"coordinates": [
[0, 0.284483, 0.983744],
[1, 0.038844, 0.739444]
]
}
}
}
},
"400": {
"description": "Malformed filter"
},
"403": {
"description": "Non-interactive request"
},
}
})
def post(self):
try:
df = current_app.data.filter_dataframe(request.get_json()["filter"])
except KeyError:
return make_response("Malformed filter", HTTPStatus.BAD_REQUEST)
if len(df.obs.index) > current_app.data.features["layout"]["obs"]["interactiveLimit"]:
return make_response("Non-interactive request", HTTPStatus.FORBIDDEN)
return make_response((jsonify({"layout": current_app.data.layout(df)})), HTTPStatus.OK)
class AnnotationsObsAPI(Resource): class AnnotationsObsAPI(Resource):
@swagger.doc({ @swagger.doc({
"summary": "Fetch annotations (metadata) for all observations.", "summary": "Fetch annotations (metadata) for all observations.",
@@ -367,112 +305,6 @@ class AnnotationsVarAPI(Resource):
return make_response(jsonify(annotation_response), HTTPStatus.OK) return make_response(jsonify(annotation_response), HTTPStatus.OK)
class DiffExpObsAPI(Resource):
@swagger.doc({
"summary": "Generate differential expression (DE) statistics for two specified subsets of data, "
"as indicated by the two provided observation complex filters",
"tags": ["diffexp"],
# TODO sort out params
# "parameters": [
# # {
# # "in": "body",
# # "name": "mode",
# # "type": "string",
# # "required": True,
# # "description": "topN or varFilter"
# # },
# {
# "in": "query",
# "name": "count",
# "type": "int32",
# "description": "TopN mode: how many vars to return"
# },
# {
# "in": "body",
# "name": "varFilter",
# "schema": FilterModel,
# "description": "varFilter: Complex filter, only var for which vars to return"
# },
# {
# "in": "body",
# "name": "set1",
# "schema": FilterModel,
# "required": True,
# "description": "Complex filter, only obs - observations in set1"
# },
# {
# "in": "body",
# "name": "set2",
# "schema": FilterModel,
# "description": "Complex filter, only obs - observations in set2. If not included, inverse of set1."
# },
# ],
"responses": {
"200": {
"description": "Statistics are encoded as an array of arrays, with fields ordered as: "
"varIndex, avgDiff, pVal, pValAdj, set1AvgExp, set2AvgExp",
"examples": {
"application/json": [
[328, -2.569489, 2.655706e-63, 3.642036e-57, 383.393, 583.9],
[1250, -2.569489, 2.655706e-63, 3.642036e-57, 383.393, 583.9],
]
}
},
"400": {
"description": "malformed filter"
},
"403": {
"description": "non-interactive request"
},
"501": {
"description": "diffexp is not implemented"
}
}
})
def post(self):
args = request.get_json()
# confirm mode is present and legal
try:
mode = DiffExpMode(args["mode"])
except KeyError:
return make_response("Error: mode is required", HTTPStatus.BAD_REQUEST)
except ValueError:
return make_response(f"Error: invalid mode option {args['mode']}", HTTPStatus.BAD_REQUEST)
# Validate filters
if mode == DiffExpMode.VAR_FILTER:
if "varFilter" not in args:
return make_response("varFilter is required when mode is set to varFilter ", HTTPStatus.BAD_REQUEST)
if Axis.OBS in args["varFilter"]["filter"]:
return make_response("Obs filter not allowed in varFilter", HTTPStatus.BAD_REQUEST)
if "set1" not in args:
return make_response("set1 is required.", HTTPStatus.BAD_REQUEST)
if Axis.VAR in args["set1"]["filter"]:
return make_response("Var filter not allowed for set1", HTTPStatus.BAD_REQUEST)
# set2
if "set2" not in args:
return make_response("Set2 as inverse of set1 is not implemented", HTTPStatus.NOT_IMPLEMENTED)
if Axis.VAR in args["set2"]["filter"]:
return make_response("Var filter not allowed for set2", HTTPStatus.BAD_REQUEST)
set1_filter = args["set1"]["filter"]
set2_filter = args.get("set2", {"filter": {}})["filter"]
if "varFilter" in args:
set1_filter[Axis.VAR] = args["varFilter"]["filter"][Axis.VAR]
set2_filter[Axis.VAR] = args["varFilter"]["filter"][Axis.VAR]
df1 = current_app.data.filter_dataframe(set1_filter, include_uns=False)
# TODO inverse
df2 = current_app.data.filter_dataframe(set2_filter, include_uns=False)
# exceeds size limit
if df1.shape[0] + df2.shape[0] > current_app.data.features["diffexp"]["interactiveLimit"]:
return make_response("Non-interactive request", HTTPStatus.FORBIDDEN)
# mode
count = args.get("count", None)
try:
diffexp = current_app.data.diffexp(df1, df2, count)
except ValueError as ve:
return make_response(ve.message, HTTPStatus.BAD_REQUEST)
return make_response(jsonify(diffexp), HTTPStatus.OK)
class DataObsAPI(Resource): class DataObsAPI(Resource):
@swagger.doc({ @swagger.doc({
"summary": "Get data (expression values) from the dataframe.", "summary": "Get data (expression values) from the dataframe.",
@@ -682,15 +514,193 @@ class DataVarAPI(Resource):
return make_response((jsonify(current_app.data.data_frame(df, axis=Axis.VAR))), HTTPStatus.OK) return make_response((jsonify(current_app.data.data_frame(df, axis=Axis.VAR))), HTTPStatus.OK)
class DiffExpObsAPI(Resource):
@swagger.doc({
"summary": "Generate differential expression (DE) statistics for two specified subsets of data, "
"as indicated by the two provided observation complex filters",
"tags": ["diffexp"],
# TODO sort out params
# "parameters": [
# # {
# # "in": "body",
# # "name": "mode",
# # "type": "string",
# # "required": True,
# # "description": "topN or varFilter"
# # },
# {
# "in": "query",
# "name": "count",
# "type": "int32",
# "description": "TopN mode: how many vars to return"
# },
# {
# "in": "body",
# "name": "varFilter",
# "schema": FilterModel,
# "description": "varFilter: Complex filter, only var for which vars to return"
# },
# {
# "in": "body",
# "name": "set1",
# "schema": FilterModel,
# "required": True,
# "description": "Complex filter, only obs - observations in set1"
# },
# {
# "in": "body",
# "name": "set2",
# "schema": FilterModel,
# "description": "Complex filter, only obs - observations in set2. If not included, inverse of set1."
# },
# ],
"responses": {
"200": {
"description": "Statistics are encoded as an array of arrays, with fields ordered as: "
"varIndex, avgDiff, pVal, pValAdj, set1AvgExp, set2AvgExp",
"examples": {
"application/json": [
[328, -2.569489, 2.655706e-63, 3.642036e-57, 383.393, 583.9],
[1250, -2.569489, 2.655706e-63, 3.642036e-57, 383.393, 583.9],
]
}
},
"400": {
"description": "malformed filter"
},
"403": {
"description": "non-interactive request"
},
"501": {
"description": "diffexp is not implemented"
}
}
})
def post(self):
args = request.get_json()
# confirm mode is present and legal
try:
mode = DiffExpMode(args["mode"])
except KeyError:
return make_response("Error: mode is required", HTTPStatus.BAD_REQUEST)
except ValueError:
return make_response(f"Error: invalid mode option {args['mode']}", HTTPStatus.BAD_REQUEST)
# Validate filters
if mode == DiffExpMode.VAR_FILTER:
if "varFilter" not in args:
return make_response("varFilter is required when mode is set to varFilter ", HTTPStatus.BAD_REQUEST)
if Axis.OBS in args["varFilter"]["filter"]:
return make_response("Obs filter not allowed in varFilter", HTTPStatus.BAD_REQUEST)
if "set1" not in args:
return make_response("set1 is required.", HTTPStatus.BAD_REQUEST)
if Axis.VAR in args["set1"]["filter"]:
return make_response("Var filter not allowed for set1", HTTPStatus.BAD_REQUEST)
# set2
if "set2" not in args:
return make_response("Set2 as inverse of set1 is not implemented", HTTPStatus.NOT_IMPLEMENTED)
if Axis.VAR in args["set2"]["filter"]:
return make_response("Var filter not allowed for set2", HTTPStatus.BAD_REQUEST)
set1_filter = args["set1"]["filter"]
set2_filter = args.get("set2", {"filter": {}})["filter"]
if "varFilter" in args:
set1_filter[Axis.VAR] = args["varFilter"]["filter"][Axis.VAR]
set2_filter[Axis.VAR] = args["varFilter"]["filter"][Axis.VAR]
df1 = current_app.data.filter_dataframe(set1_filter, include_uns=False)
# TODO inverse
df2 = current_app.data.filter_dataframe(set2_filter, include_uns=False)
# exceeds size limit
if df1.shape[0] + df2.shape[0] > current_app.data.features["diffexp"]["interactiveLimit"]:
return make_response("Non-interactive request", HTTPStatus.FORBIDDEN)
# mode
count = args.get("count", None)
try:
diffexp = current_app.data.diffexp(df1, df2, count)
except ValueError as ve:
return make_response(ve.message, HTTPStatus.BAD_REQUEST)
return make_response(jsonify(diffexp), HTTPStatus.OK)
class LayoutObsAPI(Resource):
@swagger.doc({
"summary": "Get the default layout for all observations.",
"tags": ["layout"],
"parameters": [],
"responses": {
"200": {
"description": "layout",
"examples": {
"application/json": {
"layout": {
"ndims": 2,
"coordinates": [
[0, 0.284483, 0.983744],
[1, 0.038844, 0.739444]
]
}
}
}
}
}
})
def get(self):
return make_response((jsonify({"layout": current_app.data.layout(current_app.data.data)})), HTTPStatus.OK)
@swagger.doc({
"summary": "Observation layout for filtered subset.",
"tags": ["layout"],
"parameters": [
{
"name": "filter",
"description": "Complex Filter",
"in": "body",
"schema": FilterModel
}
],
"responses": {
"200": {
"description": "layout",
"examples": {
"application/json": {
"layout": {
"ndims": 2,
"coordinates": [
[0, 0.284483, 0.983744],
[1, 0.038844, 0.739444]
]
}
}
}
},
"400": {
"description": "Malformed filter"
},
"403": {
"description": "Non-interactive request"
},
}
})
def post(self):
try:
df = current_app.data.filter_dataframe(request.get_json()["filter"])
except KeyError:
return make_response("Malformed filter", HTTPStatus.BAD_REQUEST)
if len(df.obs.index) > current_app.data.features["layout"]["obs"]["interactiveLimit"]:
return make_response("Non-interactive request", HTTPStatus.FORBIDDEN)
return make_response((jsonify({"layout": current_app.data.layout(df)})), HTTPStatus.OK)
def get_api_resources(): def get_api_resources():
bp = Blueprint("api", __name__, url_prefix="/api/v0.2") bp = Blueprint("api", __name__, url_prefix="/api/v0.2")
api = Api(bp, add_api_spec_resource=False) api = Api(bp, add_api_spec_resource=False)
# Initialization routes
api.add_resource(SchemaAPI, "/schema") api.add_resource(SchemaAPI, "/schema")
api.add_resource(ConfigAPI, "/config") api.add_resource(ConfigAPI, "/config")
api.add_resource(LayoutObsAPI, "/layout/obs") # Data routes
api.add_resource(AnnotationsObsAPI, "/annotations/obs") api.add_resource(AnnotationsObsAPI, "/annotations/obs")
api.add_resource(DiffExpObsAPI, "/diffexp/obs")
api.add_resource(AnnotationsVarAPI, "/annotations/var") api.add_resource(AnnotationsVarAPI, "/annotations/var")
api.add_resource(DataObsAPI, "/data/obs") api.add_resource(DataObsAPI, "/data/obs")
api.add_resource(DataVarAPI, "/data/var") api.add_resource(DataVarAPI, "/data/var")
# Computation routes
api.add_resource(DiffExpObsAPI, "/diffexp/obs")
api.add_resource(LayoutObsAPI, "/layout/obs")
return api return api
+54 -45
View File
@@ -11,6 +11,15 @@ from scipy import stats
from server.app.driver.driver import CXGDriver from server.app.driver.driver import CXGDriver
from server.app.util.constants import Axis, DEFAULT_TOP_N, DiffExpMode from server.app.util.constants import Axis, DEFAULT_TOP_N, DiffExpMode
"""
Sort order for methods
1. Initialize
2. Helper
3. Filter
4. Data & Metadata
5. Computation
"""
class ScanpyEngine(CXGDriver): class ScanpyEngine(CXGDriver):
@@ -151,7 +160,7 @@ class ScanpyEngine(CXGDriver):
Filter data based on index. ex. [1, 3, [111:200]] Filter data based on index. ex. [1, 3, [111:200]]
:param filter: subset of filter dict for obs/var:index :param filter: subset of filter dict for obs/var:index
:param index: np logical vector containing true for passing false for failing filter :param index: np logical vector containing true for passing false for failing filter
:param axis: Axis :param axis: string obs or var
:return: np logical vector for whether the data passes the filter :return: np logical vector for whether the data passes the filter
""" """
if axis == Axis.OBS: if axis == Axis.OBS:
@@ -194,9 +203,8 @@ class ScanpyEngine(CXGDriver):
def annotation(self, df, axis, fields=None): def annotation(self, df, axis, fields=None):
""" """
Gets annotation value for each observation Gets annotation value for each observation
:param axis:
:param df: from filter_cells, dataframe :param df: from filter_cells, dataframe
:param axis: string obs or var
:param fields: list of keys for annotation to return, returns all annotation values if not set. :param fields: list of keys for annotation to return, returns all annotation values if not set.
:return: dict: names - list of fields in order, data - list of lists or metadata :return: dict: names - list of fields in order, data - list of lists or metadata
[observation ids, val1, val2...] [observation ids, val1, val2...]
@@ -211,24 +219,37 @@ class ScanpyEngine(CXGDriver):
} }
# @cache.memoize() # @cache.memoize()
def layout(self, df): def data_frame(self, df, axis):
""" """
Computes a n-d layout for cells through dimensionality reduction. Retrieves data for each variable for observations in data frame
:param df: from filter_cells, dataframe :param df: from filter_cells, dataframe
:return: [cellid, x, y, ...] :param axis: string obs or var
""" :return: {
# TODO Filtering cells is fine, but filtering genes does nothing because the neighbors are "var": list of variable ids,
# calculated using the original vars (geneset) and this doesn’t get updated when you use less. "obs": [cellid, var1 expression, var2 expression, ...],
# Need to recalculate neighbors (long) if user requests new layout filtered by var
getattr(sc.tl, self.layout_method)(df, random_state=123)
df_layout = df.obsm[f"X_{self.layout_method}"]
normalized_layout = DataFrame((df_layout - df_layout.min()) / (df_layout.max() - df_layout.min()),
index=df.obs.index)
return {
"ndims": normalized_layout.shape[1],
# reset_index gets obs' id into output
"coordinates": normalized_layout.reset_index().values.tolist()
} }
"""
var_idx = df.var.index.tolist()
obs_idx = df.obs.index.tolist()
values = df.X
df_shape = df.shape
if df_shape[0] == 1:
values = values[None, :]
elif df_shape[1] == 1:
values = values[:, None]
if axis == Axis.OBS:
expression = DataFrame(values, index=obs_idx)
result = {
"var": var_idx,
"obs": expression.reset_index().values.tolist()
}
else:
expression = DataFrame(values.T, index=var_idx)
result = {
"obs": obs_idx,
"var": expression.reset_index().values.tolist(),
}
return result
# @cache.memoize() # @cache.memoize()
def diffexp(self, df1, df2, top_n=None): def diffexp(self, df1, df2, top_n=None):
@@ -238,7 +259,7 @@ class ScanpyEngine(CXGDriver):
only the top N vars will be returned. only the top N vars will be returned.
:param df1: from filter_cells, dataframe containing first set of observations :param df1: from filter_cells, dataframe containing first set of observations
:param df2: from filter_cells, dataframe containing second set of observations :param df2: from filter_cells, dataframe containing second set of observations
:param topN: Limit results to top N (Top var mode only) :param top_n: Limit results to top N (Top var mode only)
:return: top genes, stats and expression values for variables :return: top genes, stats and expression values for variables
""" """
# If not the same genes, test is wrong! # If not the same genes, test is wrong!
@@ -280,33 +301,21 @@ class ScanpyEngine(CXGDriver):
return sorted(result, key=lambda gene: gene[0]) return sorted(result, key=lambda gene: gene[0])
# @cache.memoize() # @cache.memoize()
def data_frame(self, df, axis): def layout(self, df):
""" """
Retrieves data for each variable for observations in data frame Computes a n-d layout for cells through dimensionality reduction.
:param df: from filter_cells, dataframe :param df: from filter_cells, dataframe
:return: { :return: [cellid, x, y, ...]
"var": list of variable ids,
"obs": [cellid, var1 expression, var2 expression, ...],
}
""" """
var_idx = df.var.index.tolist() # TODO Filtering cells is fine, but filtering genes does nothing because the neighbors are
obs_idx = df.obs.index.tolist() # calculated using the original vars (geneset) and this doesn’t get updated when you use less.
values = df.X # Need to recalculate neighbors (long) if user requests new layout filtered by var
df_shape = df.shape getattr(sc.tl, self.layout_method)(df, random_state=123)
if df_shape[0] == 1: df_layout = df.obsm[f"X_{self.layout_method}"]
values = values[None, :] normalized_layout = DataFrame((df_layout - df_layout.min()) / (df_layout.max() - df_layout.min()),
elif df_shape[1] == 1: index=df.obs.index)
values = values[:, None] return {
if axis == Axis.OBS: "ndims": normalized_layout.shape[1],
expression = DataFrame(values, index=obs_idx) # reset_index gets obs' id into output
result = { "coordinates": normalized_layout.reset_index().values.tolist()
"var": var_idx,
"obs": expression.reset_index().values.tolist()
} }
else:
expression = DataFrame(values.T, index=var_idx)
result = {
"obs": obs_idx,
"var": expression.reset_index().values.tolist(),
}
return result