mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-28 14:48:12 +08:00
renaming backend, cellxgene to server, client respectively
This commit is contained in:
@@ -0,0 +1,61 @@
|
||||
import os
|
||||
|
||||
from flask import Flask
|
||||
from flask_compress import Compress
|
||||
from flask_cors import CORS
|
||||
from flask_restful_swagger_2 import get_swagger_blueprint
|
||||
|
||||
from .web import webapp
|
||||
from .rest_api.rest import get_api_resources
|
||||
|
||||
app = Flask(__name__)
|
||||
Compress(app)
|
||||
CORS(app)
|
||||
|
||||
# Config
|
||||
CONFIG_FILE = os.environ.get("CXG_CONFIG_FILE", default="scanpy-test.cfg")
|
||||
CXG_DIR = os.environ.get("CXG_DIRECTORY", default="/Users/charlotteweaver/Documents/Git/cxg-v2/data/")
|
||||
SECRET_KEY = os.environ.get("CXG_SECRET_KEY", default="SparkleAndShine")
|
||||
ENGINE = os.environ.get("CXG_ENGINE", default="scanpy")
|
||||
TITLE = os.environ.get("DATASET_TITLE", default="PBMC 3K")
|
||||
# TODO remove the 2 when this is prod
|
||||
CXG_API_BASE = os.environ.get("CXG_API_BASE2", default="http://0.0.0.0:5005/api/")
|
||||
|
||||
if not CONFIG_FILE:
|
||||
raise ValueError("No config file set for Flask application")
|
||||
|
||||
# TODO check what is actually being configured here
|
||||
app.config.from_pyfile(os.path.join(CXG_DIR, "config", CONFIG_FILE), silent=True)
|
||||
app.config.update(
|
||||
SECRET_KEY=SECRET_KEY,
|
||||
CXG_API_BASE=CXG_API_BASE,
|
||||
ENGINE=ENGINE,
|
||||
DATA=CXG_DIR,
|
||||
DATASET_TITLE=TITLE
|
||||
)
|
||||
|
||||
app.config['PROFILE'] = True
|
||||
# app.wsgi_app = ProfilerMiddleware(app.wsgi_app, restrictions=[15])
|
||||
|
||||
# Application Data
|
||||
data = None
|
||||
if app.config["ENGINE"] == "scanpy":
|
||||
from .scanpy_engine.scanpy_engine import ScanpyEngine
|
||||
data = ScanpyEngine(app.config["DATA"], schema="data_schema.json")
|
||||
|
||||
REACTIVE_LIMIT = 1_000_000
|
||||
|
||||
# A list of swagger document objects
|
||||
docs = []
|
||||
resources = get_api_resources()
|
||||
docs.append(resources.get_swagger_doc())
|
||||
|
||||
|
||||
app.register_blueprint(webapp.bp)
|
||||
app.register_blueprint(resources.blueprint)
|
||||
app.register_blueprint(
|
||||
get_swagger_blueprint(docs, '/api/swagger', produces=["application/json"], title="cellxgene rest api",
|
||||
description='An API connecting ExpressionMatrix2 clustering algorithm to cellxgene'))
|
||||
|
||||
|
||||
app.add_url_rule('/', endpoint='index')
|
||||
@@ -0,0 +1,78 @@
|
||||
from abc import ABCMeta, abstractmethod
|
||||
|
||||
|
||||
class CXGDriver(metaclass=ABCMeta):
|
||||
def __init__(self, data, schema=None, graph_method=None, diffexp_method=None):
|
||||
self.data = self._load_data(data)
|
||||
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def _load_data(data):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def _load_or_infer_schema(data):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def _set_cell_ids(self):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def cells(self):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def cellids(self):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def genes(self):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def filter_cells(self, filter):
|
||||
"""
|
||||
Filter cells from data and return a subset of the data
|
||||
:param filter:
|
||||
:return: filtered dataframe
|
||||
"""
|
||||
pass
|
||||
|
||||
# Should this return the order of metadata fields as the first value?
|
||||
@abstractmethod
|
||||
def metadata(self, df, fields=None):
|
||||
"""
|
||||
Generator for metadata. Gets the metadata values cell by cell and returns all value
|
||||
or only certain values if names is not None
|
||||
|
||||
:param df: from filter_cells, dataframe
|
||||
:param fields: list of keys for metadata to return, returns all metadata values if not set.
|
||||
:return: Iterator for cellid + list of cells metadata values ex. [cell-id, val1, val2, val3]
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def create_graph(self, df):
|
||||
"""
|
||||
Computes a n-d layout for cells through dimensionality reduction.
|
||||
:param df: from filter_cells, dataframe
|
||||
:return: Iterator for [cellid-1, pos1, pos2], [cellid-2, pos1, pos2]
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def diffexp(self, df1, df2):
|
||||
"""
|
||||
Computes the top differentially expressed genes between two clusters
|
||||
|
||||
:param df1: First set of cells
|
||||
:param df2: Second set of cells
|
||||
:return: Up in the air: I recommend [gene name, mean_expression_cells1,
|
||||
mean_expression_cells2, average_difference, statistic_value]
|
||||
"""
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def expression(self, df):
|
||||
pass
|
||||
@@ -0,0 +1,441 @@
|
||||
from flask import (
|
||||
Blueprint, request
|
||||
)
|
||||
from flask_restful_swagger_2 import Api, swagger, Resource
|
||||
|
||||
from ..util.utils import make_payload
|
||||
from ..util.filter import parse_filter
|
||||
|
||||
|
||||
class InitializeAPI(Resource):
|
||||
@swagger.doc({
|
||||
'summary': 'get metadata schema, ranges for values, and cell count to initialize cellxgene app',
|
||||
'tags': ['initialize'],
|
||||
'parameters': [],
|
||||
'responses': {
|
||||
'200': {
|
||||
'description': 'initialization data for UI',
|
||||
'examples': {
|
||||
'application/json': {
|
||||
"data": {
|
||||
"cellcount": 3589,
|
||||
"options": {
|
||||
"Sample.type": {
|
||||
"options": {
|
||||
"Glioblastoma": 3589
|
||||
}
|
||||
},
|
||||
"Selection": {
|
||||
"options": {
|
||||
"Astrocytes(HEPACAM)": 714,
|
||||
"Endothelial(BSC)": 123,
|
||||
"Microglia(CD45)": 1108,
|
||||
"Neurons(Thy1)": 685,
|
||||
"Oligodendrocytes(GC)": 294,
|
||||
"Unpanned": 665
|
||||
}
|
||||
},
|
||||
"Splice_sites_AT.AC": {
|
||||
"range": {
|
||||
"max": 1025,
|
||||
"min": 152
|
||||
}
|
||||
},
|
||||
"Splice_sites_Annotated": {
|
||||
"range": {
|
||||
"max": 1075869,
|
||||
"min": 26
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema": {
|
||||
"CellName": {
|
||||
"displayname": "Name",
|
||||
"type": "string",
|
||||
"variabletype": "categorical"
|
||||
},
|
||||
"Class": {
|
||||
"displayname": "Class",
|
||||
"type": "string",
|
||||
"variabletype": "categorical"
|
||||
},
|
||||
"ERCC_reads": {
|
||||
"displayname": "ERCC Reads",
|
||||
"type": "int",
|
||||
"variabletype": "continuous"
|
||||
},
|
||||
"ERCC_to_non_ERCC": {
|
||||
"displayname": "ERCC:Non-ERCC",
|
||||
"type": "float",
|
||||
"variabletype": "continuous"
|
||||
},
|
||||
"Genes_detected": {
|
||||
"displayname": "Genes Detected",
|
||||
"type": "int",
|
||||
"variabletype": "continuous"
|
||||
}
|
||||
},
|
||||
"genes": ["1/2-SBSRNA4", "A1BG", "A1BG-AS1"]
|
||||
|
||||
},
|
||||
"status": {
|
||||
"error": False,
|
||||
"errormessage": ""
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
def get(self):
|
||||
from app import data, REACTIVE_LIMIT
|
||||
return make_payload({
|
||||
"schema": data.schema,
|
||||
"cellcount": data.cell_count,
|
||||
"reactivelimit": REACTIVE_LIMIT,
|
||||
"genes": data.genes(),
|
||||
"ranges": data.metadata_ranges(),
|
||||
|
||||
})
|
||||
|
||||
|
||||
class CellsAPI(Resource):
|
||||
@swagger.doc({
|
||||
'summary': 'filter based on metadata fields to get a subset cells, expression data, and metadata',
|
||||
'tags': ['cells'],
|
||||
'description': "Cells takes query parameters defined in the schema retrieved from the /initialize enpoint. "
|
||||
"<br>For categorical metadata keys filter based on `key=value` <br>"
|
||||
" For continuous metadata keys filter by `key=min,max`<br> Either value "
|
||||
"can be replaced by a \*. To have only a minimum value `key=min,\*` To have only a maximum "
|
||||
"value `key=\*,max` <br>Graph data (if retrieved) is normalized"
|
||||
" To only retrieve cells that don't have a value for the key filter by `key`",
|
||||
'parameters': [],
|
||||
|
||||
'responses': {
|
||||
'200': {
|
||||
'description': 'initialization data for UI',
|
||||
'examples': {
|
||||
'application/json': {
|
||||
"data": {
|
||||
"badmetadatacount": 0,
|
||||
"cellcount": 0,
|
||||
"cellids": ["..."],
|
||||
"metadata": [
|
||||
{
|
||||
"CellName": "1001000173.G8",
|
||||
"Class": "Neoplastic",
|
||||
"Cluster_2d": "11",
|
||||
"Cluster_2d_color": "#8C564B",
|
||||
"Cluster_CNV": "1",
|
||||
"Cluster_CNV_color": "#1F77B4",
|
||||
"ERCC_reads": "152104",
|
||||
"ERCC_to_non_ERCC": "0.562454470489481",
|
||||
"Genes_detected": "1962",
|
||||
"Location": "Tumor",
|
||||
"Location.color": "#FF7F0E",
|
||||
"Multimapping_reads_percent": "2.67",
|
||||
"Neoplastic": "Neoplastic",
|
||||
"Non_ERCC_reads": "270429",
|
||||
"Sample.name": "BT_S2",
|
||||
"Sample.name.color": "#AEC7E8",
|
||||
"Sample.type": "Glioblastoma",
|
||||
"Sample.type.color": "#1F77B4",
|
||||
"Selection": "Unpanned",
|
||||
"Selection.color": "#98DF8A",
|
||||
"Splice_sites_AT.AC": "102",
|
||||
"Splice_sites_Annotated": "122397",
|
||||
"Splice_sites_GC.AG": "761",
|
||||
"Splice_sites_GT.AG": "125741",
|
||||
"Splice_sites_non_canonical": "56",
|
||||
"Splice_sites_total": "126660",
|
||||
"Total_reads": "1741039",
|
||||
"Unique_reads": "1400382",
|
||||
"Unique_reads_percent": "80.43",
|
||||
"Unmapped_mismatch": "2.15",
|
||||
"Unmapped_other": "0.18",
|
||||
"Unmapped_short": "14.56",
|
||||
"housekeeping_cluster": "2",
|
||||
"housekeeping_cluster_color": "#AEC7E8",
|
||||
"recluster_myeloid": "NA",
|
||||
"recluster_myeloid_color": "NA"
|
||||
},
|
||||
],
|
||||
"reactive": True,
|
||||
"graph": [
|
||||
[
|
||||
"1001000173.G8",
|
||||
0.93836,
|
||||
0.28623
|
||||
],
|
||||
|
||||
[
|
||||
"1001000173.D4",
|
||||
0.1662,
|
||||
0.79438
|
||||
]
|
||||
|
||||
],
|
||||
"status": {
|
||||
"error": False,
|
||||
"errormessage": ""
|
||||
}
|
||||
|
||||
},
|
||||
}
|
||||
},
|
||||
},
|
||||
|
||||
'400': {
|
||||
'description': 'bad query params',
|
||||
}
|
||||
}
|
||||
})
|
||||
def get(self):
|
||||
from app import data
|
||||
payload = {
|
||||
"cellids": [],
|
||||
"metadata": [],
|
||||
"cellcount": 0,
|
||||
"graph": [],
|
||||
"ranges": {},
|
||||
}
|
||||
# get query params
|
||||
filter = parse_filter(request.args, data.schema)
|
||||
filtered_data = data.filter_cells(filter)
|
||||
payload["metadata"] = data.metadata(filtered_data)
|
||||
payload["ranges"] = data.metadata_ranges(filtered_data)
|
||||
payload["graph"] = data.create_graph(filtered_data)
|
||||
payload["cellids"] = data.cellids(filtered_data)
|
||||
payload["cellcount"] = len(payload["cellids"])
|
||||
return make_payload(payload)
|
||||
|
||||
|
||||
class ExpressionAPI(Resource):
|
||||
@swagger.doc({
|
||||
'summary': 'Json with gene list and expression data by cell, limited to first 40 cells',
|
||||
'tags': ['expression'],
|
||||
'parameters': [
|
||||
{
|
||||
'name': 'include_unexpressed_genes',
|
||||
'description': "Include genes that have 0 expression across all cells in set",
|
||||
'in': 'path',
|
||||
'type': 'bool',
|
||||
}
|
||||
],
|
||||
'responses': {
|
||||
'200': {
|
||||
'description': 'Json for heatmap',
|
||||
'examples': {
|
||||
'application/json': {
|
||||
"data": {
|
||||
"cells": [
|
||||
{
|
||||
"cellname": "1/2-SBSRNA4",
|
||||
"e": [0, 0, 214, 0, 0]
|
||||
},
|
||||
],
|
||||
"genes": [
|
||||
"1001000173.G8",
|
||||
"1001000173.D4",
|
||||
"1001000173.B4",
|
||||
"1001000173.A2",
|
||||
"1001000173.E2"
|
||||
],
|
||||
"nonzero_gene_count": 2857
|
||||
},
|
||||
"status": {
|
||||
"error": False,
|
||||
"errormessage": ""
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
def get(self):
|
||||
from app import data
|
||||
expression_data = data.expression()
|
||||
return make_payload(expression_data)
|
||||
|
||||
@swagger.doc({
|
||||
'summary': 'Json with gene list and expression data by cell',
|
||||
'tags': ['expression'],
|
||||
'parameters': [
|
||||
{
|
||||
'name': 'body',
|
||||
'in': 'body',
|
||||
"schema": {
|
||||
"example": {
|
||||
"celllist": ["1001000173.G8", "1001000173.D4"],
|
||||
"genelist": ["1/2-SBSRNA4", "A1BG", "A1BG-AS1", "A1CF", "A2LD1", "A2M", "A2ML1", "A2MP1",
|
||||
"A4GALT"],
|
||||
"include_unexpressed_genes": True,
|
||||
}
|
||||
|
||||
}
|
||||
},
|
||||
],
|
||||
'responses': {
|
||||
'200': {
|
||||
'description': 'Json for expressiondata',
|
||||
'examples': {
|
||||
'application/json': {
|
||||
"data": {
|
||||
"cells": [
|
||||
{
|
||||
"cellname": "1001000173.D4",
|
||||
"e": [0, 0]
|
||||
},
|
||||
{
|
||||
"cellname": "1001000173.G8",
|
||||
"e": [0, 0]
|
||||
}
|
||||
],
|
||||
"genes": [
|
||||
"ABCD4",
|
||||
"ZWINT"
|
||||
],
|
||||
"nonzero_gene_count": 2857
|
||||
},
|
||||
"status": {
|
||||
"error": False,
|
||||
"errormessage": ""
|
||||
}
|
||||
|
||||
}
|
||||
}
|
||||
},
|
||||
'400': {
|
||||
'description': 'Required parameter missing/incorrect',
|
||||
}
|
||||
}
|
||||
})
|
||||
def post(self):
|
||||
from app import data
|
||||
args = request.get_json()
|
||||
cell_list = args.get('celllist', [])
|
||||
gene_list = args.get('genelist', [])
|
||||
if not cell_list and not gene_list:
|
||||
return make_payload([], "must include celllist and/or genelist parameter", 400)
|
||||
|
||||
expression_data = data.expression(cell_list, gene_list)
|
||||
|
||||
if cell_list and len(expression_data['cells']) < len(cell_list):
|
||||
return make_payload([], "Some cell ids not available", 400)
|
||||
if gene_list and len(expression_data['genes']) < len(gene_list):
|
||||
return make_payload([], "Some genes not available", 400)
|
||||
|
||||
return make_payload(expression_data)
|
||||
|
||||
|
||||
class DifferentialExpressionAPI(Resource):
|
||||
@swagger.doc({
|
||||
'summary': 'Get the top expressed genes for two cell sets. Calculated using t-test',
|
||||
'tags': ['expression'],
|
||||
'parameters': [
|
||||
{
|
||||
'name': 'body',
|
||||
'in': 'body',
|
||||
'schema': {
|
||||
"example": {
|
||||
"celllist1": ["1001000176.C12", "1001000176.C7", "1001000177.F11"],
|
||||
"celllist2": ["1001000012.D2", "1001000017.F10", "1001000033.C3", "1001000229.D4"],
|
||||
"num_genes": 5,
|
||||
"pval": 0.000001,
|
||||
},
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
'200': {
|
||||
'description': 'top expressed genes for cellset1, cellset2',
|
||||
'examples': {
|
||||
'application/json': {
|
||||
"data": {
|
||||
"celllist1": {
|
||||
"ave_diff": [
|
||||
432.0132935431362,
|
||||
12470.5623982637,
|
||||
957.0246880086814
|
||||
],
|
||||
"mean_expression_cellset1": [
|
||||
438.6185567010309,
|
||||
13315.536082474227,
|
||||
1076.5773195876288
|
||||
],
|
||||
"mean_expression_cellset2": [
|
||||
6.605263157894737,
|
||||
844.9736842105264,
|
||||
119.55263157894737
|
||||
],
|
||||
"pval": [
|
||||
3.8906598089944563e-35,
|
||||
1.9086226376018916e-25,
|
||||
7.847480544069826e-21
|
||||
],
|
||||
"topgenes": [
|
||||
"TMSB10",
|
||||
"FTL",
|
||||
"TMSB4X"
|
||||
]
|
||||
},
|
||||
"celllist2": {
|
||||
"ave_diff": [
|
||||
-6860.599158979924,
|
||||
-519.1314432989691,
|
||||
-10278.328269126423
|
||||
],
|
||||
"mean_expression_cellset1": [
|
||||
2.8350515463917527,
|
||||
0.6185567010309279,
|
||||
23.09278350515464
|
||||
],
|
||||
"mean_expression_cellset2": [
|
||||
6863.434210526316,
|
||||
519.75,
|
||||
10301.421052631578
|
||||
],
|
||||
"pval": [
|
||||
4.662891833748732e-44,
|
||||
3.6278087029927103e-37,
|
||||
8.396825170618402e-35
|
||||
],
|
||||
"topgenes": [
|
||||
"SPARCL1",
|
||||
"C1orf61",
|
||||
"CLU"
|
||||
]
|
||||
}
|
||||
},
|
||||
"status": {
|
||||
"error": False,
|
||||
"errormessage": ""
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
})
|
||||
def post(self):
|
||||
from app import data
|
||||
args = request.get_json()
|
||||
cell_list_1 = args.get('celllist1', [])
|
||||
cell_list_2 = args.get('celllist2', [])
|
||||
num_genes = args.get("num_genes", 7)
|
||||
pval = args.get('pval', 0.5)
|
||||
if not (cell_list_1 and cell_list_2):
|
||||
return make_payload([],
|
||||
"must include celllist1 and celllist2 parameters",
|
||||
400)
|
||||
data = data.diffexp(cell_list_1, cell_list_2, pval, num_genes)
|
||||
return make_payload(data)
|
||||
|
||||
|
||||
def get_api_resources():
|
||||
bp = Blueprint('api', __name__, url_prefix='/api/v2.0')
|
||||
api = Api(bp, add_api_spec_resource=False)
|
||||
api.add_resource(InitializeAPI, "/initialize")
|
||||
api.add_resource(CellsAPI, "/cells")
|
||||
api.add_resource(ExpressionAPI, "/expression")
|
||||
api.add_resource(DifferentialExpressionAPI, "/diffexp")
|
||||
return api
|
||||
@@ -0,0 +1,183 @@
|
||||
import os
|
||||
|
||||
import numpy as np
|
||||
import scanpy.api as sc
|
||||
from scipy import stats
|
||||
|
||||
from ..util.schema_parse import parse_schema
|
||||
from ..driver.driver import CXGDriver
|
||||
|
||||
|
||||
class ScanpyEngine(CXGDriver):
|
||||
|
||||
def __init__(self, data, schema=None, graph_method="umap", diffexp_method="ttest"):
|
||||
self.data = self._load_data(data)
|
||||
self.schema = self._load_or_infer_schema(data, schema)
|
||||
self._set_cell_ids()
|
||||
self.cell_count = self.data.shape[0]
|
||||
self.gene_count = self.data.shape[1]
|
||||
self.graph_method = graph_method
|
||||
self.diffexp_method = diffexp_method
|
||||
|
||||
@staticmethod
|
||||
def _load_data(data):
|
||||
return sc.read(os.path.join(data, "data.h5ad"))
|
||||
|
||||
@staticmethod
|
||||
def _load_or_infer_schema(data, schema):
|
||||
data_schema = None
|
||||
if not schema:
|
||||
pass
|
||||
else:
|
||||
data_schema = parse_schema(os.path.join(data, schema))
|
||||
return data_schema
|
||||
|
||||
def _set_cell_ids(self):
|
||||
self.data.obs['cxg_cell_id'] = list(range(self.data.obs.shape[0]))
|
||||
self.data.obs["cell_name"] = list(self.data.obs.index)
|
||||
self.data.obs.set_index('cxg_cell_id', inplace=True)
|
||||
|
||||
def cells(self):
|
||||
return list(self.data.obs.index)
|
||||
|
||||
def cellids(self, df=None):
|
||||
if df:
|
||||
return list(df.obs.index)
|
||||
else:
|
||||
return list(self.data.obs.index)
|
||||
|
||||
def genes(self):
|
||||
return self.data.var.index.tolist()
|
||||
|
||||
def filter_cells(self, filter):
|
||||
"""
|
||||
Filter cells from data and return a subset of the data
|
||||
:param filter:
|
||||
"""
|
||||
cell_idx = np.ones((self.cell_count,), dtype=bool)
|
||||
for key, value in filter.items():
|
||||
if value["variable_type"] == "categorical":
|
||||
key_idx = np.in1d(getattr(self.data.obs, key), value["query"])
|
||||
cell_idx = np.logical_and(cell_idx, key_idx)
|
||||
else:
|
||||
min_ = value["query"]["min"]
|
||||
max_ = value["query"]["max"]
|
||||
if min_:
|
||||
key_idx = np.array((getattr(self.data.obs, key) >= min_).data)
|
||||
cell_idx = np.logical_and(cell_idx, key_idx)
|
||||
if max_:
|
||||
key_idx = np.array((getattr(self.data.obs, key) <= min_).data)
|
||||
cell_idx = np.logical_and(cell_idx, key_idx)
|
||||
return self.data[cell_idx, :]
|
||||
|
||||
def metadata_ranges(self, df=None):
|
||||
metadata_ranges = {}
|
||||
if not df:
|
||||
df = self.data
|
||||
for field in self.schema:
|
||||
if self.schema[field]["variabletype"] == "categorical":
|
||||
group_by = field
|
||||
if group_by == "CellName":
|
||||
group_by = 'cell_name'
|
||||
metadata_ranges[field] = {"options": df.obs.groupby(group_by).size().to_dict()}
|
||||
else:
|
||||
metadata_ranges[field] = {
|
||||
"range": {
|
||||
"min": df.obs[field].min(),
|
||||
"max": df.obs[field].max()
|
||||
}
|
||||
}
|
||||
return metadata_ranges
|
||||
|
||||
def metadata(self, df, fields=None):
|
||||
"""
|
||||
Generator for metadata. Gets the metadata values cell by cell and returns all value
|
||||
or only certain values if names is not None
|
||||
|
||||
"""
|
||||
metadata = df.obs.to_dict(orient="records")
|
||||
for idx in range(len(metadata)):
|
||||
metadata[idx]["CellName"] = metadata[idx].pop("cell_name", None)
|
||||
return metadata
|
||||
|
||||
def create_graph(self, df):
|
||||
"""
|
||||
Computes a n-d layout for cells through dimensionality reduction.
|
||||
"""
|
||||
getattr(sc.tl, self.graph_method)(df)
|
||||
graph = df.obsm["X_{graph_method}".format(graph_method=self.graph_method)]
|
||||
normalized_graph = (graph - graph.min()) / (graph.max() - graph.min())
|
||||
return np.hstack((df.obs["cell_name"].values.reshape(len(df.obs.index), 1), normalized_graph)).tolist()
|
||||
|
||||
def diffexp(self, cell_list_1, cell_list_2, pval, num_genes):
|
||||
cells_idx_1 = np.in1d(self.data.obs["cell_name"], cell_list_1)
|
||||
cells_idx_2 = np.in1d(self.data.obs["cell_name"], cell_list_2)
|
||||
expression_1 = self.data.X[cells_idx_1, :]
|
||||
expression_2 = self.data.X[cells_idx_2, :]
|
||||
diff_exp = stats.ttest_ind(expression_1, expression_2)
|
||||
set1 = np.logical_and(diff_exp.pvalue < pval, diff_exp.statistic > 0)
|
||||
set2 = np.logical_and(diff_exp.pvalue < pval, diff_exp.statistic < 0)
|
||||
stat1 = diff_exp.statistic[set1]
|
||||
stat2 = diff_exp.statistic[set2]
|
||||
sort_set1 = np.argsort(stat1)[::-1]
|
||||
sort_set2 = np.argsort(stat2)
|
||||
pval1 = diff_exp.pvalue[set1][sort_set1]
|
||||
pval2 = diff_exp.pvalue[set2][sort_set2]
|
||||
mean_ex1_set1 = np.mean(expression_1[:, set1], axis=0)[sort_set1]
|
||||
mean_ex2_set1 = np.mean(expression_2[:, set1], axis=0)[sort_set1]
|
||||
mean_ex1_set2 = np.mean(expression_1[:, set2], axis=0)[sort_set2]
|
||||
mean_ex2_set2 = np.mean(expression_2[:, set2], axis=0)[sort_set2]
|
||||
mean_diff1 = mean_ex1_set1 - mean_ex2_set1
|
||||
mean_diff2 = mean_ex1_set2 - mean_ex2_set2
|
||||
genes_cellset_1 = self.data.var_names[set1][sort_set1]
|
||||
genes_cellset_2 = self.data.var_names[set2][sort_set2]
|
||||
return {
|
||||
"celllist1": {
|
||||
"topgenes": genes_cellset_1.tolist()[:num_genes],
|
||||
"mean_expression_cellset1": mean_ex1_set1.tolist()[:num_genes],
|
||||
"mean_expression_cellset2": mean_ex2_set1.tolist()[:num_genes],
|
||||
"pval": pval1.tolist()[:num_genes],
|
||||
"ave_diff": mean_diff1.tolist()[:num_genes]
|
||||
},
|
||||
"celllist2": {
|
||||
"topgenes": genes_cellset_2.tolist()[:num_genes],
|
||||
"mean_expression_cellset1": mean_ex1_set2.tolist()[:num_genes],
|
||||
"mean_expression_cellset2": mean_ex2_set2.tolist()[:num_genes],
|
||||
"pval": pval2.tolist()[:num_genes],
|
||||
"ave_diff": mean_diff2.tolist()[:num_genes]
|
||||
},
|
||||
}
|
||||
|
||||
def expression(self, cells=None, genes=None):
|
||||
"""
|
||||
:param df:
|
||||
:return:
|
||||
"""
|
||||
if cells:
|
||||
cells_idx = np.in1d(self.data.obs["cell_name"], cells)
|
||||
else:
|
||||
cells_idx = np.ones((self.cell_count,), dtype=bool)
|
||||
if genes:
|
||||
genes_idx = np.in1d(self.data.var_names, genes)
|
||||
else:
|
||||
genes_idx = np.ones((self.gene_count,), dtype=bool)
|
||||
index = np.ix_(cells_idx, genes_idx)
|
||||
expression = self.data.X[index]
|
||||
|
||||
if not genes:
|
||||
genes = self.data.var.index.tolist()
|
||||
if not cells:
|
||||
cells = self.data.obs["cell_name"].tolist()
|
||||
|
||||
cell_data = []
|
||||
for idx, cell in enumerate(cells):
|
||||
cell_data.append({
|
||||
"cellname": cell,
|
||||
"e": list(expression[idx]),
|
||||
})
|
||||
|
||||
return {
|
||||
"genes": genes,
|
||||
"cells": cell_data,
|
||||
"nonzero_gene_count": int(np.sum(expression.any(axis=0)))
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
class QueryStringError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def _convert_variable(datatype, variable):
|
||||
"""
|
||||
Convert variable to number (float/int)
|
||||
Used for dataset metadata and for query string
|
||||
:param datatype: type to convert to
|
||||
:param variable: value of variable
|
||||
:return: converted variable
|
||||
:raises: ValueError
|
||||
"""
|
||||
try:
|
||||
if variable and datatype == "int":
|
||||
variable = int(variable)
|
||||
elif variable and datatype == "float":
|
||||
variable = float(variable)
|
||||
return variable
|
||||
except ValueError:
|
||||
raise
|
||||
|
||||
|
||||
def parse_filter(filter, schema):
|
||||
"""
|
||||
{key: variable_type
|
||||
value_type
|
||||
query
|
||||
|
||||
:param filter:
|
||||
:param schema:
|
||||
:return:
|
||||
"""
|
||||
query = {}
|
||||
for key in filter:
|
||||
value = filter.getlist(key)
|
||||
if key not in schema:
|
||||
raise QueryStringError("Error: key {} not in metadata schema".format(key))
|
||||
query[key] = {
|
||||
"variable_type": schema[key]["variabletype"],
|
||||
"value_type": schema[key]["type"]
|
||||
}
|
||||
if query[key]["variable_type"] == "categorical":
|
||||
query[key]["query"] = _convert_variable(query[key]["value_type"], value)
|
||||
elif query[key]["variable_type"] == "continuous":
|
||||
value = value[0]
|
||||
try:
|
||||
min, max = value.split(",")
|
||||
except ValueError:
|
||||
raise QueryStringError("Error: min,max format required for range for key {}, got {}".format(key, value))
|
||||
if min == "*":
|
||||
min = None
|
||||
if max == "*":
|
||||
max = None
|
||||
try:
|
||||
query[key]["query"] = {
|
||||
"min": _convert_variable(query[key]["value_type"], min),
|
||||
"max": _convert_variable(query[key]["value_type"], max)
|
||||
}
|
||||
except ValueError:
|
||||
raise QueryStringError(
|
||||
"Error: expected type {} for key {}, got {}".format(query[key]["type"], key, value)
|
||||
)
|
||||
return query
|
||||
@@ -0,0 +1,7 @@
|
||||
import json
|
||||
|
||||
|
||||
def parse_schema(filename):
|
||||
with open(filename) as fh:
|
||||
schema = json.load(fh)
|
||||
return schema
|
||||
@@ -0,0 +1,40 @@
|
||||
import json
|
||||
|
||||
from numpy import float32, integer
|
||||
from flask import make_response, jsonify, Response
|
||||
|
||||
|
||||
class Float32JSONEncoder(json.JSONEncoder):
|
||||
def default(self, obj):
|
||||
if isinstance(obj, float32):
|
||||
return float(obj)
|
||||
elif isinstance(obj, integer):
|
||||
return int(obj)
|
||||
return json.JSONEncoder.default(self, obj)
|
||||
|
||||
|
||||
def make_payload(data, errormessage="", errorcode=200):
|
||||
"""
|
||||
Creates JSON respons for requests
|
||||
:param data: json data
|
||||
:param errormessage: error message
|
||||
:param errorcode: http error code
|
||||
:return: flask json repsonse
|
||||
"""
|
||||
error = False
|
||||
if errormessage:
|
||||
error = True
|
||||
# Questionable
|
||||
data = json.loads(json.dumps(data, cls=Float32JSONEncoder))
|
||||
return make_response(jsonify({
|
||||
"data": data,
|
||||
"status": {
|
||||
"error": error,
|
||||
"errormessage": errormessage,
|
||||
}
|
||||
}), errorcode)
|
||||
|
||||
|
||||
def make_streaming_response(data_generator, errorcode=200, content_type="application/json"):
|
||||
# TODO headers
|
||||
return Response(data_generator, status=errorcode, content_type=content_type)
|
||||
@@ -0,0 +1,97 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<title>CellxGene REST API - Swagger definition</title>
|
||||
<link href="https://fonts.googleapis.com/css?family=Open+Sans:400,700|Source+Code+Pro:300,600|Titillium+Web:400,600,700"
|
||||
rel="stylesheet">
|
||||
<link rel="stylesheet" href="https://cdnjs.cloudflare.com/ajax/libs/swagger-ui/3.2.1/swagger-ui.css"
|
||||
crossorigin="anonymous"/>
|
||||
<style>
|
||||
html {
|
||||
box-sizing: border-box;
|
||||
overflow: -moz-scrollbars-vertical;
|
||||
overflow-y: scroll;
|
||||
}
|
||||
|
||||
*,
|
||||
*:before,
|
||||
*:after {
|
||||
box-sizing: inherit;
|
||||
}
|
||||
|
||||
body {
|
||||
margin: 0;
|
||||
background: #fafafa;
|
||||
}
|
||||
</style>
|
||||
</head>
|
||||
|
||||
<body>
|
||||
|
||||
<svg xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink"
|
||||
style="position:absolute;width:0;height:0">
|
||||
<defs>
|
||||
<symbol viewBox="0 0 20 20" id="unlocked">
|
||||
<path d="M15.8 8H14V5.6C14 2.703 12.665 1 10 1 7.334 1 6 2.703 6 5.6V6h2v-.801C8 3.754 8.797 3 10 3c1.203 0 2 .754 2 2.199V8H4c-.553 0-1 .646-1 1.199V17c0 .549.428 1.139.951 1.307l1.197.387C5.672 18.861 6.55 19 7.1 19h5.8c.549 0 1.428-.139 1.951-.307l1.196-.387c.524-.167.953-.757.953-1.306V9.199C17 8.646 16.352 8 15.8 8z"></path>
|
||||
</symbol>
|
||||
|
||||
<symbol viewBox="0 0 20 20" id="locked">
|
||||
<path d="M15.8 8H14V5.6C14 2.703 12.665 1 10 1 7.334 1 6 2.703 6 5.6V8H4c-.553 0-1 .646-1 1.199V17c0 .549.428 1.139.951 1.307l1.197.387C5.672 18.861 6.55 19 7.1 19h5.8c.549 0 1.428-.139 1.951-.307l1.196-.387c.524-.167.953-.757.953-1.306V9.199C17 8.646 16.352 8 15.8 8zM12 8H8V5.199C8 3.754 8.797 3 10 3c1.203 0 2 .754 2 2.199V8z"/>
|
||||
</symbol>
|
||||
|
||||
<symbol viewBox="0 0 20 20" id="close">
|
||||
<path d="M14.348 14.849c-.469.469-1.229.469-1.697 0L10 11.819l-2.651 3.029c-.469.469-1.229.469-1.697 0-.469-.469-.469-1.229 0-1.697l2.758-3.15-2.759-3.152c-.469-.469-.469-1.228 0-1.697.469-.469 1.228-.469 1.697 0L10 8.183l2.651-3.031c.469-.469 1.228-.469 1.697 0 .469.469.469 1.229 0 1.697l-2.758 3.152 2.758 3.15c.469.469.469 1.229 0 1.698z"/>
|
||||
</symbol>
|
||||
|
||||
<symbol viewBox="0 0 20 20" id="large-arrow">
|
||||
<path d="M13.25 10L6.109 2.58c-.268-.27-.268-.707 0-.979.268-.27.701-.27.969 0l7.83 7.908c.268.271.268.709 0 .979l-7.83 7.908c-.268.271-.701.27-.969 0-.268-.269-.268-.707 0-.979L13.25 10z"/>
|
||||
</symbol>
|
||||
|
||||
<symbol viewBox="0 0 20 20" id="large-arrow-down">
|
||||
<path d="M17.418 6.109c.272-.268.709-.268.979 0s.271.701 0 .969l-7.908 7.83c-.27.268-.707.268-.979 0l-7.908-7.83c-.27-.268-.27-.701 0-.969.271-.268.709-.268.979 0L10 13.25l7.418-7.141z"/>
|
||||
</symbol>
|
||||
|
||||
|
||||
<symbol viewBox="0 0 24 24" id="jump-to">
|
||||
<path d="M19 7v4H5.83l3.58-3.59L8 6l-6 6 6 6 1.41-1.41L5.83 13H21V7z"/>
|
||||
</symbol>
|
||||
|
||||
<symbol viewBox="0 0 24 24" id="expand">
|
||||
<path d="M10 18h4v-2h-4v2zM3 6v2h18V6H3zm3 7h12v-2H6v2z"/>
|
||||
</symbol>
|
||||
|
||||
</defs>
|
||||
</svg>
|
||||
|
||||
<div id="swagger-ui"></div>
|
||||
|
||||
<script src="https://cdnjs.cloudflare.com/ajax/libs/swagger-ui/3.2.1/swagger-ui-bundle.js"
|
||||
crossorigin="anonymous"></script>
|
||||
<script src="https://cdnjs.cloudflare.com/ajax/libs/swagger-ui/3.2.1/swagger-ui-standalone-preset.js"
|
||||
crossorigin="anonymous"></script>
|
||||
<script>
|
||||
window.onload = function () {
|
||||
const ui = SwaggerUIBundle({
|
||||
url: window.location.origin + "/api/swagger.json",
|
||||
dom_id: '#swagger-ui',
|
||||
deepLinking: true,
|
||||
presets: [
|
||||
SwaggerUIBundle.presets.apis,
|
||||
SwaggerUIStandalonePreset
|
||||
],
|
||||
plugins: [
|
||||
SwaggerUIBundle.plugins.DownloadUrl
|
||||
],
|
||||
layout: "StandaloneLayout"
|
||||
});
|
||||
|
||||
window.ui = ui
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
|
||||
</html>
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
from flask import (
|
||||
Blueprint, render_template, url_for, current_app
|
||||
)
|
||||
|
||||
bp = Blueprint('webapp', __name__, template_folder='templates')
|
||||
|
||||
|
||||
@bp.route('/')
|
||||
def index():
|
||||
url_base = current_app.config["CXG_API_BASE"]
|
||||
dataset_title = current_app.config["DATASET_TITLE"]
|
||||
return render_template("index.html", prefix=url_base, datasetTitle=dataset_title)
|
||||
|
||||
|
||||
# renders swagger documentation
|
||||
@bp.route('/swagger')
|
||||
def swag():
|
||||
return render_template("swagger.html")
|
||||
|
||||
|
||||
# renders swagger documentation
|
||||
@bp.route('/favicon.png')
|
||||
def favicon():
|
||||
return url_for("static", filename="img/favicon.png")
|
||||
@@ -0,0 +1,3 @@
|
||||
from app import app
|
||||
|
||||
app.run(host='0.0.0.0', debug=True, port=5005)
|
||||
@@ -0,0 +1,54 @@
|
||||
import unittest
|
||||
import requests
|
||||
import json
|
||||
|
||||
|
||||
class EndPoints(unittest.TestCase):
|
||||
"""Test Case for endpoints"""
|
||||
|
||||
def setUp(self):
|
||||
# Local
|
||||
self.url_base = "http://0.0.0.0:5005/api/" + "v2.0/"
|
||||
self.session = requests.Session()
|
||||
|
||||
def test_cells(self):
|
||||
url = "{base}{endpoint}?{params}".format(base=self.url_base, endpoint="cells", params="&".join(
|
||||
["louvain=B cells"]))
|
||||
result = self.session.get(url)
|
||||
assert result.status_code == 200
|
||||
result_data = result.json()
|
||||
assert "B cells" in result_data["data"]["ranges"]["louvain"]["options"]
|
||||
url = "{base}{endpoint}?{params}".format(base=self.url_base, endpoint="cells", params="&".join(
|
||||
["louvain=B cells", "louvain=Megakaryocytes"]))
|
||||
result = self.session.get(url)
|
||||
assert result.status_code == 200
|
||||
result_data = result.json()
|
||||
assert "Megakaryocytes" in result_data["data"]["ranges"]["louvain"]["options"]
|
||||
|
||||
def test_initialize(self):
|
||||
url = "{base}{endpoint}".format(base=self.url_base, endpoint="initialize")
|
||||
result = self.session.get(url)
|
||||
assert result.status_code == 200
|
||||
result_data = result.json()
|
||||
assert result_data["data"]["cellcount"] == 2638
|
||||
assert len(result_data["data"]['ranges']['CellName']['options']) == 2638
|
||||
|
||||
|
||||
def test_expression_get(self):
|
||||
url = "{base}{endpoint}".format(base=self.url_base, endpoint="expression")
|
||||
result = self.session.get(url)
|
||||
assert result.status_code == 200
|
||||
|
||||
def test_expression_post(self):
|
||||
url = "{base}{endpoint}".format(base=self.url_base, endpoint="expression")
|
||||
result = self.session.post(url, data=json.dumps({"celllist": ["AAACATACAACCAC-1", "AACCGATGGTCATG-1"], "genelist": ["BACH1", "MIS18A", "ATP5O"]}), headers={'content-type': 'application/json'})
|
||||
assert result.status_code == 200
|
||||
result_data = result.json()
|
||||
assert len(result_data["data"]["cells"]) == 2
|
||||
assert len(result_data["data"]["cells"][0]['e']) == 3
|
||||
|
||||
def test_diffexp(self):
|
||||
url = "{base}{endpoint}".format(base=self.url_base, endpoint="diffexp")
|
||||
result = self.session.post(url, data=json.dumps({"celllist1": ["AAACATACAACCAC-1", "AACCGATGGTCATG-1"], "celllist2": ["CCGATAGACCTAAG-1", "GGTGGAGAAGTAGA-1"]}), headers={'content-type': 'application/json'})
|
||||
assert result.status_code == 200
|
||||
|
||||
Reference in New Issue
Block a user