Fixing locust scale tests for cellxgene loading apis and adding a Github Actions workflow to run the tests every Sunday. (#1988)

This commit is contained in:
maniarathi
2020-11-24 09:08:57 -08:00
committed by GitHub
parent 16718f392f
commit ea70a35a01
4 changed files with 125 additions and 101 deletions
+30
View File
@@ -0,0 +1,30 @@
name: "Scale test cellxgene APIs for initial loading"
on:
schedule:
- cron: "0 0 * * Sun"
jobs:
locust-build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v2
- name: Set up Python 3.7
uses: actions/setup-python@v1
with:
python-version: 3.7
- name: Install dependencies
run: |
pip install -r server/test/locust/requirements-locust.txt
- name: Dev Scale Test
run: |
locust -f server/test/locust/locustfile.py --headless -u 30 -r 10 --host https://api.cellxgene.dev.single-cell.czi.technology/cellxgene/e/ --run-time 5m 2>&1 | tee locust_dev_stats.txt
- name: Slack success webhook
env:
SLACK_WEBHOOK: ${{ secrets.SLACK_WEBHOOK }}
run: |
DEV_STATS=$(tail -n 61 locust_dev_stats.txt)
DEV_MSG="\`\`\`CELLXGENE EXPLORER DEV SCALE TEST RESULTS: ${DEV_STATS}\`\`\`"
curl -X POST -H 'Content-type: application/json' --data "{'text':'${DEV_MSG}'}" $SLACK_WEBHOOK
+2 -8
View File
@@ -10,12 +10,6 @@ Locust test config
# multi-dataset, for dataroot tests. these are varied in size/shape # multi-dataset, for dataroot tests. these are varied in size/shape
DataSets = [ DataSets = [
"/d/pbmc3k.cxg", "GSE60361.cxg",
"/d/TM_droplet_processed.cxg", "WongAdultRetina.cxg",
"/d/pancreas.cxg",
"/d/immune_bone_marrow_processed.cxg",
"/d/10X_mouse_13MM_processed.cxg",
"/d/GSE60361.cxg",
"/d/Reprogrammed_Dendritic_Cells.cxg",
"/d/WongAdultRetina.cxg",
] ]
+93 -92
View File
@@ -1,143 +1,144 @@
from locust import HttpLocust, TaskSet, TaskSequence, seq_task, task
from locust.wait_time import between
import random
import json import json
from gevent.pool import Group import random
import requests
from config import DataSets
from locust import HttpUser, SequentialTaskSet, task, between, TaskSet
from locust.clients import HttpSession
from requests.packages.urllib3.exceptions import InsecureRequestWarning
import server.test.unit.decode_fbs as decode_fbs import server.test.unit.decode_fbs as decode_fbs
from config import DataSets
requests.packages.urllib3.disable_warnings(InsecureRequestWarning)
""" """
Simple locust stress test defition for cellxgene Simple locust stress test definition for cellxgene
""" """
API = "/api/v0.2"
API_SUFFIX = "api/v0.2"
class ViewDataset(TaskSet): class CellXGeneTasks(TaskSet):
""" """
Simulate use against a single dataset Simulate use against a single dataset
""" """
def on_start(self): def on_start(self):
self.client.verify = False self.client.verify = False
self.dataset = random.choice(DataSets) self.dataset = random.choice(DataSets)
with self.client.get(f"{self.dataset}{API}/config", catch_response=True) as r: with self.client.get(
if r.status_code == 200: f"{self.dataset}/{API_SUFFIX}/schema", stream=True, catch_response=True
self.config = r.json()["config"] ) as schema_response:
r.success() if schema_response.status_code == 200:
else: self.schema = schema_response.json()["schema"]
self.config = None
r.failure(f"bad response code {r.status_code}")
with self.client.get(f"{self.dataset}{API}/schema", catch_response=True) as r:
if r.status_code == 200:
self.schema = r.json()["schema"]
r.success()
else: else:
self.schema = None self.schema = None
r.failure(f"bad response code {r.status_code}")
with self.client.get( with self.client.get(
f"{self.dataset}{API}/annotations/var?annotation-name={self.var_index_name()}", f"{self.dataset}/{API_SUFFIX}/config", stream=True, catch_response=True
) as config_response:
if config_response.status_code == 200:
self.config = config_response.json()["config"]
else:
self.config = None
with self.client.get(
f"{self.dataset}/{API_SUFFIX}/annotations/var?annotation-name={self.var_index_name()}",
headers={"Accept": "application/octet-stream"}, headers={"Accept": "application/octet-stream"},
catch_response=True, catch_response=True,
) as r: ) as var_index_response:
if r.status_code == 200: if var_index_response.status_code == 200:
df = decode_fbs.decode_matrix_FBS(r.content) df = decode_fbs.decode_matrix_FBS(var_index_response.content)
gene_names_idx = df["col_idx"].index(self.var_index_name()) gene_names_idx = df["col_idx"].index(self.var_index_name())
self.gene_names = df["columns"][gene_names_idx] self.gene_names = df["columns"][gene_names_idx]
else: else:
self.gene_names = None self.gene_names = []
r.failure(f"bad response code {r.status_code}")
def var_index_name(self): def var_index_name(self):
if self.schema is None: if self.schema is not None:
return None return self.schema["annotations"]["var"]["index"]
return self.schema["annotations"]["var"]["index"] return None
def obs_annotation_names(self): def obs_annotation_names(self):
if self.schema is None: if self.schema is not None:
return [col["name"] for col in self.schema["annotations"]["obs"]["columns"]]
return []
def layout_names(self):
if self.schema is not None:
return [layout["name"] for layout in self.schema["layout"]["obs"]]
else:
return [] return []
return [col["name"] for col in self.schema["annotations"]["obs"]["columns"]]
@task(2) @task(2)
class InitializeClient(TaskSequence): class InitializeClient(SequentialTaskSet):
""" """
Initial loading of cellxgene - when the user hits the main route. Initial loading of cellxgene - when the user hits the main route.
Currently this sequence skips some of the static assets, which are quite Currently this sequence skips some of the static assets, which are quite small and should be served by the
small and should be served by the HTTP server directly. HTTP server directly.
1. load index.html, etc. 1. Load index.html, etc.
2. concurrently load /config, /schema 2. Concurrently load /config, /schema
3. concurrently load /layout/obs, /annotations/var?annotation-name=<the index> 3. Concurrently load /layout/obs, /annotations/var?annotation-name=<the index>
-- does intitial render -- -- Does initial render --
4. concurrently load all /annotations/obs 4. Concurrently load all /annotations/obs and all /layouts/obs
-- fully initialized -- -- Fully initialized --
""" """
# users hit all of the init routes as fast as they can, subject to the ordering constraints # Users hit all of the init routes as fast as they can, subject to the ordering constraints and network latency.
# and network latency
wait_time = between(0.01, 0.1) wait_time = between(0.01, 0.1)
def on_start(self): def on_start(self):
self.dataset = self.parent.dataset self.dataset = self.parent.dataset
self.client.verify = False self.client.verify = False
self.api_less_client = HttpSession(
base_url=self.client.base_url.replace("api.", "").replace("cellxgene/", ""),
request_success=self.client.request_success,
request_failure=self.client.request_failure,
)
@seq_task(1) @task
def index(self): def index(self):
self.client.get(f"{self.dataset}/", stream=True).close() self.api_less_client.get(f"{self.dataset}", stream=True)
@seq_task(2) @task
def loadConfigSchema(self): def loadConfigAndSchema(self):
def config(): self.client.get(f"{self.dataset}/{API_SUFFIX}/schema", stream=True, catch_response=True)
self.client.get(f"{self.dataset}{API}/config", stream=True).close() self.client.get(f"{self.dataset}/{API_SUFFIX}/config", stream=True, catch_response=True)
def schema(): @task
self.client.get(f"{self.dataset}{API}/schema", stream=True).close()
group = Group()
group.spawn(config)
group.spawn(schema)
group.join()
@seq_task(3)
def loadBootstrapData(self): def loadBootstrapData(self):
def layout(): self.client.get(
self.client.get( f"{self.dataset}/{API_SUFFIX}/layout/obs", headers={"Accept": "application/octet-stream"}, stream=True
f"{self.dataset}{API}/layout/obs", headers={"Accept": "application/octet-stream"}, stream=True )
).close() self.client.get(
f"{self.dataset}/{API_SUFFIX}/annotations/var?annotation-name={self.parent.var_index_name()}",
def varAnnotationIndex(): headers={"Accept": "application/octet-stream"},
self.client.get( catch_response=True,
f"{self.dataset}{API}/annotations/var?annotation-name={self.parent.var_index_name()}", )
headers={"Accept": "application/octet-stream"},
stream=True,
).close()
group = Group()
group.spawn(layout)
group.spawn(varAnnotationIndex)
group.join()
@seq_task(4)
def loadObsAnnotations(self):
def obs_annotation(name):
self.client.get(
f"{self.dataset}{API}/annotations/obs?annotation-name={name}",
headers={"Accept": "application/octet-stream"},
stream=True,
).close()
@task
def loadObsAnnotationsAndLayouts(self):
obs_names = self.parent.obs_annotation_names() obs_names = self.parent.obs_annotation_names()
group = Group()
for name in obs_names: for name in obs_names:
group.spawn(obs_annotation, name) self.client.get(
group.join() f"{self.dataset}/{API_SUFFIX}/annotations/obs?annotation-name={name}",
headers={"Accept": "application/octet-stream"},
stream=True,
)
@seq_task(5) layouts = self.parent.layolut_names()
for name in layouts:
self.client.get(
f"{self.dataset}/{API_SUFFIX}/annotations/obs?layout-name={name}",
headers={"Accept": "application/octet-stream"},
stream=True,
)
@task
def done(self): def done(self):
self.interrupt() self.interrupt()
@@ -146,19 +147,19 @@ class ViewDataset(TaskSet):
""" """
Simulate user occasionally loading some expression data for a gene Simulate user occasionally loading some expression data for a gene
""" """
gene_name = random.choice(self.gene_names) gene_name = random.choice(self.gene_names)
filter = {"filter": {"var": {"annotation_value": [{"name": self.var_index_name(), "values": [gene_name]}]}}} filter = {"filter": {"var": {"annotation_value": [{"name": self.var_index_name(), "values": [gene_name]}]}}}
self.client.put( self.client.put(
f"{self.dataset}{API}/data/var", f"{self.dataset}/{API_SUFFIX}/data/var",
data=json.dumps(filter), data=json.dumps(filter),
headers={"Content-Type": "application/json", "Accept": "application/octet-stream"}, headers={"Content-Type": "application/json", "Accept": "application/octet-stream"},
stream=True, stream=True,
).close() ).close()
class CellxgeneUser(HttpLocust): class CellxgeneUser(HttpUser):
task_set = ViewDataset tasks = [CellXGeneTasks]
# most ops do not require back-end interaction, so slow cadence # Most ops do not require back-end interaction, so slow cadence for users
# for users
wait_time = between(10, 60) wait_time = between(10, 60)
@@ -27,7 +27,6 @@ class WebsiteUser(HttpUser):
dataset_urls = [ dataset_urls = [
"human_cell_landscape.cxg", "human_cell_landscape.cxg",
"Single_cell_drug_screening_a549-42-remixed.cxg", "Single_cell_drug_screening_a549-42-remixed.cxg",
"kampmann_lab_human_AD_snRNAseq_EC_inhibitoryNeurons-53-remixed.cxg",
"krasnow_lab_human_lung_cell_atlas_smartseq2-2-remixed.cxg", "krasnow_lab_human_lung_cell_atlas_smartseq2-2-remixed.cxg",
"Single_cell_gene_expression_profiling_of_SARS_CoV_2_infected_human_cell_lines_H1299-27-remixed.cxg", "Single_cell_gene_expression_profiling_of_SARS_CoV_2_infected_human_cell_lines_H1299-27-remixed.cxg",
] ]