From 38c4ecac8c6ec30a217235a74c5d48f011700e12 Mon Sep 17 00:00:00 2001 From: Bruce Martin Date: Wed, 18 Mar 2020 12:27:07 -0600 Subject: [PATCH] locust load test script (#1244) * initial version of a locustfile driver * README --- server/locust/README.md | 39 +++++++ server/locust/config.py | 18 ++++ server/locust/locustfile.py | 141 ++++++++++++++++++++++++++ server/locust/requirements-locust.txt | 1 + 4 files changed, 199 insertions(+) create mode 100644 server/locust/README.md create mode 100644 server/locust/config.py create mode 100644 server/locust/locustfile.py create mode 100644 server/locust/requirements-locust.txt diff --git a/server/locust/README.md b/server/locust/README.md new file mode 100644 index 00000000..b677af6a --- /dev/null +++ b/server/locust/README.md @@ -0,0 +1,39 @@ +# Locust Load Test + +This directory contains scripts to load test cellxgene's backend. It +primary simulates initial data loading and expression data fetch, which +are the most common data routes. It currently does not include tests +for differential expression or re-clustering routes. + +## Prerequisites + +You need: + +- Python 3.6+, and pip +- cellxgene installed +- install the locust dependencies in `requirements-locust.txt` + +## To test + +1. Choose to run cellxgene in either single dataset or data root mode. +2. Edit config.py to indicate which datasets to load: + - in single dataset mode, just set `DataSets=[""]` + - in dataroot (multi-dataset) mode, add the route names, eg, `DataSets=['foo.cxg', 'bar.cxg']` +3. Launch cellxgene in the appropriate mode +4. launch locust, specifying the correct --host argument +5. point your web browser to the locust http server, usually `http://localhost:8089/` + +### Single dataset mode + +- Edit config.py and set `DataSets=[""]` +- in a shell, run `cellxgene launch somefile.h5ad` +- launch locust in another shell, `locust --host http://localhost:5005/` (or wherever you are running cellxgene) +- point a browser to the locust port, usually http://localhost:8089/ +- run test + +### Multi-dataset mode + +- Edit config.py and set `DataSets=["datapath1", ...]` +- in a shell, run `cellxgene launch --dataroot path` + +The remainder of the steps are same as single dataset. diff --git a/server/locust/config.py b/server/locust/config.py new file mode 100644 index 00000000..c62a6514 --- /dev/null +++ b/server/locust/config.py @@ -0,0 +1,18 @@ +""" +Locust test config +""" + + +""" Data routes that will be tested """ + +# single dataset, for non-dataroot tests +# DataSets = [""] + +# multi-dataset, for dataroot tests. these are varied in size/shape +DataSets = [ + "/pbmc3k.cxg", + "/TM_droplet_processed.cxg", + "/pancreas.cxg", + "/immune_bone_marrow_processed.cxg", + "/10x_mouse_13MM_processed.cxg", +] diff --git a/server/locust/locustfile.py b/server/locust/locustfile.py new file mode 100644 index 00000000..0ac38955 --- /dev/null +++ b/server/locust/locustfile.py @@ -0,0 +1,141 @@ +from locust import HttpLocust, TaskSet, TaskSequence, seq_task, task +from locust.wait_time import between +import random +import json +from gevent.pool import Group + +import server.test.decode_fbs as decode_fbs +from config import DataSets + + +""" +Simple locust stress test defition for cellxgene +""" +API = "/api/v0.2" + + +class ViewDataset(TaskSet): + """ + Simulate use against a single dataset + """ + + def on_start(self): + self.dataset = random.choice(DataSets) + + with self.client.get(f"{self.dataset}{API}/config", catch_response=True) as r: + self.config = r.json()["config"] + + with self.client.get(f"{self.dataset}{API}/schema", catch_response=True) as r: + self.schema = r.json()["schema"] + + with self.client.get( + f"{self.dataset}{API}/annotations/var?annotation-name={self.var_index_name()}", + headers={"Accept": "application/octet-stream"}, + catch_response=True + ) as r: + df = decode_fbs.decode_matrix_FBS(r.content) + gene_names_idx = df["col_idx"].index(self.var_index_name()) + self.gene_names = df["columns"][gene_names_idx] + + def var_index_name(self): + return self.schema["annotations"]["var"]["index"] + + def obs_annotation_names(self): + return [col["name"] for col in self.schema["annotations"]["obs"]["columns"]] + + @task(2) + class InitializeClient(TaskSequence): + """ + Initial loading of cellxgene - when the user hits the main route. + + Currently this sequence skips some of the static assets, which are quite + small and should be served by the HTTP server directly. + + 1. load index.html, etc. + 2. concurrently load /config, /schema + 3. concurrently load /layout/obs, /annotations/var?annotation-name= + -- does intitial render -- + 4. concurrently load all /annotations/obs + -- fully initialized -- + """ + + # users hit all of the init routes as fast as they can, subject to the ordering constraints + # and network latency + wait_time = between(0.01, 0.1) + + def on_start(self): + self.dataset = self.parent.dataset + + @seq_task(1) + def index(self): + self.client.get(f"{self.dataset}/").close() + + @seq_task(2) + def loadConfigSchema(self): + def config(): + self.client.get(f"{self.dataset}{API}/config").close() + + def schema(): + self.client.get(f"{self.dataset}{API}/schema").close() + + group = Group() + group.spawn(config) + group.spawn(schema) + group.join() + + @seq_task(3) + def loadBootstrapData(self): + def layout(): + self.client.get( + f"{self.dataset}{API}/layout/obs", headers={"Accept": "application/octet-stream"} + ).close() + + def varAnnotationIndex(): + self.client.get( + f"{self.dataset}{API}/annotations/var?annotation-name={self.parent.var_index_name()}", + headers={"Accept": "application/octet-stream"}, + ).close() + + group = Group() + group.spawn(layout) + group.spawn(varAnnotationIndex) + group.join() + + @seq_task(4) + def loadObsAnnotations(self): + def obs_annotation(name): + self.client.get( + f"{self.dataset}{API}/annotations/obs?annotation-name={name}", + headers={"Accept": "application/octet-stream"}, + ).close() + + obs_names = self.parent.obs_annotation_names() + group = Group() + for name in obs_names: + group.spawn(obs_annotation, name) + group.join() + + @seq_task(5) + def done(self): + self.interrupt() + + @task(1) + def load_expression(self): + """ + Simulate user occasionally loading some expression data for a gene + """ + gene_name = random.choice(self.gene_names) + filter = {"filter": {"var": {"annotation_value": [{"name": self.var_index_name(), "values": [gene_name]}]}}} + self.client.put( + f"{self.dataset}{API}/data/var", + data=json.dumps(filter), + headers={"Content-Type": "application/json", "Accept": "application/octet-stream"}, + ).close() + + +class CellxgeneUser(HttpLocust): + task_set = ViewDataset + + # most ops do not require back-end interaction, so slow cadence + # for users + wait_time = between(10, 60) diff --git a/server/locust/requirements-locust.txt b/server/locust/requirements-locust.txt new file mode 100644 index 00000000..282a9af7 --- /dev/null +++ b/server/locust/requirements-locust.txt @@ -0,0 +1 @@ +locustio