Add server plugin system (#1447)

* Add server plugin system

Plugins are optional modules loaded at runtime. Specification:
* Plugins are loaded from the server.plugins module (directory
  server/plugins)
* The import_plugins method is run as part of the initialization of the
  server module in __init__.py

* Add plugins to the EB build process

* Remove bit of dead code

* Respond to feedback from @bmccandless
This commit is contained in:
Matt Weiden
2020-05-05 17:05:42 -07:00
committed by GitHub
parent 67d7b8160f
commit 5947306ca0
8 changed files with 245 additions and 168 deletions
+5
View File
@@ -1 +1,6 @@
from server.common.utils import import_plugins
__version__ = "0.15.0" __version__ = "0.15.0"
import_plugins("server.plugins")
+28 -3
View File
@@ -1,12 +1,16 @@
import contextlib import contextlib
import errno import errno
import socket import importlib.util
from urllib.parse import urlsplit, urljoin import logging
import os import os
import pkgutil
import socket
import warnings
from flask import json from flask import json
from urllib.parse import urlsplit, urljoin
import numpy as np import numpy as np
import pandas as pd import pandas as pd
import warnings
def find_available_port(host, port=5005): def find_available_port(host, port=5005):
@@ -142,3 +146,24 @@ def series_to_schema(array):
else: else:
raise TypeError(f"Annotations of type {dtype} are unsupported.") raise TypeError(f"Annotations of type {dtype} are unsupported.")
return schema return schema
def import_plugins(plugin_module):
"""
Load optional plugin modules from server.common.plugins
If you would like to customize cellxgene, you can add submodules to server.common.plugins before running the app.
This code will import each, loading the code in each. If no plugins are defined, initializing the app continues as
normal.
"""
loaded_modules = []
try:
pkg = importlib.import_module(plugin_module)
for loader, name, is_pkg in pkgutil.walk_packages(pkg.__path__):
full_name = f"{plugin_module}.{name}"
module = importlib.import_module(full_name)
logging.info(f"Imported plugin {full_name}")
loaded_modules.append(module)
except ModuleNotFoundError:
logging.debug(f"No plugins found in module: {plugin_module}")
return loaded_modules
+3
View File
@@ -41,6 +41,9 @@ build: clean
cp -r customize/ebextensions/* artifact.dir/.ebextensions; \ cp -r customize/ebextensions/* artifact.dir/.ebextensions; \
fi; \ fi; \
fi; \ fi; \
if [ -d customize/plugins ] ; then \
cp -r customize/plugins artifact.dir/server/; \
fi; \
(cd artifact.dir; \ (cd artifact.dir; \
cp -r server/common/web/static static; \ cp -r server/common/web/static static; \
zip -r ../artifact.zip . --exclude server/test/\* server/eb/\* ; ); \ zip -r ../artifact.zip . --exclude server/test/\* server/eb/\* ; ); \
+125 -118
View File
@@ -27,115 +27,124 @@ https://docs.aws.amazon.com/elasticbeanstalk/latest/dg/eb-cli3-install.html
These steps are meant to serve as an example. These steps are meant to serve as an example.
There are many more options to these commands that may be important or necessary for your environment. There are many more options to these commands that may be important or necessary for your environment.
1. Make your matrix files available to the EB servers. ### 1. Make your matrix files available to the EB servers.
The following choices are known to work. The following choices are known to work.
* S3 Bucket. * S3 Bucket.
* POSIX filesystem (such as Lustre) * POSIX filesystem (such as Lustre)
* Lustre filesystem backed by S3 * Lustre filesystem backed by S3
S3 is convenient and the relatively inexpensive option. S3 is convenient and the relatively inexpensive option.
Lustre is higher performance, but more expensive, and slightly more complex to setup and manage. Lustre is higher performance, but more expensive, and slightly more complex to setup and manage.
AWS supports a feature to back the Lustre filesystem with S3, which give an easy to manage and high AWS supports a feature to back the Lustre filesystem with S3, which give an easy to manage and high
performance option. performance option.
Once the storage is in place, the next step is to copy your matrix files to that location. Once the storage is in place, the next step is to copy your matrix files to that location.
Currently cellxgene supports a flat file organization. Each matrix file is located from Currently cellxgene supports a flat file organization. Each matrix file is located from
the same s3 prefix or filesystem directory. This location is specified in the configuration as the dataroot. the same s3 prefix or filesystem directory. This location is specified in the configuration as the dataroot.
2. Create an elastic beanstalk application. For example: ### 2. Create an elastic beanstalk application. For example:
``` ```
EB_APP=cellxgene-app EB_APP=cellxgene-app
eb init -p python-3.6 $EB_APP eb init -p python-3.6 $EB_APP
``` ```
3. Configuring cellxgene ### 3. Configuring cellxgene
All the cellxgene configuration options can be set from a configuration file. All the cellxgene configuration options can be set from a configuration file.
This file can be generated like this: This file can be generated like this:
```cellxgene launch --dump-default-config > myconfig.yaml``` ```cellxgene launch --dump-default-config > myconfig.yaml```
The config file may then be customized before the app is deployed. The config file may then be customized before the app is deployed.
There are two ways to set the config file location, evaluated in this order: There are two ways to set the config file location, evaluated in this order:
First, if your config file is named "config.yaml" and exists in `customize/config.yaml`, First, if your config file is named "config.yaml" and exists in `customize/config.yaml`,
then it will be bundled with the application zip file and installed along then it will be bundled with the application zip file and installed along
side the app on the EB servers. side the app on the EB servers.
Second, a potentially more flexible approach is to place your config file in a location accessible to the EB Second, a potentially more flexible approach is to place your config file in a location accessible to the EB
servers, such as in S3. For example: s3://my-bucket/my-datasets/config.yaml. servers, such as in S3. For example: s3://my-bucket/my-datasets/config.yaml.
Set the CXG_CONFIG_FILE environment variable to specify this location. Set the CXG_CONFIG_FILE environment variable to specify this location.
Another option is to set the CXG_DATAROOT environment variable. The dataroot Another option is to set the CXG_DATAROOT environment variable. The dataroot
is the location where the matrix files are located. is the location where the matrix files are located.
This environment variable will override the dataroot in the config file (if specified). This environment variable will override the dataroot in the config file (if specified).
- Note: Certain features, such as user annotations, are automatically disabled by the EB app, - Note: Certain features, such as user annotations, are automatically disabled by the EB app,
and cannot be enabled using configuration. They may be enabled manually by modifying app.py, however and cannot be enabled using configuration. They may be enabled manually by modifying app.py, however
this is not supported or recommended at this time. this is not supported or recommended at this time.
4. Customization ### 4. Customization
The deployment can be customized in several ways, by adding files to a directory called The deployment can be customized in several ways, by adding files to a directory called
`customize` which is placed in this directory. `customize` which is placed in this directory.
config file: #### config file
This was described in the previous section. This was described in the previous section.
static files: #### static files
The cellxgene server can serve additional static webpages that will be associated with the app. The cellxgene server can serve additional static webpages that will be associated with the app.
These include the about_legal_tos (terms of service), and about_legal_privacy, for example. These include the about_legal_tos (terms of service), and about_legal_privacy, for example.
To use this feature, do the following: To use this feature, do the following:
* In this directory, create a sub directory called "customize/deploy/". * In this directory, create a sub directory called "customize/deploy/".
* Copy the files you want to serve into this directory * Copy the files you want to serve into this directory
* modify your configuration file to set the location to these file: /static/deploy/<filename> * modify your configuration file to set the location to these file: /static/deploy/<filename>
Example: you want to include an "about_legal_tos" and "about_legal_privacy" page to cellxgene. Example: you want to include an "about_legal_tos" and "about_legal_privacy" page to cellxgene.
Assume files called "tos.html" and "privacy.html" exist. Assume files called "tos.html" and "privacy.html" exist.
``` ```
$ mkdir static $ mkdir static
$ cp <source_dir>/tos.html customize/deploy/tos.html $ cp <source_dir>/tos.html customize/deploy/tos.html
$ cp <source_dir>/privacy.html customize/deploy/privacy.html $ cp <source_dir>/privacy.html customize/deploy/privacy.html
# edit config.yaml # edit config.yaml
$ grep "/static/deploy" config.yaml $ grep "/static/deploy" config.yaml
about_legal_tos: /static/deploy/tos.html about_legal_tos: /static/deploy/tos.html
about_legal_privacy: /static/deploy/privacy.html about_legal_privacy: /static/deploy/privacy.html
``` ```
Inline javascript scripts: #### Inline javascript scripts
Additional scripts can be added using the server/inline_scripts config parameters. Additional scripts can be added using the server/inline_scripts config parameters.
To include these scripts in the deployment, use the following steps: To include these scripts in the deployment, use the following steps:
* In this directory, create a sub directory called "customize/inline_scripts". * In this directory, create a sub directory called "customize/inline_scripts".
* Copy the script files into this directory * Copy the script files into this directory
* modify your configuration file to set the location to these file (leaving off customize/inline_scripts) * modify your configuration file to set the location to these file (leaving off customize/inline_scripts)
For example, to add an inline script called "myscript.js": For example, to add an inline script called "myscript.js":
``` ```
$ mkdir scripts $ mkdir scripts
$ cp <source_dir>/myscript.js customize/inline_scripts/myscript.js $ cp <source_dir>/myscript.js customize/inline_scripts/myscript.js
# edit the config.yaml # edit the config.yaml
$ grep inline_scripts config.yaml $ grep inline_scripts config.yaml
inline_scripts : [ myscript.js ] inline_scripts : [ myscript.js ]
``` ```
ebextensions: #### Plugins
Optionally, you can add plugins to the server python code. To include a plugin in the deployment use the following steps:
```
$ mkdir plugins
$ cp <source_dir>/<my_plugin>.py customize/plugins/<my_plugin>.py
```
#### ebextensions
Any additional config files intended for the `.ebextensions` directory of the artifact can be added Any additional config files intended for the `.ebextensions` directory of the artifact can be added
to the `customize/ebextensions` directory. Any file found here will be copied over. to the `customize/ebextensions` directory. Any file found here will be copied over.
requirements.txt: #### requirements.txt
A custom requirements.txt can be supplied in customize/requirements.txt. A custom requirements.txt can be supplied in customize/requirements.txt.
This file must fully specify the versions of all the python modules used by the server in the deployment. This file must fully specify the versions of all the python modules used by the server in the deployment.
@@ -144,14 +153,14 @@ Therefore the custom/requirements.txt must all have exact versions specified (e.
This file can be generated the first time using a process like this: This file can be generated the first time using a process like this:
``` ```
# assume you are running in this directory # assume you are running in this directory
$ virtualenv temp $ virtualenv temp
$ source temp/bin/activate $ source temp/bin/activate
$ pip install -r ../requirements.txt $ pip install -r ../requirements.txt
$ mkdir -p customize $ mkdir -p customize
$ pip freeze > customize/requirements.txt $ pip freeze > customize/requirements.txt
$ deactivate $ deactivate
$ rm -rf temp/ $ rm -rf temp/
``` ```
Keep the customize/requirememts.txt file, and reuse it for each deployment. Keep the customize/requirememts.txt file, and reuse it for each deployment.
@@ -159,64 +168,62 @@ If a future cellxgene version updates its requirements by modifying a module ver
or adding a new dependency, then the `make build` process will detect any or adding a new dependency, then the `make build` process will detect any
incompatibilities and raise an error. incompatibilities and raise an error.
5. Create the artifact.zip file for the application ### 5. Create the artifact.zip file for the application
``` ```
$ make build $ make build
``` ```
6. Flask secret key ### 6. Flask secret key
The application requires as secret key to be provided to flask, the web framework used by cellxgene. The application requires as secret key to be provided to flask, the web framework used by cellxgene.
There are three ways to provide the secret key: There are three ways to provide the secret key:
- In the configuration file: update the server/flask_secret_key attribute. - In the configuration file: update the server/flask_secret_key attribute.
- An environment variable: CXG_SECRET_KEY - An environment variable: `CXG_SECRET_KEY`
- Managed by the AWS Secret Manager - Managed by the AWS Secret Manager
If using the AWS Secret Manager, then the secret name is passed as an environment variable: CXG_AWS_SECRET_NAME. If using the AWS Secret Manager, then the secret name is passed as an environment variable: CXG_AWS_SECRET_NAME.
The secret must contain a key with the name "flask_secret_key". The secret must contain a key with the name "flask_secret_key".
The region name for the AWS Secret Manager must be specified (e.g. us-east-1). The region name for the AWS Secret Manager must be specified (e.g. us-east-1).
The most straightforward way is to specified it with the CXG_AWS_SECRET_REGION_NAME environment variable. The most straightforward way is to specified it with the CXG_AWS_SECRET_REGION_NAME environment variable.
If this environment variable is not defined, then the app attempts to determine the region from the If this environment variable is not defined, then the app attempts to determine the region from the
dataroot (if in s3), or the config file location (if in s3). dataroot (if in s3), or the config file location (if in s3).
### 7. Create an environment
7. Create an environment ```
# name of the environment
$ EB_ENV=cellxgene-env
``` # type of ec2 instance to run the cellxgene server (for example)
# name of the environment $ EB_INSTANCE=m5.large
$ EB_ENV=cellxgene-env
# type of ec2 instance to run the cellxgene server (for example) # One or both of the following environment variables needs to be set
$ EB_INSTANCE=m5.large $ CXG_DATAROOT=<location to your S3 bucket>
$ CXG_CONFIG_FILE=<location to your config file>
# One or both of the following environment variables needs to be set # Potentially also set envvars for the sercret key.
$ CXG_DATAROOT=<location to your S3 bucket>
$ CXG_CONFIG_FILE=<location to your config file>
# Potentially also set envvars for the sercret key. $ eb create $EB_ENV --instance-type $EB_INSTANCE \
--envvars CXG_DATAROOT=$CXG_DATAROOT,CXG_CONFIG_FILE=$CXG_CONFIG_FILE
```
$ eb create $EB_ENV --instance-type $EB_INSTANCE \ ### 8. Give the elastic beanstalk environment access to the dataroot.
--envvars CXG_DATAROOT=$CXG_DATAROOT,CXG_CONFIG_FILE=$CXG_CONFIG_FILE
```
8. Give the elastic beanstalk environment access to the dataroot. If using S3, this link may provide some useful information:
https://aws.amazon.com/premiumsupport/knowledge-center/elastic-beanstalk-s3-bucket-instance/
If using Lustre, then this link may provide a place to start:
https://aws.amazon.com/fsx/lustre/
If using S3, this link may provide some useful information: ### 9. Deploy the application
https://aws.amazon.com/premiumsupport/knowledge-center/elastic-beanstalk-s3-bucket-instance/
If using Lustre, then this link may provide a place to start: ```
https://aws.amazon.com/fsx/lustre/ $ eb deploy $EB_ENV
```
9. Deploy the application ### 10. Open the application in a browser
``` ```
$ eb deploy $EB_ENV $ eb open $EB_ENV
``` ```
10. Open the application in a browser
```
$ eb open $EB_ENV
```
+6
View File
@@ -1,4 +1,6 @@
import random
import shutil import shutil
import string
import tempfile import tempfile
from os import path, popen from os import path, popen
@@ -74,3 +76,7 @@ def app_config(data_locator, backed=False):
config.update(**args) config.update(**args)
config.complete_config() config.complete_config()
return config return config
def random_string(n):
return "".join(random.choice(string.ascii_letters) for _ in range(n))
-1
View File
@@ -7,7 +7,6 @@ from server.test.test_datasets.fixtures import pbmc3k_colors
class TestCxgAdaptor(unittest.TestCase): class TestCxgAdaptor(unittest.TestCase):
def test_get_colors(self): def test_get_colors(self):
data = self.get_data("pbmc3k.cxg") data = self.get_data("pbmc3k.cxg")
self.assertDictEqual(data.get_colors(), pbmc3k_colors) self.assertDictEqual(data.get_colors(), pbmc3k_colors)
+3 -5
View File
@@ -1,6 +1,4 @@
import random
import shutil import shutil
import string
import unittest import unittest
import anndata import anndata
@@ -8,7 +6,7 @@ import anndata
from server.common.data_locator import DataLocator from server.common.data_locator import DataLocator
from server.converters.cxgtool import write_cxg from server.converters.cxgtool import write_cxg
from server.data_cxg.cxg_adaptor import CxgAdaptor from server.data_cxg.cxg_adaptor import CxgAdaptor
from server.test import PROJECT_ROOT, app_config from server.test import PROJECT_ROOT, app_config, random_string
from server.test.test_datasets.fixtures import pbmc3k_colors from server.test.test_datasets.fixtures import pbmc3k_colors
@@ -31,8 +29,8 @@ class TestCxgAdaptor(unittest.TestCase):
self.assertEqual(data.get_colors(), {}) self.assertEqual(data.get_colors(), {})
def convert_pbmc3k(self, **kwargs): def convert_pbmc3k(self, **kwargs):
random_string = "".join(random.choice(string.ascii_letters) for _ in range(8)) rand_str = random_string(8)
data_locator = f"/tmp/test_{random_string}.cxg" data_locator = f"/tmp/test_{rand_str}.cxg"
self.fixtures.append(data_locator) self.fixtures.append(data_locator)
source_h5ad = anndata.read_h5ad(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad") source_h5ad = anndata.read_h5ad(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad")
write_cxg(adata=source_h5ad, container=data_locator, title="pbmc3k", **kwargs) write_cxg(adata=source_h5ad, container=data_locator, title="pbmc3k", **kwargs)
+34
View File
@@ -0,0 +1,34 @@
import os
import shutil
import unittest
from server.common.utils import import_plugins
from server.test import PROJECT_ROOT, random_string
class TestPlugins(unittest.TestCase):
""" Test plugin import functionality """
plugins_dir = f"{PROJECT_ROOT}/server/test/plugins"
test_plugin_path = f"{plugins_dir}/foo.py"
secret = random_string(8)
@classmethod
def setUpClass(cls) -> None:
if not os.path.isdir(cls.plugins_dir):
os.mkdir(cls.plugins_dir)
with open(cls.test_plugin_path, "w") as fh:
fh.write(f'SECRET = "{cls.secret}"\n')
@classmethod
def tearDownClass(cls) -> None:
if os.path.isdir(cls.plugins_dir):
shutil.rmtree(cls.plugins_dir)
def test_import_plugins(self):
self.assertTrue(os.path.isfile(self.test_plugin_path))
loaded_modules = import_plugins("server.test.plugins")
# test that import plugins found the file
self.assertEqual(["server.test.plugins.foo"], [ele.__name__ for ele in loaded_modules])
# test that the module was properly executed
self.assertEqual(self.secret, loaded_modules[0].SECRET)