Add server plugin system (#1447)

* Add server plugin system

Plugins are optional modules loaded at runtime. Specification:
* Plugins are loaded from the server.plugins module (directory
  server/plugins)
* The import_plugins method is run as part of the initialization of the
  server module in __init__.py

* Add plugins to the EB build process

* Remove bit of dead code

* Respond to feedback from @bmccandless
This commit is contained in:
Matt Weiden
2020-05-05 17:05:42 -07:00
committed by GitHub
parent 67d7b8160f
commit 5947306ca0
8 changed files with 245 additions and 168 deletions
+5
View File
@@ -1 +1,6 @@
from server.common.utils import import_plugins
__version__ = "0.15.0"
import_plugins("server.plugins")
+28 -3
View File
@@ -1,12 +1,16 @@
import contextlib
import errno
import socket
from urllib.parse import urlsplit, urljoin
import importlib.util
import logging
import os
import pkgutil
import socket
import warnings
from flask import json
from urllib.parse import urlsplit, urljoin
import numpy as np
import pandas as pd
import warnings
def find_available_port(host, port=5005):
@@ -142,3 +146,24 @@ def series_to_schema(array):
else:
raise TypeError(f"Annotations of type {dtype} are unsupported.")
return schema
def import_plugins(plugin_module):
"""
Load optional plugin modules from server.common.plugins
If you would like to customize cellxgene, you can add submodules to server.common.plugins before running the app.
This code will import each, loading the code in each. If no plugins are defined, initializing the app continues as
normal.
"""
loaded_modules = []
try:
pkg = importlib.import_module(plugin_module)
for loader, name, is_pkg in pkgutil.walk_packages(pkg.__path__):
full_name = f"{plugin_module}.{name}"
module = importlib.import_module(full_name)
logging.info(f"Imported plugin {full_name}")
loaded_modules.append(module)
except ModuleNotFoundError:
logging.debug(f"No plugins found in module: {plugin_module}")
return loaded_modules
+3
View File
@@ -41,6 +41,9 @@ build: clean
cp -r customize/ebextensions/* artifact.dir/.ebextensions; \
fi; \
fi; \
if [ -d customize/plugins ] ; then \
cp -r customize/plugins artifact.dir/server/; \
fi; \
(cd artifact.dir; \
cp -r server/common/web/static static; \
zip -r ../artifact.zip . --exclude server/test/\* server/eb/\* ; ); \
+125 -118
View File
@@ -27,115 +27,124 @@ https://docs.aws.amazon.com/elasticbeanstalk/latest/dg/eb-cli3-install.html
These steps are meant to serve as an example.
There are many more options to these commands that may be important or necessary for your environment.
1. Make your matrix files available to the EB servers.
### 1. Make your matrix files available to the EB servers.
The following choices are known to work.
The following choices are known to work.
* S3 Bucket.
* POSIX filesystem (such as Lustre)
* Lustre filesystem backed by S3
* S3 Bucket.
* POSIX filesystem (such as Lustre)
* Lustre filesystem backed by S3
S3 is convenient and the relatively inexpensive option.
Lustre is higher performance, but more expensive, and slightly more complex to setup and manage.
AWS supports a feature to back the Lustre filesystem with S3, which give an easy to manage and high
performance option.
S3 is convenient and the relatively inexpensive option.
Lustre is higher performance, but more expensive, and slightly more complex to setup and manage.
AWS supports a feature to back the Lustre filesystem with S3, which give an easy to manage and high
performance option.
Once the storage is in place, the next step is to copy your matrix files to that location.
Currently cellxgene supports a flat file organization. Each matrix file is located from
the same s3 prefix or filesystem directory. This location is specified in the configuration as the dataroot.
Once the storage is in place, the next step is to copy your matrix files to that location.
Currently cellxgene supports a flat file organization. Each matrix file is located from
the same s3 prefix or filesystem directory. This location is specified in the configuration as the dataroot.
2. Create an elastic beanstalk application. For example:
### 2. Create an elastic beanstalk application. For example:
```
EB_APP=cellxgene-app
eb init -p python-3.6 $EB_APP
```
```
EB_APP=cellxgene-app
eb init -p python-3.6 $EB_APP
```
3. Configuring cellxgene
### 3. Configuring cellxgene
All the cellxgene configuration options can be set from a configuration file.
This file can be generated like this:
All the cellxgene configuration options can be set from a configuration file.
This file can be generated like this:
```cellxgene launch --dump-default-config > myconfig.yaml```
```cellxgene launch --dump-default-config > myconfig.yaml```
The config file may then be customized before the app is deployed.
The config file may then be customized before the app is deployed.
There are two ways to set the config file location, evaluated in this order:
There are two ways to set the config file location, evaluated in this order:
First, if your config file is named "config.yaml" and exists in `customize/config.yaml`,
then it will be bundled with the application zip file and installed along
side the app on the EB servers.
First, if your config file is named "config.yaml" and exists in `customize/config.yaml`,
then it will be bundled with the application zip file and installed along
side the app on the EB servers.
Second, a potentially more flexible approach is to place your config file in a location accessible to the EB
servers, such as in S3. For example: s3://my-bucket/my-datasets/config.yaml.
Set the CXG_CONFIG_FILE environment variable to specify this location.
Second, a potentially more flexible approach is to place your config file in a location accessible to the EB
servers, such as in S3. For example: s3://my-bucket/my-datasets/config.yaml.
Set the CXG_CONFIG_FILE environment variable to specify this location.
Another option is to set the CXG_DATAROOT environment variable. The dataroot
is the location where the matrix files are located.
This environment variable will override the dataroot in the config file (if specified).
Another option is to set the CXG_DATAROOT environment variable. The dataroot
is the location where the matrix files are located.
This environment variable will override the dataroot in the config file (if specified).
- Note: Certain features, such as user annotations, are automatically disabled by the EB app,
and cannot be enabled using configuration. They may be enabled manually by modifying app.py, however
this is not supported or recommended at this time.
- Note: Certain features, such as user annotations, are automatically disabled by the EB app,
and cannot be enabled using configuration. They may be enabled manually by modifying app.py, however
this is not supported or recommended at this time.
4. Customization
### 4. Customization
The deployment can be customized in several ways, by adding files to a directory called
`customize` which is placed in this directory.
config file:
#### config file
This was described in the previous section.
static files:
#### static files
The cellxgene server can serve additional static webpages that will be associated with the app.
These include the about_legal_tos (terms of service), and about_legal_privacy, for example.
To use this feature, do the following:
* In this directory, create a sub directory called "customize/deploy/".
* Copy the files you want to serve into this directory
* modify your configuration file to set the location to these file: /static/deploy/<filename>
* In this directory, create a sub directory called "customize/deploy/".
* Copy the files you want to serve into this directory
* modify your configuration file to set the location to these file: /static/deploy/<filename>
Example: you want to include an "about_legal_tos" and "about_legal_privacy" page to cellxgene.
Assume files called "tos.html" and "privacy.html" exist.
Example: you want to include an "about_legal_tos" and "about_legal_privacy" page to cellxgene.
Assume files called "tos.html" and "privacy.html" exist.
```
$ mkdir static
$ cp <source_dir>/tos.html customize/deploy/tos.html
$ cp <source_dir>/privacy.html customize/deploy/privacy.html
```
$ mkdir static
$ cp <source_dir>/tos.html customize/deploy/tos.html
$ cp <source_dir>/privacy.html customize/deploy/privacy.html
# edit config.yaml
$ grep "/static/deploy" config.yaml
about_legal_tos: /static/deploy/tos.html
about_legal_privacy: /static/deploy/privacy.html
```
# edit config.yaml
$ grep "/static/deploy" config.yaml
about_legal_tos: /static/deploy/tos.html
about_legal_privacy: /static/deploy/privacy.html
```
Inline javascript scripts:
#### Inline javascript scripts
Additional scripts can be added using the server/inline_scripts config parameters.
To include these scripts in the deployment, use the following steps:
* In this directory, create a sub directory called "customize/inline_scripts".
* Copy the script files into this directory
* modify your configuration file to set the location to these file (leaving off customize/inline_scripts)
* In this directory, create a sub directory called "customize/inline_scripts".
* Copy the script files into this directory
* modify your configuration file to set the location to these file (leaving off customize/inline_scripts)
For example, to add an inline script called "myscript.js":
For example, to add an inline script called "myscript.js":
```
$ mkdir scripts
$ cp <source_dir>/myscript.js customize/inline_scripts/myscript.js
# edit the config.yaml
$ grep inline_scripts config.yaml
inline_scripts : [ myscript.js ]
```
```
$ mkdir scripts
$ cp <source_dir>/myscript.js customize/inline_scripts/myscript.js
# edit the config.yaml
$ grep inline_scripts config.yaml
inline_scripts : [ myscript.js ]
```
ebextensions:
#### Plugins
Optionally, you can add plugins to the server python code. To include a plugin in the deployment use the following steps:
```
$ mkdir plugins
$ cp <source_dir>/<my_plugin>.py customize/plugins/<my_plugin>.py
```
#### ebextensions
Any additional config files intended for the `.ebextensions` directory of the artifact can be added
to the `customize/ebextensions` directory. Any file found here will be copied over.
requirements.txt:
#### requirements.txt
A custom requirements.txt can be supplied in customize/requirements.txt.
This file must fully specify the versions of all the python modules used by the server in the deployment.
@@ -144,14 +153,14 @@ Therefore the custom/requirements.txt must all have exact versions specified (e.
This file can be generated the first time using a process like this:
```
# assume you are running in this directory
$ virtualenv temp
$ source temp/bin/activate
$ pip install -r ../requirements.txt
$ mkdir -p customize
$ pip freeze > customize/requirements.txt
$ deactivate
$ rm -rf temp/
# assume you are running in this directory
$ virtualenv temp
$ source temp/bin/activate
$ pip install -r ../requirements.txt
$ mkdir -p customize
$ pip freeze > customize/requirements.txt
$ deactivate
$ rm -rf temp/
```
Keep the customize/requirememts.txt file, and reuse it for each deployment.
@@ -159,64 +168,62 @@ If a future cellxgene version updates its requirements by modifying a module ver
or adding a new dependency, then the `make build` process will detect any
incompatibilities and raise an error.
5. Create the artifact.zip file for the application
### 5. Create the artifact.zip file for the application
```
$ make build
```
```
$ make build
```
6. Flask secret key
### 6. Flask secret key
The application requires as secret key to be provided to flask, the web framework used by cellxgene.
There are three ways to provide the secret key:
The application requires as secret key to be provided to flask, the web framework used by cellxgene.
There are three ways to provide the secret key:
- In the configuration file: update the server/flask_secret_key attribute.
- An environment variable: CXG_SECRET_KEY
- Managed by the AWS Secret Manager
- In the configuration file: update the server/flask_secret_key attribute.
- An environment variable: `CXG_SECRET_KEY`
- Managed by the AWS Secret Manager
If using the AWS Secret Manager, then the secret name is passed as an environment variable: CXG_AWS_SECRET_NAME.
The secret must contain a key with the name "flask_secret_key".
The region name for the AWS Secret Manager must be specified (e.g. us-east-1).
The most straightforward way is to specified it with the CXG_AWS_SECRET_REGION_NAME environment variable.
If this environment variable is not defined, then the app attempts to determine the region from the
dataroot (if in s3), or the config file location (if in s3).
If using the AWS Secret Manager, then the secret name is passed as an environment variable: CXG_AWS_SECRET_NAME.
The secret must contain a key with the name "flask_secret_key".
The region name for the AWS Secret Manager must be specified (e.g. us-east-1).
The most straightforward way is to specified it with the CXG_AWS_SECRET_REGION_NAME environment variable.
If this environment variable is not defined, then the app attempts to determine the region from the
dataroot (if in s3), or the config file location (if in s3).
### 7. Create an environment
7. Create an environment
```
# name of the environment
$ EB_ENV=cellxgene-env
```
# name of the environment
$ EB_ENV=cellxgene-env
# type of ec2 instance to run the cellxgene server (for example)
$ EB_INSTANCE=m5.large
# type of ec2 instance to run the cellxgene server (for example)
$ EB_INSTANCE=m5.large
# One or both of the following environment variables needs to be set
$ CXG_DATAROOT=<location to your S3 bucket>
$ CXG_CONFIG_FILE=<location to your config file>
# One or both of the following environment variables needs to be set
$ CXG_DATAROOT=<location to your S3 bucket>
$ CXG_CONFIG_FILE=<location to your config file>
# Potentially also set envvars for the sercret key.
# Potentially also set envvars for the sercret key.
$ eb create $EB_ENV --instance-type $EB_INSTANCE \
--envvars CXG_DATAROOT=$CXG_DATAROOT,CXG_CONFIG_FILE=$CXG_CONFIG_FILE
```
$ eb create $EB_ENV --instance-type $EB_INSTANCE \
--envvars CXG_DATAROOT=$CXG_DATAROOT,CXG_CONFIG_FILE=$CXG_CONFIG_FILE
```
### 8. Give the elastic beanstalk environment access to the dataroot.
8. Give the elastic beanstalk environment access to the dataroot.
If using S3, this link may provide some useful information:
https://aws.amazon.com/premiumsupport/knowledge-center/elastic-beanstalk-s3-bucket-instance/
If using Lustre, then this link may provide a place to start:
https://aws.amazon.com/fsx/lustre/
If using S3, this link may provide some useful information:
https://aws.amazon.com/premiumsupport/knowledge-center/elastic-beanstalk-s3-bucket-instance/
### 9. Deploy the application
If using Lustre, then this link may provide a place to start:
https://aws.amazon.com/fsx/lustre/
```
$ eb deploy $EB_ENV
```
9. Deploy the application
### 10. Open the application in a browser
```
$ eb deploy $EB_ENV
```
10. Open the application in a browser
```
$ eb open $EB_ENV
```
```
$ eb open $EB_ENV
```
+6
View File
@@ -1,4 +1,6 @@
import random
import shutil
import string
import tempfile
from os import path, popen
@@ -74,3 +76,7 @@ def app_config(data_locator, backed=False):
config.update(**args)
config.complete_config()
return config
def random_string(n):
return "".join(random.choice(string.ascii_letters) for _ in range(n))
-1
View File
@@ -7,7 +7,6 @@ from server.test.test_datasets.fixtures import pbmc3k_colors
class TestCxgAdaptor(unittest.TestCase):
def test_get_colors(self):
data = self.get_data("pbmc3k.cxg")
self.assertDictEqual(data.get_colors(), pbmc3k_colors)
+3 -5
View File
@@ -1,6 +1,4 @@
import random
import shutil
import string
import unittest
import anndata
@@ -8,7 +6,7 @@ import anndata
from server.common.data_locator import DataLocator
from server.converters.cxgtool import write_cxg
from server.data_cxg.cxg_adaptor import CxgAdaptor
from server.test import PROJECT_ROOT, app_config
from server.test import PROJECT_ROOT, app_config, random_string
from server.test.test_datasets.fixtures import pbmc3k_colors
@@ -31,8 +29,8 @@ class TestCxgAdaptor(unittest.TestCase):
self.assertEqual(data.get_colors(), {})
def convert_pbmc3k(self, **kwargs):
random_string = "".join(random.choice(string.ascii_letters) for _ in range(8))
data_locator = f"/tmp/test_{random_string}.cxg"
rand_str = random_string(8)
data_locator = f"/tmp/test_{rand_str}.cxg"
self.fixtures.append(data_locator)
source_h5ad = anndata.read_h5ad(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad")
write_cxg(adata=source_h5ad, container=data_locator, title="pbmc3k", **kwargs)
+34
View File
@@ -0,0 +1,34 @@
import os
import shutil
import unittest
from server.common.utils import import_plugins
from server.test import PROJECT_ROOT, random_string
class TestPlugins(unittest.TestCase):
""" Test plugin import functionality """
plugins_dir = f"{PROJECT_ROOT}/server/test/plugins"
test_plugin_path = f"{plugins_dir}/foo.py"
secret = random_string(8)
@classmethod
def setUpClass(cls) -> None:
if not os.path.isdir(cls.plugins_dir):
os.mkdir(cls.plugins_dir)
with open(cls.test_plugin_path, "w") as fh:
fh.write(f'SECRET = "{cls.secret}"\n')
@classmethod
def tearDownClass(cls) -> None:
if os.path.isdir(cls.plugins_dir):
shutil.rmtree(cls.plugins_dir)
def test_import_plugins(self):
self.assertTrue(os.path.isfile(self.test_plugin_path))
loaded_modules = import_plugins("server.test.plugins")
# test that import plugins found the file
self.assertEqual(["server.test.plugins.foo"], [ele.__name__ for ele in loaded_modules])
# test that the module was properly executed
self.assertEqual(self.secret, loaded_modules[0].SECRET)