mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-05 01:28:13 +08:00
Refactor czi_hosted and server into backend directory, pull common code into backend/common, refactor tests (#2102)
* move local_server -> backend/server server-> backend/czi_hosted, pull common code into backend/common update imports, tests and make commands
This commit is contained in:
@@ -0,0 +1,30 @@
|
||||
files:
|
||||
"/etc/httpd/conf.d/enable_mod_deflate.conf":
|
||||
mode: "000644"
|
||||
owner: root
|
||||
group: root
|
||||
content: |
|
||||
<IfModule mod_deflate.c>
|
||||
|
||||
AddOutputFilterByType DEFLATE text/plain
|
||||
AddOutputFilterByType DEFLATE text/html
|
||||
AddOutputFilterByType DEFLATE application/xhtml+xml
|
||||
AddOutputFilterByType DEFLATE text/xml
|
||||
AddOutputFilterByType DEFLATE application/xml
|
||||
AddOutputFilterByType DEFLATE application/xml+rss
|
||||
AddOutputFilterByType DEFLATE application/x-javascript
|
||||
AddOutputFilterByType DEFLATE text/javascript
|
||||
AddOutputFilterByType DEFLATE text/css
|
||||
AddOutputFilterByType DEFLATE application/octet-stream
|
||||
|
||||
DeflateCompressionLevel 9
|
||||
|
||||
BrowserMatch ^Mozilla/4 gzip-only-text/html
|
||||
BrowserMatch ^Mozilla/4\.0[678] no-gzip
|
||||
BrowserMatch \bMSI[E] !no-gzip !gzip-only-text/html
|
||||
|
||||
<IfModule mod_headers.c>
|
||||
Header append Vary User-Agent env=!dont-vary
|
||||
</IfModule>
|
||||
|
||||
</IfModule>
|
||||
@@ -0,0 +1,10 @@
|
||||
# Configure WSGI so that it will work with numpy, scanpy, etc, which all use the
|
||||
# Python SWIG, and therefore will deadlock on start. For more information, see
|
||||
# https://modwsgi.readthedocs.io/en/develop/user-guides/application-issues.html#python-simplified-gil-state-api
|
||||
files:
|
||||
"/etc/httpd/conf.d/wsgi_custom.conf":
|
||||
mode: "000644"
|
||||
owner: root
|
||||
group: root
|
||||
content: |
|
||||
WSGIApplicationGroup %{GLOBAL}
|
||||
@@ -0,0 +1,5 @@
|
||||
|
||||
# Elastic Beanstalk Files
|
||||
.elasticbeanstalk/*
|
||||
!.elasticbeanstalk/*.cfg.yml
|
||||
!.elasticbeanstalk/*.global.yml
|
||||
@@ -0,0 +1,59 @@
|
||||
include ../../../common.mk
|
||||
|
||||
.PHONY: clean
|
||||
clean:
|
||||
rm -f artifact.zip
|
||||
rm -rf artifact.dir
|
||||
|
||||
|
||||
# Build the ElasticBeanstalk configuration and deployment bundle,
|
||||
# such that deployment can be done with a simple `eb deploy`.
|
||||
# Presumes that a top-level `make build-client` has been done to
|
||||
# create the client static assets.
|
||||
|
||||
cwd := $(shell pwd)
|
||||
|
||||
.PHONY: build
|
||||
build: clean
|
||||
mkdir artifact.dir; \
|
||||
(cd ../../.. ; \
|
||||
git ls-files backend/czi_hosted/ | cpio -pdm $(cwd)/artifact.dir ; ); \
|
||||
$(call copy_client_assets,../../../client/build,artifact.dir/backend/czi_hosted) ; \
|
||||
set -e ; \
|
||||
cp app.py artifact.dir/application.py; \
|
||||
cp -r ../../../backend/common artifact.dir/backend/common; \
|
||||
cp ../requirements.txt artifact.dir; \
|
||||
cp -r .ebextensions artifact.dir; \
|
||||
if [ -d customize ] ; then \
|
||||
if [ -f customize/config.yaml ] ; then \
|
||||
cp customize/config.yaml artifact.dir; \
|
||||
fi ; \
|
||||
if [ -f customize/Dockerfile ] ; then \
|
||||
cp customize/Dockerfile artifact.dir; \
|
||||
fi ; \
|
||||
if [ -f customize/requirements.txt ] ; then \
|
||||
pip install requirements-parser ; \
|
||||
pip install packaging ; \
|
||||
python3 check_requirements.py ../requirements.txt customize/requirements.txt; \
|
||||
cp customize/requirements.txt artifact.dir; \
|
||||
fi ; \
|
||||
if [ -d customize/deploy ] ; then \
|
||||
mkdir -p artifact.dir/backend/czi_hosted/common/web/static/cellxgene; \
|
||||
cp -r customize/deploy artifact.dir/backend/czi_hosted/common/web/static/cellxgene; \
|
||||
fi; \
|
||||
if [ -d customize/inline_scripts ] ; then \
|
||||
cp -r customize/inline_scripts/* artifact.dir/backend/czi_hosted/common/web/templates; \
|
||||
fi; \
|
||||
if [ -d customize/ebextensions ] ; then \
|
||||
cp -r customize/ebextensions/* artifact.dir/.ebextensions; \
|
||||
fi; \
|
||||
fi; \
|
||||
if [ -d customize/plugins ] ; then \
|
||||
cp -r customize/plugins artifact.dir/backend/czi_hosted/; \
|
||||
fi; \
|
||||
(cd artifact.dir; \
|
||||
cp -r backend/czi_hosted/common/web/static static; \
|
||||
zip -r ../artifact.zip . --exclude backend/czi_hosted/test/\* backend/czi_hosted/eb/\* ; ); \
|
||||
if ! [ -f .elasticbeanstalk/config.yml ] ; then \
|
||||
mkdir -p .elasticbeanstalk ; cat config_deploy.yaml >> .elasticbeanstalk/config.yml ; fi
|
||||
|
||||
@@ -0,0 +1,300 @@
|
||||
# AWS Elastic Beanstalk
|
||||
|
||||
This directory contains scripts to aid in creating and deploying cellxgene on
|
||||
AWS Elastic Beanstalk.
|
||||
|
||||
This will result in a variant of cellxgene, running on AWS EC2 instances, serving data from S3.
|
||||
All datasets must be in the CXG (tiledb) format (see `cellxene convert --help`),
|
||||
and located under a single S3 prefix, which is accessible to the instance.
|
||||
In the current incarnation, no access control is available
|
||||
(outside of anything you configure yourself), so this is most appropriate for public datasets.
|
||||
|
||||
This is early development work, and will change significantly in the near future.
|
||||
We would love feedback on it, but please assume it will change.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
1. Some familiarity with AWS EB, S3, and IAM are needed.
|
||||
|
||||
2. Install the awsebcli.
|
||||
Instruction are here:
|
||||
https://docs.aws.amazon.com/elasticbeanstalk/latest/dg/eb-cli3-install.html
|
||||
|
||||
3. In the top level directory, run `make build-client` to create the client static assets.
|
||||
|
||||
## Steps
|
||||
|
||||
These steps are meant to serve as an example.
|
||||
There are many more options to these commands that may be important or necessary for your environment.
|
||||
|
||||
### 1. Make your matrix files available to the EB servers.
|
||||
|
||||
The following choices are known to work.
|
||||
|
||||
- S3 Bucket.
|
||||
- POSIX filesystem (such as Lustre)
|
||||
- Lustre filesystem backed by S3
|
||||
|
||||
S3 is convenient and the relatively inexpensive option.
|
||||
Lustre is higher performance, but more expensive, and slightly more complex to setup and manage.
|
||||
AWS supports a feature to back the Lustre filesystem with S3, which gives an easy to manage, high
|
||||
performance option.
|
||||
|
||||
Once the storage is in place, the next step is to copy your data files to that location.
|
||||
Currently cellxgene supports a flat file organization. Each matrix file is located under
|
||||
the same s3 prefix or filesystem directory. This location is specified in the configuration
|
||||
as the dataroot.
|
||||
|
||||
### 2. Create an elastic beanstalk application. For example:
|
||||
|
||||
```
|
||||
EB_APP=cellxgene-app
|
||||
eb init -p python-3.6 $EB_APP
|
||||
```
|
||||
|
||||
### 3. Configuring cellxgene
|
||||
|
||||
All the cellxgene configuration options can be set from a configuration file.
|
||||
A yaml config file containing all of the default configuration options can be generated like this:
|
||||
|
||||
`cellxgene launch --dump-default-config > myconfig.yaml`
|
||||
|
||||
The config file may then be customized before the app is deployed.
|
||||
|
||||
There are two ways to set the config file location, evaluated in this order:
|
||||
|
||||
First, if your config file is named "config.yaml" and exists in `customize/config.yaml`,
|
||||
then it will be bundled with the application zip file and installed along
|
||||
side the app on the EB servers.
|
||||
|
||||
Second, a potentially more flexible approach is to place your config file in a location accessible
|
||||
to the EB servers, such as in S3. For example: s3://my-bucket/my-datasets/config.yaml.
|
||||
Set the CXG_CONFIG_FILE environment variable to specify this location.
|
||||
|
||||
Another option is to set the CXG_DATAROOT environment variable. The dataroot
|
||||
is the location where the matrix files are located.
|
||||
This environment variable will override the dataroot in the config file (if specified).
|
||||
|
||||
### 4. Customization
|
||||
|
||||
The deployment can be customized in several ways, by adding files to a directory called
|
||||
`customize` which is placed in this directory.
|
||||
|
||||
#### config file
|
||||
|
||||
This was described in the previous section.
|
||||
|
||||
#### static files
|
||||
|
||||
The cellxgene server can serve additional static webpages that will be associated with the app.
|
||||
These include the about_legal_tos (terms of service), and about_legal_privacy, for example.
|
||||
To use this feature, do the following:
|
||||
|
||||
- In this directory, create a sub directory called "customize/deploy/".
|
||||
- Copy the files you want to serve into this directory
|
||||
- modify your configuration file to set the location to these file: /static/cellxgene/deploy/<filename>
|
||||
|
||||
Example: you want to include an "about_legal_tos" and "about_legal_privacy" page to cellxgene.
|
||||
Assume files called "tos.html" and "privacy.html" exist.
|
||||
|
||||
```
|
||||
$ mkdir -p customize/deploy
|
||||
$ cp <source_dir>/tos.html customize/deploy/tos.html
|
||||
$ cp <source_dir>/privacy.html customize/deploy/privacy.html
|
||||
|
||||
# edit config.yaml
|
||||
$ grep "/static/cellxgene/deploy" config.yaml
|
||||
about_legal_tos: /static/cellxgene/deploy/tos.html
|
||||
about_legal_privacy: /static/cellxgene/deploy/privacy.html
|
||||
```
|
||||
|
||||
#### Inline javascript scripts
|
||||
|
||||
Additional scripts can be added using the server/inline_scripts config parameters.
|
||||
To include these scripts in the deployment, use the following steps:
|
||||
|
||||
- In this directory, create a sub directory called "customize/inline_scripts".
|
||||
- Copy the script files into this directory
|
||||
- Modify your configuration file to set the location to these file (leaving off customize/inline_scripts)
|
||||
|
||||
For example, to add an inline script called "myscript.js":
|
||||
|
||||
```
|
||||
$ mkdir -p customize/inline_scripts
|
||||
$ cp <source_dir>/myscript.js customize/inline_scripts/myscript.js
|
||||
# edit the config.yaml
|
||||
$ grep inline_scripts config.yaml
|
||||
inline_scripts : [ myscript.js ]
|
||||
```
|
||||
|
||||
#### Plugins
|
||||
|
||||
Optionally, you can add plugins to the server python code. To include a plugin in the deployment use the following steps:
|
||||
|
||||
```
|
||||
$ mkdir -p customize/plugins
|
||||
$ cp <source_dir>/<my_plugin>.py customize/plugins/<my_plugin>.py
|
||||
```
|
||||
|
||||
#### ebextensions
|
||||
|
||||
Any additional config files intended for the `.ebextensions` directory of the artifact can be added
|
||||
to the `customize/ebextensions` directory. Any file found here will be copied over.
|
||||
|
||||
#### requirements.txt
|
||||
|
||||
A custom requirements.txt can be supplied in customize/requirements.txt.
|
||||
This file must fully specify the versions of all the python modules used by the server in the deployment.
|
||||
This is useful to ensure that the dependencies do not change from one deployment to the next.
|
||||
Therefore the custom/requirements.txt must all have exact versions specified (e.g. anndata==0.7.1).
|
||||
|
||||
This file can be generated the first time using a process like this:
|
||||
|
||||
```
|
||||
# assume you are running in this directory
|
||||
$ virtualenv temp
|
||||
$ source temp/bin/activate
|
||||
$ pip install -r ../requirements.txt
|
||||
$ mkdir -p customize
|
||||
$ pip freeze > customize/requirements.txt
|
||||
$ deactivate
|
||||
$ rm -rf temp/
|
||||
```
|
||||
|
||||
Keep the customize/requirememts.txt file, and reuse it for each deployment.
|
||||
If a future cellxgene version updates its requirements by modifying a module version
|
||||
or adding a new dependency, then the `make build` process will detect any
|
||||
incompatibilities and raise an error.
|
||||
|
||||
#### File structure for customizations
|
||||
|
||||
The following diagram shows the file structure for the customization directory.
|
||||
|
||||
```
|
||||
customization
|
||||
+-- config.yaml
|
||||
+-- deploy/
|
||||
+-- inline_scripts/
|
||||
+-- plugins/
|
||||
+-- ebextensions/
|
||||
+-- requirements.txt
|
||||
```
|
||||
|
||||
### 5. Create the artifact.zip file for the application
|
||||
|
||||
```
|
||||
$ make build
|
||||
```
|
||||
|
||||
### 6. Flask secret key
|
||||
|
||||
The application requires a secret key to be provided to flask, the web framework used by cellxgene.
|
||||
There are three ways to provide the secret key:
|
||||
|
||||
- In the configuration file, update the server/flask_secret_key attribute.
|
||||
- In the configuration file, update the external/aws_secrets_manager section to set the
|
||||
secret name and key that defines the flask secret key.
|
||||
- An environment variable: `CXG_SECRET_KEY`
|
||||
|
||||
### 7. Create an environment
|
||||
|
||||
```
|
||||
# name of the environment
|
||||
$ EB_ENV=cellxgene-env
|
||||
|
||||
# type of ec2 instance to run the cellxgene server (for example)
|
||||
$ EB_INSTANCE=m5.large
|
||||
|
||||
# One or both of the following environment variables needs to be set
|
||||
$ CXG_DATAROOT=<location to your S3 bucket>
|
||||
$ CXG_CONFIG_FILE=<location to your config file>
|
||||
|
||||
# Potentially also set an environment variable for the flask secret key,
|
||||
# and other environemet variable described in the configuration file.
|
||||
|
||||
$ eb create $EB_ENV --instance-type $EB_INSTANCE \
|
||||
--envvars CXG_DATAROOT=$CXG_DATAROOT,CXG_CONFIG_FILE=$CXG_CONFIG_FILE
|
||||
```
|
||||
|
||||
### 8. Give the elastic beanstalk environment access to the dataroot.
|
||||
|
||||
If using S3, this link may provide some useful information:
|
||||
https://aws.amazon.com/premiumsupport/knowledge-center/elastic-beanstalk-s3-bucket-instance/
|
||||
If using Lustre, then this link may provide a place to start:
|
||||
https://aws.amazon.com/fsx/lustre/
|
||||
|
||||
### 9. Deploy the application
|
||||
|
||||
```
|
||||
$ eb deploy $EB_ENV
|
||||
```
|
||||
|
||||
### 10. Open the application in a browser
|
||||
|
||||
```
|
||||
$ eb open $EB_ENV
|
||||
```
|
||||
|
||||
## Advanced Features
|
||||
|
||||
### Authentication
|
||||
|
||||
Authentication can be configured in the configuration file. Authentication is required
|
||||
for User Annotations (see below). User Annotations is a feature where annotations can be
|
||||
created by the user
|
||||
, and
|
||||
then associated with the user's id.
|
||||
When the user revisits the site, their annotations will be available.
|
||||
|
||||
There are three main authentication modes: null, session, or oauth.
|
||||
In the configuration file specify the authentication mode by setting
|
||||
`server / authentication / type`.
|
||||
|
||||
#### null
|
||||
|
||||
Authentication is disabled: user annotations cannot be enabled.
|
||||
|
||||
#### session
|
||||
|
||||
The user is associated with their client browser session. This approach is
|
||||
simple to setup, but not recommended for hosted cellxgene, since the user will not have access to
|
||||
their annotations when running from a different browser, or if their cookies get cleared.
|
||||
|
||||
#### oauth
|
||||
|
||||
A user logs into cellxgene using an identity provider (like Google), or logs in using
|
||||
an email/password. This is the best option, but requires making use of an oauth service and
|
||||
additional configuration of the cellxgene server.
|
||||
|
||||
To see what this looks like, please look at https://cellxgene.cziscience.com/,
|
||||
and view one of the cellxgene datasets.
|
||||
For this server, Auth0 (auth0.com) is used for authentication, but there are other options.
|
||||
There are good sources of documentation online that describe how to use one of these
|
||||
services.
|
||||
|
||||
The `params_oauth` section in the configuration file describes characteristics of the
|
||||
authentication service, like "client_id" and "client_secret".
|
||||
For security, the client_secret needs to be protected. One option is to
|
||||
store it in the AWS Secrets Manager.
|
||||
|
||||
### User Annotations
|
||||
|
||||
User annotations can be configured in the configuration file both generally and for a specific data route. The annotations feature is only available when Authorization is enabled.
|
||||
To enable Annotations, it is necessary to create a relational database and add the database uri (typically `postgresql://[user[:password]@][netloc][:port][/dbname]`) to the secrets manager under `DB_URI`.
|
||||
The hosted version of cellxgene runs on AWS's [Aurora PostgreSQL](https://docs.aws.amazon.com/AmazonRDS/latest/AuroraUserGuide/Aurora.AuroraPostgreSQL.html) but any sqlalchemy compatible relational database should work.
|
||||
Once the database is set up apply the cellxgene schema to your database by running the following inside the cellxgene repo
|
||||
`PROJECT_ROOT=$(git rev-parse --show-toplevel)`
|
||||
`python3`
|
||||
Inside the python console
|
||||
`from sqlalchemy import create_engine`
|
||||
`from server.db.cellxgene_orm import Base`
|
||||
`uri = "[DB_URI]”`
|
||||
`engine = create_engine(uri)`
|
||||
|
||||
Base.metadata.create_all(engine)`
|
||||
|
||||
To check the schema was properly applied (or just to check what is in the database at any point)
|
||||
ssh into your database. For a postgres database this entails running:
|
||||
`psql [DB_URI]`
|
||||
|
||||
You'll also need to update your IAM policies to allow the instance to write to the s3 bucket.
|
||||
@@ -0,0 +1,191 @@
|
||||
"""cellxgene AWS elastic beanstalk application"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import hashlib
|
||||
import base64
|
||||
from urllib.parse import urlparse
|
||||
from flask import json
|
||||
import logging
|
||||
from flask_talisman import Talisman
|
||||
from flask_cors import CORS
|
||||
|
||||
|
||||
if os.path.isdir("/opt/python/log"):
|
||||
# This is the standard location where Amazon EC2 instances store the application logs.
|
||||
logging.basicConfig(
|
||||
filename="/opt/python/log/app.log",
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s.%(msecs)03d %(levelname)s %(module)s - %(funcName)s: %(message)s",
|
||||
datefmt="%Y-%m-%d %H:%M:%S",
|
||||
)
|
||||
|
||||
SERVERDIR = os.path.dirname(os.path.realpath(__file__))
|
||||
sys.path.append(SERVERDIR)
|
||||
|
||||
try:
|
||||
from backend.czi_hosted.common.config.app_config import AppConfig
|
||||
from backend.czi_hosted.app.app import Server
|
||||
from backend.common.utils.data_locator import DataLocator, discover_s3_region_name
|
||||
except Exception:
|
||||
logging.critical("Exception importing server modules", exc_info=True)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
class WSGIServer(Server):
|
||||
def __init__(self, app_config):
|
||||
super().__init__(app_config)
|
||||
|
||||
@staticmethod
|
||||
def _before_adding_routes(app, app_config):
|
||||
script_hashes = WSGIServer.get_csp_hashes(app, app_config)
|
||||
server_config = app_config.server_config
|
||||
|
||||
# add the api_base_url to the connect_src csp header.
|
||||
extra_connect_src = []
|
||||
api_base_url = server_config.get_api_base_url()
|
||||
if api_base_url:
|
||||
parse_api_base_url = urlparse(api_base_url)
|
||||
extra_connect_src = [f"{parse_api_base_url.scheme}://{parse_api_base_url.netloc}"]
|
||||
|
||||
# This hash should be in sync with the script within
|
||||
# `client/configuration/webpack/obsoleteHTMLTemplate.html`
|
||||
|
||||
# It is _very_ difficult to generate the correct hash manually,
|
||||
# consider forcing CSP to fail on the local server by intercepting the response via Requestly
|
||||
# this should print the failing script's hash to console.
|
||||
# See more here: https://github.com/chanzuckerberg/cellxgene/pull/1745
|
||||
obsolete_browser_script_hash = ["'sha256-/rmgOi/skq9MpiZxPv6lPb1PNSN+Uf4NaUHO/IjyfwM='"]
|
||||
csp = {
|
||||
"default-src": ["'self'"],
|
||||
"connect-src": ["'self'"] + extra_connect_src,
|
||||
"script-src": ["'self'", "'unsafe-eval'"] + obsolete_browser_script_hash + script_hashes,
|
||||
"style-src": ["'self'", "'unsafe-inline'"],
|
||||
"img-src": ["'self'", "https://cellxgene.cziscience.com", "data:"],
|
||||
"object-src": ["'none'"],
|
||||
"base-uri": ["'none'"],
|
||||
"frame-ancestors": ["'none'"],
|
||||
}
|
||||
|
||||
if not app.debug:
|
||||
csp["upgrade-insecure-requests"] = ""
|
||||
|
||||
if server_config.app__csp_directives:
|
||||
for k, v in server_config.app__csp_directives.items():
|
||||
if not isinstance(v, list):
|
||||
v = [v]
|
||||
csp[k] = csp.get(k, []) + v
|
||||
|
||||
# Add the web_base_url to the CORS header
|
||||
web_base_url = server_config.get_web_base_url()
|
||||
if web_base_url:
|
||||
web_base_url_parse = urlparse(web_base_url)
|
||||
allowed_origin = f"{web_base_url_parse.scheme}://{web_base_url_parse.netloc}"
|
||||
CORS(app, supports_credentials=True, origins=allowed_origin)
|
||||
|
||||
Talisman(
|
||||
app, force_https=server_config.app__force_https, frame_options="DENY", content_security_policy=csp,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def load_static_csp_hashes(app):
|
||||
csp_hashes = None
|
||||
try:
|
||||
with app.open_resource("../common/web/csp-hashes.json") as f:
|
||||
csp_hashes = json.load(f)
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
if not isinstance(csp_hashes, dict):
|
||||
csp_hashes = {}
|
||||
script_hashes = [f"'{hash}'" for hash in csp_hashes.get("script-hashes", [])]
|
||||
if len(script_hashes) == 0:
|
||||
logging.error("Content security policy hashes are missing, falling back to unsafe-inline policy")
|
||||
|
||||
return script_hashes
|
||||
|
||||
@staticmethod
|
||||
def compute_inline_csp_hashes(app, app_config):
|
||||
dataset_configs = [app_config.default_dataset_config] + list(app_config.dataroot_config.values())
|
||||
hashes = []
|
||||
for dataset_config in dataset_configs:
|
||||
inline_scripts = dataset_config.app__inline_scripts
|
||||
for script in inline_scripts:
|
||||
with app.open_resource(f"../common/web/templates/{script}") as f:
|
||||
content = f.read()
|
||||
# we use jinja2 template include, which trims final newline if present.
|
||||
if content[-1] == 0x0A:
|
||||
content = content[0:-1]
|
||||
hash = base64.b64encode(hashlib.sha256(content).digest())
|
||||
hashes.append(f"'sha256-{hash.decode('utf-8')}'")
|
||||
return hashes
|
||||
|
||||
@staticmethod
|
||||
def get_csp_hashes(app, app_config):
|
||||
script_hashes = WSGIServer.load_static_csp_hashes(app)
|
||||
script_hashes += WSGIServer.compute_inline_csp_hashes(app, app_config)
|
||||
return script_hashes
|
||||
|
||||
|
||||
try:
|
||||
app_config = AppConfig()
|
||||
|
||||
has_config = False
|
||||
# config file: look first for "config.yaml" in the current working directory
|
||||
config_file = "config.yaml"
|
||||
config_location = DataLocator(config_file)
|
||||
if config_location.exists():
|
||||
with config_location.local_handle() as lh:
|
||||
logging.info(f"Configuration from {config_file}")
|
||||
app_config.update_from_config_file(lh)
|
||||
has_config = True
|
||||
|
||||
else:
|
||||
# config file: second, use the CXG_CONFIG_FILE
|
||||
config_file = os.getenv("CXG_CONFIG_FILE")
|
||||
if config_file:
|
||||
region_name = discover_s3_region_name(config_file)
|
||||
config_location = DataLocator(config_file, region_name)
|
||||
if config_location.exists():
|
||||
with config_location.local_handle() as lh:
|
||||
logging.info(f"Configuration from {config_file}")
|
||||
app_config.update_from_config_file(lh)
|
||||
has_config = True
|
||||
else:
|
||||
logging.critical(f"Configuration file not found {config_file}")
|
||||
sys.exit(1)
|
||||
|
||||
if not has_config:
|
||||
logging.critical("No config file found")
|
||||
sys.exit(1)
|
||||
|
||||
dataroot = os.getenv("CXG_DATAROOT")
|
||||
if dataroot:
|
||||
logging.info("Configuration from CXG_DATAROOT")
|
||||
app_config.update_server_config(multi_dataset__dataroot=dataroot)
|
||||
|
||||
# overwrite configuration for the eb app
|
||||
app_config.update_default_dataset_config(embeddings__enable_reembedding=False,)
|
||||
app_config.update_server_config(multi_dataset__allowed_matrix_types=["cxg"],)
|
||||
|
||||
# complete config
|
||||
app_config.complete_config(logging.info)
|
||||
|
||||
server = WSGIServer(app_config)
|
||||
debug = False
|
||||
application = server.app
|
||||
|
||||
except Exception:
|
||||
logging.critical("Caught exception during initialization", exc_info=True)
|
||||
sys.exit(1)
|
||||
|
||||
if app_config.is_multi_dataset():
|
||||
logging.info(f"starting server with multi_dataset__dataroot={app_config.server_config.multi_dataset__dataroot}")
|
||||
else:
|
||||
logging.info(f"starting server with single_dataset__datapath={app_config.server_config.single_dataset__datapath}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
application.run(host=app_config.server_config.app__host, debug=debug, threaded=not debug, use_debugger=False)
|
||||
except Exception:
|
||||
logging.critical("Caught exception during initialization", exc_info=True)
|
||||
sys.exit(1)
|
||||
@@ -0,0 +1,38 @@
|
||||
import sys
|
||||
import argparse
|
||||
import yaml
|
||||
|
||||
from backend.czi_hosted.common.config.app_config import AppConfig
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser("A script to check hosted configuration files")
|
||||
parser.add_argument("config_file", help="the configuration file")
|
||||
parser.add_argument(
|
||||
"-s",
|
||||
"--show",
|
||||
default=False,
|
||||
action="store_true",
|
||||
help="print the configuration. NOTE: this may print secret values to stdout",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
app_config = AppConfig()
|
||||
try:
|
||||
app_config.update_from_config_file(args.config_file)
|
||||
app_config.complete_config()
|
||||
except Exception as e:
|
||||
print(f"Error: {str(e)}")
|
||||
print("FAIL:", args.config_file)
|
||||
sys.exit(1)
|
||||
|
||||
if args.show:
|
||||
yaml_config = app_config.config_to_dict()
|
||||
yaml.dump(yaml_config, sys.stdout)
|
||||
|
||||
print("PASS:", args.config_file)
|
||||
sys.exit(0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,98 @@
|
||||
"""This is a simple script to ensure the custom requirements.txt do not violate
|
||||
the server requirements.txt. A hosted cellxgene deployment may specify the exact
|
||||
version requirements on all the modules, and may add additional modules.
|
||||
This script is meant to aid in making that list of custom requirements easier to maintain.
|
||||
If cellxgene adds a new dependency, or changes the version requirements of an existing
|
||||
dependency, then this script can check if the custom requirements are still valid"""
|
||||
|
||||
import sys
|
||||
import requirements
|
||||
from packaging.version import Version
|
||||
import pkg_resources
|
||||
|
||||
|
||||
def check(expected, custom):
|
||||
"""checks that the custom requirements meet all the requirements of the expected requirements.
|
||||
The custom set of requirements may contain additional entries than expected.
|
||||
The requirements in custom must all be exact (==).
|
||||
An expected requirement must be present in custom, and must match all the specs
|
||||
for that requirement.
|
||||
|
||||
expected : name of the expected requirement.txt file
|
||||
custom : name of the custom requirements.txt file
|
||||
"""
|
||||
edict = parse_requirements(expected)
|
||||
cdict = parse_requirements(custom)
|
||||
|
||||
okay = True
|
||||
|
||||
# cdict must only have exact requirements (==)
|
||||
for cname, cspecs in cdict.items():
|
||||
if len(cspecs) != 1 or cspecs[0][0] != "==":
|
||||
print(f"Error, spec must be an exact requirement {custom}: {cname} {str(cspecs)}")
|
||||
okay = False
|
||||
|
||||
for ename, especs in edict.items():
|
||||
if ename not in cdict:
|
||||
print(f"Error, missing requirement from {custom}: {ename} {str(especs)}")
|
||||
okay = False
|
||||
continue
|
||||
|
||||
cver = Version(cdict[ename][0][1])
|
||||
for espec in especs:
|
||||
rokay = check_version(cver, espec[0], Version(espec[1]))
|
||||
if not rokay:
|
||||
print(f"Error, failed requirement from {custom}: {ename} {espec}, {cver}")
|
||||
okay = False
|
||||
|
||||
if okay:
|
||||
print("requirements check successful")
|
||||
sys.exit(0)
|
||||
else:
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def parse_requirements(fname):
|
||||
"""Read a requirements file and return a dict of modules name / specification"""
|
||||
try:
|
||||
with open(fname, "r") as fd:
|
||||
try:
|
||||
# pylint: disable=no-member
|
||||
rdict = {req.name: req.specs for req in requirements.parse(fd)}
|
||||
except pkg_resources.RequirementParseError:
|
||||
print(f"Unable to parse the requirements file: {fname}")
|
||||
sys.exit(1)
|
||||
except Exception as e:
|
||||
print(f"Unable to open file {fname}: {str(e)}")
|
||||
sys.exit(1)
|
||||
|
||||
return rdict
|
||||
|
||||
|
||||
# pylint: disable=too-many-return-statements
|
||||
def check_version(cver, optype, ever):
|
||||
"""
|
||||
Simple version check.
|
||||
Note: There is more complexity to comparing version (PEP440).
|
||||
However the use cases in cellxgene are limited, and do not require a general solution.
|
||||
"""
|
||||
|
||||
if optype == "==":
|
||||
return cver == ever
|
||||
if optype == "!=":
|
||||
return cver != ever
|
||||
if optype == ">=":
|
||||
return cver >= ever
|
||||
if optype == ">":
|
||||
return cver > ever
|
||||
if optype == "<=":
|
||||
return cver <= ever
|
||||
if optype == "<":
|
||||
return cver < ever
|
||||
|
||||
print(f"Error, optype not handled: {optype}")
|
||||
return False
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
check(sys.argv[1], sys.argv[2])
|
||||
@@ -0,0 +1,2 @@
|
||||
deploy:
|
||||
artifact: artifact.zip
|
||||
Reference in New Issue
Block a user