pip install cellxgene
@@ -110,8 +110,8 @@
cellxgene launch https://cellxgene-example-data.czi.technology/pbmc3k.h5ad
-To explore more datasets already formatted for cellxgene, check out the Demo data or
-see Preparing your data to learn more about formatting your own
+
To explore more datasets already formatted for cellxgene, check out the Demo data or
+see Preparing your data to learn more about formatting your own
data for cellxgene.
Getting help
diff --git a/docs/_site/posts/annotations.html b/docs/_site/posts/annotations.html
index 57a31083..2a76ec9f 100644
--- a/docs/_site/posts/annotations.html
+++ b/docs/_site/posts/annotations.html
@@ -5,9 +5,9 @@
-
+
annotations | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"Creating annotations","@type":"WebPage","headline":"annotations","url":"https://chanzuckerberg.github.io/cellxgene/posts/annotations.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/contact.html b/docs/_site/posts/contact.html
index 59412b03..33c31b51 100644
--- a/docs/_site/posts/contact.html
+++ b/docs/_site/posts/contact.html
@@ -5,9 +5,9 @@
-
+
Contact | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"Contact","@type":"WebPage","headline":"Contact","url":"https://chanzuckerberg.github.io/cellxgene/posts/contact.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/contribute.html b/docs/_site/posts/contribute.html
index 8e793a58..494d2ca3 100644
--- a/docs/_site/posts/contribute.html
+++ b/docs/_site/posts/contribute.html
@@ -5,9 +5,9 @@
-
+
Code of conduct | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"An interactive explorer for single-cell transcriptomics data","@type":"WebPage","headline":"Code of conduct","url":"https://chanzuckerberg.github.io/cellxgene/posts/contribute.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/demo-data.html b/docs/_site/posts/demo-data.html
index 3879a512..479e6249 100644
--- a/docs/_site/posts/demo-data.html
+++ b/docs/_site/posts/demo-data.html
@@ -5,9 +5,9 @@
-
+
demo-data | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"Demo datasets","@type":"WebPage","headline":"demo-data","url":"https://chanzuckerberg.github.io/cellxgene/posts/demo-data.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/gallery.html b/docs/_site/posts/gallery.html
index 976c88f6..2321d81e 100644
--- a/docs/_site/posts/gallery.html
+++ b/docs/_site/posts/gallery.html
@@ -5,9 +5,9 @@
-
+
Gallery | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"An interactive explorer for single-cell transcriptomics data","@type":"WebPage","headline":"Gallery","url":"https://chanzuckerberg.github.io/cellxgene/posts/gallery.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/hosted.html b/docs/_site/posts/hosted.html
index ec586a6f..0e2d2b6a 100644
--- a/docs/_site/posts/hosted.html
+++ b/docs/_site/posts/hosted.html
@@ -5,9 +5,9 @@
-
+
Hosting cellxgene on the web | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"An interactive explorer for single-cell transcriptomics data","@type":"WebPage","headline":"Hosting cellxgene on the web","url":"https://chanzuckerberg.github.io/cellxgene/posts/hosted.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/index.html b/docs/_site/posts/index.html
deleted file mode 100644
index 13cce65b..00000000
--- a/docs/_site/posts/index.html
+++ /dev/null
@@ -1,109 +0,0 @@
-
-
-
-
-
-
-
-
-Index | cellxgene
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
- Quick start
-
-Whether you need to visualize one thousand cells or one million, cellxgene helps you gain insight into your single-cell data.
-
-To install cellxgene you need Python 3.6+. We recommend installing cellxgene into a conda or virtual environment.
-
-Install the package.
-
-
-Download an example anndata file
-
-curl -o tabula-muris.h5ad https://cellxgene-example-data.czi.technology/tabula-muris.h5ad.zip
-unzip tabula-muris.h5ad.zip
-
-
-Launch cellxgene
-cellxgene launch tabula-muris.h5ad --open
-
-
-To explore more datasets already formatted for cellxgene, check out the Demo data or
-see Preparing your data to learn more about formatting your own
-data for cellxgene.
-
-Getting help
-
-We’d love to hear from you!
-
-For questions, suggestions, or accolades, join the #cellxgene-users channel on the CZI Science Slack and say “hi!”.
-
-For any errors, report bugs on Github .
-
-
-
-
-
-
-
-
diff --git a/docs/_site/posts/install.html b/docs/_site/posts/install.html
index c7614f4b..2eb470d8 100644
--- a/docs/_site/posts/install.html
+++ b/docs/_site/posts/install.html
@@ -5,9 +5,9 @@
-
+
Install | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"An interactive explorer for single-cell transcriptomics data","@type":"WebPage","headline":"Install","url":"https://chanzuckerberg.github.io/cellxgene/posts/install.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/launch.html b/docs/_site/posts/launch.html
index ac4bea95..4aed95b3 100644
--- a/docs/_site/posts/launch.html
+++ b/docs/_site/posts/launch.html
@@ -5,9 +5,9 @@
-
+
demo-data | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"Demo datasets","@type":"WebPage","headline":"demo-data","url":"https://chanzuckerberg.github.io/cellxgene/posts/launch.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/methods.html b/docs/_site/posts/methods.html
index 2840a3d6..8c97a84c 100644
--- a/docs/_site/posts/methods.html
+++ b/docs/_site/posts/methods.html
@@ -5,9 +5,9 @@
-
+
Methods | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"An interactive explorer for single-cell transcriptomics data","@type":"WebPage","headline":"Methods","url":"https://chanzuckerberg.github.io/cellxgene/posts/methods.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/prepare.html b/docs/_site/posts/prepare.html
index adf0c529..56dce0c2 100644
--- a/docs/_site/posts/prepare.html
+++ b/docs/_site/posts/prepare.html
@@ -5,9 +5,9 @@
-
+
prepare | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"Preparing your data","@type":"WebPage","headline":"prepare","url":"https://chanzuckerberg.github.io/cellxgene/posts/prepare.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/roadmap.html b/docs/_site/posts/roadmap.html
index 5a795d78..34dc26b9 100644
--- a/docs/_site/posts/roadmap.html
+++ b/docs/_site/posts/roadmap.html
@@ -5,9 +5,9 @@
-
+
roadmap | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"Roadmap","@type":"WebPage","headline":"roadmap","url":"https://chanzuckerberg.github.io/cellxgene/posts/roadmap.html","@context":"https://schema.org"}
-
+
diff --git a/docs/_site/posts/troubleshooting.html b/docs/_site/posts/troubleshooting.html
index 8132fce3..a936237a 100644
--- a/docs/_site/posts/troubleshooting.html
+++ b/docs/_site/posts/troubleshooting.html
@@ -5,9 +5,9 @@
-
+
Troubleshooting | cellxgene
-
+
@@ -16,10 +16,10 @@
+{"publisher":{"@type":"Organization","logo":{"@type":"ImageObject","url":"https://chanzuckerberg.github.io/cellxgene/cellxgene-logo.png"}},"description":"Troubleshooting","@type":"WebPage","headline":"Troubleshooting","url":"https://chanzuckerberg.github.io/cellxgene/posts/troubleshooting.html","@context":"https://schema.org"}
-
+
diff --git a/docs/index.md b/docs/index.md
index 9afd0e61..b147a754 100644
--- a/docs/index.md
+++ b/docs/index.md
@@ -7,7 +7,7 @@ layout: default
Whether you need to visualize one thousand cells or one million, cellxgene helps you gain insight into your single-cell data.
-To install cellxgene you need Python 3.6+. We recommend [installing cellxgene into a conda or virtual environment.](install)
+To install cellxgene you need Python 3.6+. We recommend [installing cellxgene into a conda or virtual environment.](posts/install)
Install the package.
``` bash
@@ -20,8 +20,8 @@ Launch cellxgene with an example [anndata](https://anndata.readthedocs.io/en/lat
cellxgene launch https://cellxgene-example-data.czi.technology/pbmc3k.h5ad
```
-To explore more datasets already formatted for cellxgene, check out the [Demo data](demo-data) or
-see [Preparing your data](prepare) to learn more about formatting your own
+To explore more datasets already formatted for cellxgene, check out the [Demo data](posts/demo-data) or
+see [Preparing your data](posts/prepare) to learn more about formatting your own
data for cellxgene.
# Getting help
diff --git a/docs/posts/cellxgene_cziscience_com.md b/docs/posts/cellxgene_cziscience_com.md
index 833ae0ce..43144045 100644
--- a/docs/posts/cellxgene_cziscience_com.md
+++ b/docs/posts/cellxgene_cziscience_com.md
@@ -311,5 +311,29 @@ with a link to embed on your own site, please drop us a note at Nature
+
+ Single Soma Transcriptomics - AT8
+
+ bioRxiv preprint
+
+
+
+ Single Soma Transcriptomics - MAP2
+
+ bioRxiv preprint
+
+
+
+ Single Soma Transcriptomics - MAP2AT8
+
+ bioRxiv preprint
+
+
+
+ Single-cell longitudinal analysis of SARS-CoV-2 infection in human bronchial epithelial cells
+
+ bioRxiv preprint
+
+
diff --git a/docs/posts/hosted.md b/docs/posts/hosted.md
index 1168bdf0..073ede26 100644
--- a/docs/posts/hosted.md
+++ b/docs/posts/hosted.md
@@ -38,31 +38,38 @@ If you know of other solutions, drop us a note and we'll add to this list.
# Deploying cellxgene with Heroku
-## Quickstart
+## Heroku Support
-Clicking on the following button will forward you to Heroku to begin the deployment process:
+The cellxgene team has decided to end our support for our experimental deploy to Heroku button as we move towards providing a supported method of hosted cellxgene.
-
-
-
+While we no longer directly support Heroku, it is still possible to create a Heroku app via [our provided Dockerfile here](https://github.com/chanzuckerberg/cellxgene/blob/main/Dockerfile) and [Heroku's documentation](https://devcenter.heroku.com/articles/build-docker-images-heroku-yml).
-If not already logged in to Heroku, there you will be prompted to log in or sign up for an account.
+You may have to tweak the `Dockerfile` like so:
-Once logged in you will be sent to the setup page. Here you can set some of the basic settings for the app:
+```Dockerfile
+FROM ubuntu:bionic
-### Default settings
+ENV LC_ALL=C.UTF-8
+ENV LANG=C.UTF-8
-- `App name`: the unique name for your deployment
-- This will also serve as the default URL (e.g. https://cellxgene.herokapp.com/)
-- `App owner`: Who will own this app. Either you personally or an organization/team
-- `Region`: Location of the server where the app will be deployed (EU or US)
+RUN apt-get update && \
+ apt-get install -y build-essential libxml2-dev python3-dev python3-pip zlib1g-dev python3-requests && \
+ pip3 install cellxgene
-### Configuration
+# ENTRYPOINT ["cellxgene"] # Heroku doesn't work well with ENTRYPOINT
+```
-- `DATASET`: A _publicly_ accessible URL pointing to a .h5ad file to view
-- This defaults to pbm3k.h5ad
+and provide a `heroku.yml` file similar to this:
-After filling out the settings and pressing the `Deploy app` button Heroku will begin building your deployment. This process will take a few minutes, but once completed you will have a personal free hosted version of cellxgene!
+```yml
+build:
+ docker:
+ web: Dockerfile
+run:
+ web:
+ command:
+ - cellxgene launch --host 0.0.0.0 --port $PORT $DATASET # the DATATSET config var must be defined in your dashboard settings.
+```
## What is Heroku?
diff --git a/docs/posts/index.html b/docs/posts/index.html
deleted file mode 100644
index 13cce65b..00000000
--- a/docs/posts/index.html
+++ /dev/null
@@ -1,109 +0,0 @@
-
-
-
-
-
-
-
-
-Index | cellxgene
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
- Quick start
-
-Whether you need to visualize one thousand cells or one million, cellxgene helps you gain insight into your single-cell data.
-
-To install cellxgene you need Python 3.6+. We recommend installing cellxgene into a conda or virtual environment.
-
-Install the package.
-
-
-Download an example anndata file
-
-curl -o tabula-muris.h5ad https://cellxgene-example-data.czi.technology/tabula-muris.h5ad.zip
-unzip tabula-muris.h5ad.zip
-
-
-Launch cellxgene
-cellxgene launch tabula-muris.h5ad --open
-
-
-To explore more datasets already formatted for cellxgene, check out the Demo data or
-see Preparing your data to learn more about formatting your own
-data for cellxgene.
-
-Getting help
-
-We’d love to hear from you!
-
-For questions, suggestions, or accolades, join the #cellxgene-users channel on the CZI Science Slack and say “hi!”.
-
-For any errors, report bugs on Github .
-
-
-
-
-
-
-
-
diff --git a/experiments/heroku/Dockerfile b/experiments/heroku/Dockerfile
deleted file mode 100644
index 28749d8b..00000000
--- a/experiments/heroku/Dockerfile
+++ /dev/null
@@ -1,7 +0,0 @@
-FROM python:3.7
-
-WORKDIR /usr/src/app
-
-RUN pip3 install cellxgene
-
-expose 5005
diff --git a/experiments/heroku/README.md b/experiments/heroku/README.md
deleted file mode 100644
index 24668a35..00000000
--- a/experiments/heroku/README.md
+++ /dev/null
@@ -1,58 +0,0 @@
-# cellxgene cloud deployment with Heroku
-
-## Quickstart
-
-Clicking on the following button will forward you to Heroku to begin the deployment process:
-
-
-
-
-
-If not already logged in to Heroku, there you will be prompted to log in or sign up for an account.
-
-Once logged in you will be sent to the setup page. Here you can set some of the basic settings for the app:
-
-#### Default settings
-
-- `App name`: the unique name for your deployment
-- This will also serve as the default URL (e.g. https://cellxgene.herokapp.com/)
-- `App owner`: Who will own this app. Either you personally or an organization/team
-- `Region`: Location of the server where the app will be deployed (EU or US)
-
-#### Configuration
-
-- `DATASET`: A _publicly_ accessible URL pointing to a .h5ad file to view
-- This defaults to pbm3k.h5ad
-
-After filling out the settings and pressing the `Deploy app` button Heroku will begin building your deployment. This process will take a few minutes, but once completed you will have a personal free hosted version of cellxgene!
-
-## What is Heroku?
-
-Heroku is a quick and easy way to host applications on the cloud.
-
-A Heroku deployment of cellxgene means that the app is not running on your local machine. Instead, the app is installed, configured, and ran on the Heroku servers (read: cloud).
-
-On Heroku's servers, applications run on a [dyno](https://www.heroku.com/dynos) which are Heroku's implementation and abstraction of containers.
-
-Heroku is one of many options available for hosting instances of cellxgene on the web.
-Some other options include: Amazon Web Services, Google Cloud Platform, Digital Ocean, and Microsoft Azure.
-
-## Why use Heroku to deploy cellxgene?
-
-What Heroku enables is a quick, non-technical method of setting up a cellxgene instance. No command line knowledge needed. This also allows machines to access the instance via the internet, so sharing a visualized dataset is as simple as sharing a link.
-
-Because cellxgene currently heavily relies on its Python backend for providing the viewer with the necessary data and tooling, it is currently not possible to host cellxgene as a static webpage.
-
-This is a good option if you want to quickly deploy an instance of cellxgene to the web. Heroku deployments are free for small datasets up to around 250MBs in size. See below regarding larger datasets.
-
-## When should I not deploy with Heroku?
-
-- The default free dyno offered by Heroku is limited in memory to 512 MBs
- - The amount of memory needed for the dyno is roughly the same size as the h5ad file
- - Heroku offers tiered paid dynos. More can be found [here](https://www.heroku.com/pricing)
- - Note that this can get _very_ expensive for larger datasets (\$25+ a month)
-- On the free dyno, after 30 minutes of inactivity, Heroku will put your app into a hibernation mode. On the next access, Heroku will need time to boot the dyno back online.
-- Having multiple simultaneous users requires more memory. This means that the free container size is easily overwhelmed by multiple users, even with small datasets; this can be addressed by purchasing a larger container size
-- For this facilitated Heroku deployment to work, your dataset must be hosted on a publicly accessible URL
-- By default, Heroku publically shares your instance to anyone with the URL.
- - There are many ways of securing your instance. One quick and simple way is by installing [wwwhisper](https://elements.heroku.com/addons/wwwhisper), a Heroku addon
diff --git a/heroku.yml b/heroku.yml
deleted file mode 100644
index 75050a73..00000000
--- a/heroku.yml
+++ /dev/null
@@ -1,5 +0,0 @@
-build:
- docker:
- web: experiments/heroku/Dockerfile
-run:
- web: cellxgene launch $DATASET --host 0.0.0.0 --port $PORT
diff --git a/server/Makefile b/server/Makefile
index 74c44f7b..700b9a27 100644
--- a/server/Makefile
+++ b/server/Makefile
@@ -7,11 +7,35 @@ clean:
rm -f common/web/csp-hashes.json
.PHONY: unit-test
-unit-test:
+unit-test: create-test-db
PYTHONWARNINGS=ignore:ResourceWarning coverage run \
--source=app,cli,common,compute,converters,data_anndata,data_common,data_cxg \
--omit=.coverage,data_common/fbs/NetEncoding,venv \
-m unittest discover \
--start-directory test/ \
--top-level-directory ../ \
- --verbose
+ --verbose; test_result=$$?; \
+ $(MAKE) clean-test-db; \
+ exit $$test_result \
+
+
+.PHONY: test-db
+test-db: create-test-db
+ PYTHONWARNINGS=ignore:ResourceWarning coverage run \
+ --source=app,cli,common,compute,converters,data_anndata,data_common,data_cxg \
+ --omit=.coverage,data_common/fbs/NetEncoding,venv \
+ -m unittest discover \
+ --start-directory test/test_database \
+ --top-level-directory ../ \
+ --verbose; test_result=$$?; \
+ $(MAKE) clean-test-db; \
+ exit $$test_result
+
+.PHONY: create-test-db
+create-test-db:
+ -docker run -d -p 5432:5432 --name test_db -e POSTGRES_PASSWORD=test_pw postgres
+
+.PHONY: clean-test-db
+clean-test-db:
+ -docker stop test_db
+ -docker rm test_db
diff --git a/server/__init__.py b/server/__init__.py
index 27ce5322..94238d9a 100644
--- a/server/__init__.py
+++ b/server/__init__.py
@@ -1,8 +1,9 @@
-from server.common.utils import import_plugins
import logging
import sys
-__version__ = "0.15.0"
+from server.common.utils.utils import import_plugins
+
+__version__ = "0.16.0"
display_version = "cellxgene v" + __version__
try:
diff --git a/server/app/app.py b/server/app/app.py
index e290ca35..d01b1964 100644
--- a/server/app/app.py
+++ b/server/app/app.py
@@ -1,22 +1,19 @@
import datetime
import logging
+from functools import wraps
+from http import HTTPStatus
-from flask import Flask, redirect, current_app, make_response, render_template, abort
-from flask import Blueprint, request
+from flask import Flask, redirect, current_app, make_response, render_template, abort, Blueprint, request
from flask_restful import Api, Resource
from server_timing import Timing as ServerTiming
-from http import HTTPStatus
-
import server.common.rest as common_rest
-from server.common.errors import DatasetAccessError, RequestException
-from server.common.utils import path_join, Float32JSONEncoder
from server.common.data_locator import DataLocator
+from server.common.errors import DatasetAccessError, RequestException
from server.common.health import health_check
+from server.common.utils.utils import path_join, Float32JSONEncoder
from server.data_common.matrix_loader import MatrixDataLoader
-from functools import wraps
-
webbp = Blueprint("webapp", "server.common.web", template_folder="templates")
ONE_WEEK = 7 * 24 * 60 * 60
@@ -85,6 +82,7 @@ def dataset_index(url_dataroot=None, dataset=None):
try:
cache_manager = current_app.matrix_data_cache_manager
with cache_manager.data_adaptor(url_dataroot, location, app_config) as data_adaptor:
+ data_adaptor.set_uri_path(f"{url_dataroot}/{dataset}")
dataset_title = app_config.get_title(data_adaptor)
return render_template(
"index.html", datasetTitle=dataset_title, SCRIPTS=scripts, INLINE_SCRIPTS=inline_scripts
@@ -129,7 +127,7 @@ def get_data_adaptor(url_dataroot=None, dataset=None):
# sufficient to check that the datapath starts with the
# dataroot to determine that the datapath is under the dataroot.
if not datapath.startswith(dataroot):
- raise DatasetAccessError("Invalid dataset {url_dataroot}/{dataset}")
+ raise DatasetAccessError(f"Invalid dataset {url_dataroot}/{dataset}")
if datapath is None:
return common_rest.abort_and_log(HTTPStatus.BAD_REQUEST, "Invalid dataset NONE", loglevel=logging.INFO)
@@ -138,11 +136,24 @@ def get_data_adaptor(url_dataroot=None, dataset=None):
return cache_manager.data_adaptor(dataset_key, datapath, config)
+def requires_authentication(func):
+ @wraps(func)
+ def wrapped_function(self, *args, **kwargs):
+ auth = current_app.auth
+ if auth.is_user_authenticated():
+ return func(self, *args, **kwargs)
+ else:
+ return make_response("not authenticated", HTTPStatus.UNAUTHORIZED)
+
+ return wrapped_function
+
+
def rest_get_data_adaptor(func):
@wraps(func)
def wrapped_function(self, dataset=None):
try:
with get_data_adaptor(self.url_dataroot, dataset) as data_adaptor:
+ data_adaptor.set_uri_path(f"{self.url_dataroot}/{dataset}")
return func(self, data_adaptor)
except DatasetAccessError as e:
return common_rest.abort_and_log(
@@ -160,6 +171,17 @@ def dataroot_test_index():
config = current_app.app_config
server_config = config.server_config
+
+ auth = server_config.auth
+ if auth.is_valid_authentication_type():
+ if server_config.auth.is_user_authenticated():
+ data += f"Logged in as {auth.get_user_id()} / {auth.get_user_name()} / {auth.get_user_email()}
"
+ if auth.requires_client_login():
+ if server_config.auth.is_user_authenticated():
+ data += "Logout
"
+ else:
+ data += "Login
"
+
datasets = []
for dataroot_dict in server_config.multi_dataset__dataroot.values():
dataroot = dataroot_dict["dataroot"]
@@ -224,6 +246,7 @@ class AnnotationsObsAPI(DatasetResource):
def get(self, data_adaptor):
return common_rest.annotations_obs_get(request, data_adaptor)
+ @requires_authentication
@cache_control(no_store=True)
@rest_get_data_adaptor
def put(self, data_adaptor):
@@ -338,6 +361,7 @@ class Server:
lambda dataset, url_dataroot=url_dataroot: dataset_index(url_dataroot, dataset),
methods=["GET"],
)
+
else:
bp_api = Blueprint("api", __name__, url_prefix=api_version)
resources = get_api_resources(bp_api)
@@ -345,3 +369,9 @@ class Server:
self.app.matrix_data_cache_manager = server_config.matrix_data_cache_manager
self.app.app_config = app_config
+
+ auth = server_config.auth
+ self.app.auth = auth
+ if auth.requires_client_login():
+ auth.add_url_rules(self.app)
+ auth.complete_setup(self.app)
diff --git a/server/auth/__init__.py b/server/auth/__init__.py
new file mode 100644
index 00000000..1b33bebc
--- /dev/null
+++ b/server/auth/__init__.py
@@ -0,0 +1,7 @@
+
+# import the built in auth types so they can be registered
+
+import server.auth.auth_none # noqa: F401
+import server.auth.auth_test # noqa: F401
+import server.auth.auth_session # noqa: F401
+import server.auth.auth_oauth # noqa: F401
diff --git a/server/auth/auth.py b/server/auth/auth.py
new file mode 100644
index 00000000..bb86ea64
--- /dev/null
+++ b/server/auth/auth.py
@@ -0,0 +1,87 @@
+from abc import ABC, abstractmethod
+
+
+class AuthTypeBase(ABC):
+ """Base type for all authentication types."""
+
+ def __init__(self):
+ super().__init__()
+
+ @abstractmethod
+ def is_valid_authentication_type(self):
+ """Return True if the auth type is valid, e.g. it can return userinfo and username.
+ (AuthTypeNone is the only one type that returns False)"""
+ pass
+
+ def requires_client_login(self):
+ """Return True if the user needs to login from the client (e.g. Login button is shown)"""
+ return False
+
+ @abstractmethod
+ def complete_setup(self, app):
+ """complete any setup that may be needed by this auth type. The Flask app is passed in.
+ This is the last auth function called before the server starts to run."""
+ pass
+
+ @abstractmethod
+ def is_user_authenticated(self):
+ """Return True if the user is authenticated"""
+ pass
+
+ @abstractmethod
+ def get_user_id(self):
+ """Return the id for this user (string)"""
+ pass
+
+ @abstractmethod
+ def get_user_name(self):
+ """Return the name of the user (string)"""
+ pass
+
+ @abstractmethod
+ def get_user_email(self):
+ """Return the name of the user (string)"""
+ pass
+
+
+class AuthTypeClientBase(AuthTypeBase):
+ """Base type for all authentication types that require the client to login"""
+
+ def __init__(self):
+ super().__init__()
+
+ def requires_client_login(self):
+ return True
+
+ @abstractmethod
+ def add_url_rules(self, selfapp):
+ """Add url rules to the app (like /login, /logout, etc)"""
+ pass
+
+ @abstractmethod
+ def get_login_url(self, data_adaptor):
+ """Return the url for the login route"""
+ pass
+
+ @abstractmethod
+ def get_logout_url(self, data_adaptor):
+ """Return the url for the logout route"""
+ pass
+
+
+class AuthTypeFactory:
+ """Factory class to create an authentication type"""
+
+ auth_types = {}
+
+ @staticmethod
+ def register(name, auth_type):
+ assert(issubclass(auth_type, AuthTypeBase))
+ AuthTypeFactory.auth_types[name] = auth_type
+
+ @staticmethod
+ def create(name, app_config):
+ auth_type = AuthTypeFactory.auth_types.get(name)
+ if auth_type is None:
+ return None
+ return auth_type(app_config)
diff --git a/server/auth/auth_none.py b/server/auth/auth_none.py
new file mode 100644
index 00000000..c482e3c4
--- /dev/null
+++ b/server/auth/auth_none.py
@@ -0,0 +1,28 @@
+from server.auth.auth import AuthTypeBase, AuthTypeFactory
+
+
+class AuthTypeNone(AuthTypeBase):
+
+ def __init__(self, app_config):
+ super().__init__()
+
+ def is_valid_authentication_type(self):
+ return False
+
+ def complete_setup(self, app):
+ pass
+
+ def is_user_authenticated(self):
+ return True
+
+ def get_user_id(self):
+ return None
+
+ def get_user_name(self):
+ return None
+
+ def get_user_email(self):
+ return None
+
+
+AuthTypeFactory.register(None, AuthTypeNone)
diff --git a/server/auth/auth_oauth.py b/server/auth/auth_oauth.py
new file mode 100644
index 00000000..8ed52036
--- /dev/null
+++ b/server/auth/auth_oauth.py
@@ -0,0 +1,240 @@
+from flask import session, request, redirect, current_app, after_this_request, has_request_context, g
+from server.auth.auth import AuthTypeClientBase, AuthTypeFactory
+from server.common.errors import AuthenticationError, ConfigurationError
+from urllib.parse import urlencode
+from urllib.request import urlopen
+import json
+
+# It is not required to have authlib or jose.
+# However, it is a configuration error to use this auth type if they are not installed.
+missingimport = []
+try:
+ from authlib.integrations.flask_client import OAuth
+except ModuleNotFoundError:
+ missingimport.append("authlib")
+
+try:
+ from jose import jwt
+except ModuleNotFoundError:
+ missingimport.append("jose")
+
+
+class AuthTypeOAuth(AuthTypeClientBase):
+ """An authentication type for oauth2 logins."""
+
+ CXG_ID_TOKEN = "id_token"
+
+ def __init__(self, server_config):
+ super().__init__()
+ if missingimport:
+ raise ConfigurationError(f"oauth requires these modules: {', '.join(missingimport)}")
+ self.algorithms = ["RS256"]
+ self.api_base_url = server_config.authentication__params_oauth__api_base_url
+ self.client_id = server_config.authentication__params_oauth__client_id
+ self.client_secret = server_config.authentication__params_oauth__client_secret
+ self.callback_base_url = server_config.authentication__params_oauth__callback_base_url
+ self.session_cookie = server_config.authentication__params_oauth__session_cookie
+ self.cookie_params = server_config.authentication__params_oauth__cookie
+ self._validate_cookie_params()
+
+ # set the audience
+ self.audience = self.client_id
+
+ # load the jwks (JSON Web Key Set).
+ # The JSON Web Key Set (JWKS) is a set of keys which contains the public keys used to verify
+ # any JSON Web Token (JWT) issued by the authorization server and signed using the RS256
+ try:
+ jwksloc = f"{self.api_base_url}/.well-known/jwks.json"
+ jwksurl = urlopen(jwksloc)
+ self.jwks = json.loads(jwksurl.read())
+ except Exception:
+ raise ConfigurationError(f"error in oauth, api_url_base: {self.api_base_url}, cannot access {jwksloc}")
+
+ def _validate_cookie_params(self):
+ """check the cookie_params, and raise a ConfigurationError if there is something wrong"""
+ if self.session_cookie:
+ return
+
+ if not isinstance(self.cookie_params, dict):
+ raise ConfigurationError("either session_cookie or cookie must be set")
+ valid_keys = {"key", "max_age", "expires", "path", "domain", "secure", "httponly", "samesite"}
+ keys = set(self.cookie_params.keys())
+ unknown = keys - valid_keys
+ if unknown:
+ raise ConfigurationError(f"unexpected key in cookie params: {', '.join(unknown)}")
+ if "key" not in keys:
+ raise ConfigurationError("must have a key (name) in the cookie params")
+
+ def is_valid_authentication_type(self):
+ return True
+
+ def requires_client_login(self):
+ return True
+
+ def add_url_rules(self, app):
+ app.add_url_rule("/login", "login", self.login, methods=["GET"])
+ app.add_url_rule("/logout", "logout", self.logout, methods=["GET"])
+ app.add_url_rule("/oauth2/callback", "callback", self.callback, methods=["GET"])
+
+ def complete_setup(self, flask_app):
+ self.oauth = OAuth(flask_app)
+ if self.callback_base_url is None:
+ # In this case, assume the server is running on the same host as the client,
+ # and the oauth provider has been configured
+ # with a callback that understands a localhost callback (e.g. A http://localhost:5005).
+ server_config = flask_app.app_config.server_config
+ self.callback_base_url = f"http://{server_config.app__host}:{server_config.app__port}"
+
+ self.client = self.oauth.register(
+ "oauth",
+ client_id=self.client_id,
+ client_secret=self.client_secret,
+ api_base_url=self.api_base_url,
+ access_token_url=f"{self.api_base_url}/oauth/token",
+ authorize_url=f"{self.api_base_url}/authorize",
+ client_kwargs={
+ "scope" : "openid profile email",
+ }
+ )
+
+ def is_user_authenticated(self):
+ try:
+ payload = self.get_jwt_payload()
+ return payload is not None
+ except AuthenticationError:
+ return False
+
+ def get_user_id(self):
+ payload = self.get_jwt_payload()
+ if payload and payload.get("sub"):
+ return payload.get("sub")
+ return None
+
+ def get_user_name(self):
+ payload = self.get_jwt_payload()
+ if payload and payload.get("name"):
+ return payload.get("name")
+ return None
+
+ def get_user_email(self):
+ payload = self.get_jwt_payload()
+ if payload and payload.get("email"):
+ return payload.get("email")
+ return None
+
+ def update_response(self, response):
+ response.cache_control.update(
+ dict(public=True, max_age=0, no_store=True, no_cache=True, must_revalidate=True))
+
+ def login(self):
+ callbackurl = f'{self.callback_base_url}/oauth2/callback'
+ return_path = request.args.get("dataset", "")
+ return_to = f"{self.callback_base_url}/{return_path}"
+ # save the return path in the session cookie, accessed in the callback function
+ session["oauth_callback_redirect"] = return_to
+ response = self.client.authorize_redirect(redirect_uri=callbackurl)
+ self.update_response(response)
+ return response
+
+ def logout(self):
+ if self.session_cookie:
+ if self.CXG_ID_TOKEN in session:
+ del session[self.CXG_ID_TOKEN]
+ else:
+ @after_this_request
+ def remove_cookie(response):
+ response.set_cookie(self.cookie_params["key"], "", expires=0)
+ self.update_response(response)
+ return response
+
+ params = {'returnTo' : self.callback_base_url, 'client_id' : self.client_id}
+ response = redirect(self.client.api_base_url + '/v2/logout?' + urlencode(params))
+ self.update_response(response)
+ return response
+
+ def callback(self):
+ token = self.client.authorize_access_token()
+ id_token = token.get("id_token")
+ oauth_callback_redirect = session.pop("oauth_callback_redirect", "/")
+ resp = redirect(oauth_callback_redirect)
+
+ if self.session_cookie:
+ session[self.CXG_ID_TOKEN] = id_token
+ else:
+ args = self.cookie_params.copy()
+ del args["key"]
+ try:
+ resp.set_cookie(
+ self.cookie_params["key"],
+ id_token,
+ **args)
+ g.token = id_token
+ except Exception as e:
+ raise AuthenticationError(f"unable to set_cookie {self.cookie_params}") from e
+
+ self.update_response(resp)
+ return resp
+
+ def get_login_url(self, data_adaptor):
+ """Return the url for the login route"""
+ if current_app.app_config.is_multi_dataset():
+ return f"/login?dataset={data_adaptor.uri_path}"
+ else:
+ return "/login"
+
+ def get_logout_url(self, data_adaptor):
+ """Return the url for the logout route"""
+ return "/logout"
+
+ def get_token(self):
+ """Function to return the token"""
+ if "token" in g:
+ return g.token
+ if self.session_cookie:
+ g.token = session.get(self.CXG_ID_TOKEN)
+ else:
+ g.token = request.cookies.get(self.cookie_params["key"])
+
+ return g.token
+
+ def get_jwt_payload(self):
+ if not has_request_context():
+ return None
+
+ token = self.get_token()
+ if token is None:
+ return None
+
+ unverified_header = jwt.get_unverified_header(token)
+ rsa_key = {}
+ for key in self.jwks['keys']:
+ if key['kid'] == unverified_header['kid']:
+ rsa_key = {
+ 'kty': key['kty'],
+ 'kid': key['kid'],
+ 'use': key['use'],
+ 'n': key['n'],
+ 'e': key['e']
+ }
+ if rsa_key:
+ try:
+ payload = jwt.decode(
+ token,
+ rsa_key,
+ algorithms=self.algorithms,
+ audience=self.audience,
+ issuer=self.api_base_url + "/"
+ )
+ return payload
+
+ except jwt.JWTError as e:
+ raise AuthenticationError(f"invalid signature: {str(e)}")
+ except jwt.ExpiredSignatureError as e:
+ raise AuthenticationError(f"token expired: {str(e)}")
+ except jwt.JWTClaimsError as e:
+ raise AuthenticationError(f"invalid claims {str(e)}")
+
+ raise AuthenticationError("Unable to find the appropriate key")
+
+
+AuthTypeFactory.register("oauth", AuthTypeOAuth)
diff --git a/server/auth/auth_session.py b/server/auth/auth_session.py
new file mode 100644
index 00000000..95157323
--- /dev/null
+++ b/server/auth/auth_session.py
@@ -0,0 +1,39 @@
+from server.auth.auth import AuthTypeBase, AuthTypeFactory
+from flask import session
+from uuid import uuid4
+
+
+class AuthTypeSession(AuthTypeBase):
+ """Session based authentication. The user is always logged. The user id is a random number
+ associated with the session. This is a good choice for desktop servers."""
+
+ # key in the session token for userid
+ CXGUID = "cxguid"
+
+ def __init__(self, app_config):
+ super().__init__()
+
+ def is_valid_authentication_type(self):
+ return True
+
+ def complete_setup(self, app):
+ pass
+
+ def is_user_authenticated(self):
+ # always authenticated
+ return True
+
+ def get_user_id(self):
+ if self.CXGUID not in session:
+ session[self.CXGUID] = uuid4().hex
+ session.permanent = True
+ return session[self.CXGUID]
+
+ def get_user_name(self):
+ return "anonymous"
+
+ def get_user_email(self):
+ return None
+
+
+AuthTypeFactory.register("session", AuthTypeSession)
diff --git a/server/auth/auth_test.py b/server/auth/auth_test.py
new file mode 100644
index 00000000..e6b0b8e0
--- /dev/null
+++ b/server/auth/auth_test.py
@@ -0,0 +1,72 @@
+from server.auth.auth import AuthTypeClientBase, AuthTypeFactory
+from flask import session, request, redirect, current_app
+
+
+class AuthTypeTest(AuthTypeClientBase):
+ """An authentication type for testing client based logins. When the login route is accessed
+ the user is automatically logged in with a default or configured username"""
+
+ # key in session token with userid and username
+ CXGUID = "cxguid_test"
+ CXGUNAME = "cxguname_test"
+ CXGUEMAIL = "cxguemail_test"
+
+ def __init__(self, app_config):
+ super().__init__()
+ self.user_name = "test_account"
+ self.user_id = "id0001"
+ self.user_email = "test_account@test.com"
+
+ def is_valid_authentication_type(self):
+ return True
+
+ def requires_client_login(self):
+ return True
+
+ def add_url_rules(self, app):
+ app.add_url_rule("/login", "login", self.login, methods=["GET"])
+ app.add_url_rule("/logout", "logout", self.logout, methods=["GET"])
+
+ def complete_setup(self, app):
+ pass
+
+ def is_user_authenticated(self):
+ return self.CXGUID in session
+
+ def get_user_id(self):
+ return session.get(self.CXGUID)
+
+ def get_user_name(self):
+ return session.get(self.CXGUNAME)
+
+ def get_user_email(self):
+ return session.get(self.CXGUEMAIL)
+
+ def login(self):
+ args = request.args
+ return_to = args.get("dataset", "/")
+ session[self.CXGUID] = args.get("userid", self.user_id)
+ session[self.CXGUNAME] = args.get("username", self.user_name)
+ return redirect(return_to)
+
+ def logout(self):
+ session.clear()
+ return_to = request.args.get("dataset", "/")
+ return redirect(return_to)
+
+ def get_login_url(self, data_adaptor):
+ """Return the url for the login route"""
+ if current_app.app_config.is_multi_dataset():
+ return f"/login?dataset={data_adaptor.uri_path}"
+ else:
+ return "/login"
+
+ def get_logout_url(self, data_adaptor):
+ """Return the url for the logout route"""
+ if current_app.app_config.is_multi_dataset():
+ return f"/logout?dataset={data_adaptor.uri_path}"
+ else:
+ return "/logout"
+
+
+AuthTypeFactory.register("test", AuthTypeTest)
diff --git a/server/cli/launch.py b/server/cli/launch.py
index 00900092..d46161d4 100644
--- a/server/cli/launch.py
+++ b/server/cli/launch.py
@@ -1,19 +1,19 @@
import errno
import functools
import logging
-from os import devnull
import sys
import webbrowser
+from os import devnull
import click
from flask_compress import Compress
from flask_cors import CORS
-from server.common.utils import sort_options
-from server.common.errors import DatasetAccessError, ConfigurationError
+from server.app.app import Server
from server.common.app_config import AppConfig
from server.common.default_config import default_config
-from server.app.app import Server
+from server.common.errors import DatasetAccessError, ConfigurationError
+from server.common.utils.utils import sort_options
DEFAULT_CONFIG = AppConfig()
@@ -33,7 +33,7 @@ def annotation_args(func):
multiple=False,
metavar="",
help="CSV file to initialize editing of existing annotations; will be altered in-place. "
- "Incompatible with --annotations-dir.",
+ "Incompatible with --annotations-dir.",
)
@click.option(
"--annotations-dir",
@@ -42,7 +42,7 @@ def annotation_args(func):
multiple=False,
metavar="",
help="Directory of where to save output annotations; filename will be specified in the application. "
- "Incompatible with --annotations-file.",
+ "Incompatible with --annotations-file.",
)
@click.option(
"--experimental-annotations-ontology",
@@ -170,7 +170,7 @@ def server_args(func):
default=DEFAULT_CONFIG.server_config.app__debug,
show_default=True,
help="Run in debug mode. This is helpful for cellxgene developers, "
- "or when you want more information about an error condition.",
+ "or when you want more information about an error condition.",
)
@click.option(
"--verbose",
@@ -203,7 +203,7 @@ def server_args(func):
multiple=True,
metavar="",
help="Additional script files to include in HTML page. If not specified, "
- "no additional script files will be included.",
+ "no additional script files will be included.",
show_default=False,
)
@functools.wraps(func)
@@ -223,7 +223,7 @@ def launch_args(func):
default=DEFAULT_CONFIG.server_config.multi_dataset__dataroot,
metavar="",
help="Enable cellxgene to serve multiple files. Supply path (local directory or URL)"
- " to folder containing H5AD and/or CXG datasets.",
+ " to folder containing H5AD and/or CXG datasets.",
hidden=True,
) # TODO, unhide when dataroot is supported)
@click.argument("datapath", required=False, metavar="")
@@ -307,32 +307,32 @@ class CliLaunchServer(Server):
)
@launch_args
def launch(
- datapath,
- dataroot,
- verbose,
- debug,
- open_browser,
- port,
- host,
- embedding,
- obs_names,
- var_names,
- max_category_items,
- disable_custom_colors,
- diffexp_lfc_cutoff,
- title,
- scripts,
- about,
- disable_annotations,
- annotations_file,
- annotations_dir,
- backed,
- disable_diffexp,
- experimental_annotations_ontology,
- experimental_annotations_ontology_obo,
- experimental_enable_reembedding,
- config_file,
- dump_default_config,
+ datapath,
+ dataroot,
+ verbose,
+ debug,
+ open_browser,
+ port,
+ host,
+ embedding,
+ obs_names,
+ var_names,
+ max_category_items,
+ disable_custom_colors,
+ diffexp_lfc_cutoff,
+ title,
+ scripts,
+ about,
+ disable_annotations,
+ annotations_file,
+ annotations_dir,
+ backed,
+ disable_diffexp,
+ experimental_annotations_ontology,
+ experimental_annotations_ontology_obo,
+ experimental_enable_reembedding,
+ config_file,
+ dump_default_config,
):
"""Launch the cellxgene data viewer.
This web app lets you explore single-cell expression data.
diff --git a/server/cli/prepare.py b/server/cli/prepare.py
index a282c14d..df535db0 100644
--- a/server/cli/prepare.py
+++ b/server/cli/prepare.py
@@ -5,7 +5,7 @@ import pandas as pd
from numpy import ndarray, unique
from scipy.sparse.csc import csc_matrix
-from server.common.utils import sort_options
+from server.common.utils.utils import sort_options
@sort_options
@@ -37,7 +37,7 @@ from server.common.utils import sort_options
default=False,
is_flag=True,
help="Do not run quality control metrics. By default cellxgene runs them "
- "(saved to adata.obs and adata.var; see scanpy.pp.calculate_qc_metrics for details).",
+ "(saved to adata.obs and adata.var; see scanpy.pp.calculate_qc_metrics for details).",
)
@click.option(
"--make-obs-names-unique/--no-make-obs-names-unique",
@@ -53,18 +53,18 @@ from server.common.utils import sort_options
)
@click.help_option("--help", "-h", help="Show this message and exit.")
def prepare(
- data,
- embedding,
- recipe,
- output,
- plotting,
- sparse,
- overwrite,
- set_obs_names,
- set_var_names,
- skip_qc,
- make_obs_names_unique,
- make_var_names_unique,
+ data,
+ embedding,
+ recipe,
+ output,
+ plotting,
+ sparse,
+ overwrite,
+ set_obs_names,
+ set_var_names,
+ skip_qc,
+ make_obs_names_unique,
+ make_var_names_unique,
):
"""
Preprocess data for use with cellxgene.
diff --git a/server/test/test_datasets/pbmc3k.cxg/X/__lock.tdb b/server/common/annotations/__init__.py
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/X/__lock.tdb
rename to server/common/annotations/__init__.py
diff --git a/server/common/annotations/annotations.py b/server/common/annotations/annotations.py
new file mode 100644
index 00000000..855f19ab
--- /dev/null
+++ b/server/common/annotations/annotations.py
@@ -0,0 +1,78 @@
+from abc import ABCMeta, abstractmethod
+
+import fastobo
+import fsspec
+
+from server.common.errors import OntologyLoadFailure
+from server.common.utils.type_conversion_utils import get_schema_type_hint_of_array
+
+
+class Annotations(metaclass=ABCMeta):
+ """ baseclass for annotations, including ontologies"""
+
+ """ our default ontology is the PURL for the Cell Ontology.
+ See http://www.obofoundry.org/ontology/cl.html """
+ DefaultOnotology = "http://purl.obolibrary.org/obo/cl.obo"
+
+ def __init__(self):
+ self.ontology_data = None
+
+ def load_ontology(self, path):
+ """Load and parse ontologies - currently support OBO files only."""
+ if path is None:
+ path = self.DefaultOnotology
+
+ try:
+ with fsspec.open(path) as f:
+ obo = fastobo.iter(f)
+ terms = filter(lambda stanza: type(stanza) is fastobo.term.TermFrame, obo)
+ names = [tag.name for term in terms for tag in term if type(tag) is fastobo.term.NameClause]
+ self.ontology_data = names
+
+ except FileNotFoundError as e:
+ raise OntologyLoadFailure("Unable to find OBO ontology path") from e
+
+ except SyntaxError as e:
+ raise OntologyLoadFailure("Syntax error loading OBO ontology") from e
+
+ except Exception as e:
+ raise OntologyLoadFailure("Error loading OBO file") from e
+
+ def get_schema(self, data_adaptor):
+ schema = []
+ labels = self.read_labels(data_adaptor)
+ if labels is not None and not labels.empty:
+ for col in labels.columns:
+ col_schema = dict(name=col, writable=True)
+ col_schema.update(get_schema_type_hint_of_array(labels[col]))
+ schema.append(col_schema)
+
+ return schema
+
+ @abstractmethod
+ def set_collection(self, name):
+ """set or create a new annotation collection"""
+ pass
+
+ @abstractmethod
+ def read_labels(self, data_adaptor):
+ """Return the labels as a pandas.DataFrame"""
+ pass
+
+ @abstractmethod
+ def write_labels(self, df, data_adaptor):
+ """Write the labels (df) to a persistent storage such that it can later be read"""
+ pass
+
+ def update_parameters(self, parameters, data_adaptor):
+ """Update configuration parameters that describe information about the annotations feature"""
+ params = {}
+ params["annotations"] = True
+
+ if self.ontology_data:
+ params["annotations_cell_ontology_enabled"] = True
+ params["annotations_cell_ontology_terms"] = self.ontology_data
+ else:
+ params["annotations_cell_ontology_enabled"] = False
+
+ parameters.update(params)
diff --git a/server/common/annotations/hosted_tiledb.py b/server/common/annotations/hosted_tiledb.py
new file mode 100644
index 00000000..05b017a4
--- /dev/null
+++ b/server/common/annotations/hosted_tiledb.py
@@ -0,0 +1,113 @@
+import json
+import os
+import re
+import time
+
+import pandas as pd
+import tiledb
+from flask import current_app
+
+from server.common.annotations.annotations import Annotations
+from server.converters.cxgtool import sanitize_keys, generate_schema_hints_and_convert_value_types, cxg_dtype
+from server.db.cellxgene_orm import CellxGeneDataset, Annotation
+
+
+class AnnotationsHostedTileDB(Annotations):
+ CXG_ANNO_COLLECTION = "cxg_anno_collection"
+
+ def __init__(self, directory_path, db):
+ super().__init__()
+ self.db = db
+ self.directory_path = directory_path
+
+ def check_category_names(self, df):
+ sanitize_keys(df.keys().to_list(), False)
+
+ def is_safe_collection_name(self, name):
+ """
+ return true if this is a safe collection name
+ this is ultra conservative. If we want to allow full legal file name syntax,
+ we could look at modules like `pathvalidate`
+ """
+ if name is None:
+ return False
+ return re.match(r"^[\w\-]+$", name) is not None
+
+ def set_collection(self, name):
+ self.CXG_ANNO_COLLECTION = name
+
+ def read_labels(self, data_adaptor):
+ user_id = current_app.auth.get_user_id()
+ dataset_name = data_adaptor.get_location()
+ dataset_id = str(self.db.query(
+ table_args=[CellxGeneDataset],
+ filter_args=[CellxGeneDataset.name == dataset_name]
+ )[0].id)
+
+ annotation_object = self.db.query_for_most_recent(
+ Annotation, [Annotation.user_id == user_id, Annotation.dataset_id == dataset_id]
+ )
+ if annotation_object:
+ df = tiledb.open(annotation_object.tiledb_uri)
+ pandas_df = self.convert_to_pandas_df(df)
+ return pandas_df
+ else:
+ return None
+
+ def convert_to_pandas_df(self, tileDBArray):
+ repr_meta = None
+ index_dims = None
+ if '__pandas_attribute_repr' in tileDBArray.meta:
+ # backwards compatibility... unsure if necessary at this point
+ repr_meta = json.loads(tileDBArray.meta['__pandas_attribute_repr'])
+ if '__pandas_index_dims' in tileDBArray.meta:
+ index_dims = json.loads(tileDBArray.meta['__pandas_index_dims'])
+
+ data = tileDBArray[:]
+ indexes = list()
+
+ for col_name, col_val in data.items():
+ if repr_meta and col_name in repr_meta:
+ new_col = pd.Series(col_val, dtype=repr_meta[col_name])
+ data[col_name] = new_col
+ elif index_dims and col_name in index_dims:
+ new_col = pd.Series(col_val, dtype=index_dims[col_name])
+ data[col_name] = new_col
+ indexes.append(col_name)
+
+ new_df = pd.DataFrame.from_dict(data)
+ if len(indexes) > 0:
+ new_df.set_index(indexes, inplace=True)
+
+ return new_df
+
+ def write_labels(self, df, data_adaptor):
+
+ user_id = current_app.auth.get_user_id()
+ timestamp = time.time()
+ dataset_name = data_adaptor.get_location()
+ dataset_id = self.db.get_or_create_dataset(dataset_name)
+ user_id = self.db.get_or_create_user(user_id)
+
+ uri = f"{self.directory_path}-{dataset_name}-{user_id}-{timestamp}"
+ if uri.startswith("s3://"):
+ pass
+ else:
+ os.makedirs(uri, exist_ok=True)
+ schema_hints, values = generate_schema_hints_and_convert_value_types(df)
+
+ annotation = Annotation(
+ tiledb_uri=uri,
+ user_id=user_id,
+ dataset_id=str(dataset_id),
+ schema_hints=json.dumps(schema_hints)
+ )
+ if not df.empty:
+ self.check_category_names(df)
+ # convert to tiledb datatypes
+ for col in df:
+ df[col] = df[col].astype(cxg_dtype(df[col]))
+ tiledb.from_pandas(uri, df)
+
+ self.db.session.add(annotation)
+ self.db.session.commit()
diff --git a/server/common/annotations.py b/server/common/annotations/local_file_csv.py
similarity index 70%
rename from server/common/annotations.py
rename to server/common/annotations/local_file_csv.py
index 45e53122..b463540b 100644
--- a/server/common/annotations.py
+++ b/server/common/annotations/local_file_csv.py
@@ -1,86 +1,19 @@
-from datetime import datetime
-import re
-from uuid import uuid4
-import os
-import pandas as pd
-from hashlib import blake2b
import base64
-from server import __version__ as cellxgene_version
+import os
+import re
import threading
-from server.common.errors import AnnotationsError, OntologyLoadFailure
-from server.common.utils import series_to_schema
-import fsspec
-import fastobo
-from flask import session
-from abc import ABCMeta, abstractmethod
+from datetime import datetime
+from hashlib import blake2b
+import pandas as pd
+from flask import session, has_request_context, current_app
-class Annotations(metaclass=ABCMeta):
- """ baseclass for annotations, including ontologies"""
-
- """ our default ontology is the PURL for the Cell Ontology.
- See http://www.obofoundry.org/ontology/cl.html """
- DefaultOnotology = "http://purl.obolibrary.org/obo/cl.obo"
-
- def __init__(self):
- self.ontology_data = None
-
- def load_ontology(self, path):
- """Load and parse ontologies - currently support OBO files only."""
- if path is None:
- path = self.DefaultOnotology
-
- try:
- with fsspec.open(path) as f:
- obo = fastobo.iter(f)
- terms = filter(lambda stanza: type(stanza) is fastobo.term.TermFrame, obo)
- names = [tag.name for term in terms for tag in term if type(tag) is fastobo.term.NameClause]
- self.ontology_data = names
-
- except FileNotFoundError as e:
- raise OntologyLoadFailure("Unable to find OBO ontology path") from e
-
- except SyntaxError as e:
- raise OntologyLoadFailure("Syntax error loading OBO ontology") from e
-
- except Exception as e:
- raise OntologyLoadFailure("Error loading OBO file") from e
-
- def get_schema(self, data_adaptor):
- labels = self.read_labels(data_adaptor)
- schema = []
- if labels is not None and not labels.empty:
- for col in labels.columns:
- col_schema = dict(name=col, writable=True)
- col_schema.update(series_to_schema(labels[col]))
- schema.append(col_schema)
-
- return schema
-
- @abstractmethod
- def set_collection(self, name):
- """set or create a new annotation collection"""
- pass
-
- @abstractmethod
- def read_labels(self, data_adaptor):
- """Return the labels as a pandas.DataFrame"""
- pass
-
- @abstractmethod
- def write_labels(self, df, data_adaptor):
- """Write the labels (df) to a persistent storage such that it can later be read"""
- pass
-
- @abstractmethod
- def update_parameters(self, parameters, data_adaptor):
- """Update configuration parameters that describe information about the annotations feature"""
- pass
+from server import __version__ as cellxgene_version
+from server.common.annotations.annotations import Annotations
+from server.common.errors import AnnotationsError
class AnnotationsLocalFile(Annotations):
-
- CXGUID = "cxguid"
CXG_ANNO_COLLECTION = "cxg_anno_collection"
def __init__(self, output_dir, output_file):
@@ -97,7 +30,6 @@ class AnnotationsLocalFile(Annotations):
def is_safe_collection_name(self, name):
"""
return true if this is a safe collection name
-
this is ultra conservative. If we want to allow full legal file name syntax,
we could look at modules like `pathvalidate`
"""
@@ -115,6 +47,10 @@ class AnnotationsLocalFile(Annotations):
return session.get(self.CXG_ANNO_COLLECTION)
def read_labels(self, data_adaptor):
+ if has_request_context():
+ if not current_app.auth.is_user_authenticated():
+ return pd.DataFrame()
+
fname = self._get_filename(data_adaptor)
with self.label_lock:
if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
@@ -159,18 +95,12 @@ class AnnotationsLocalFile(Annotations):
self.last_fname = fname
self.last_labels = df
- def _get_userid(self):
- if self.CXGUID not in session:
- session[self.CXGUID] = uuid4().hex
- session.permanent = True
- return session[self.CXGUID]
-
def _get_userdata_idhash(self, data_adaptor):
"""
Return a short hash that weakly identifies the user and dataset.
Used to create safe annotations output file names.
"""
- uid = self._get_userid()
+ uid = current_app.auth.get_user_id()
id = (uid + data_adaptor.get_location()).encode()
idhash = base64.b32encode(blake2b(id, digest_size=5).digest()).decode("utf-8")
return idhash
@@ -257,8 +187,9 @@ class AnnotationsLocalFile(Annotations):
elif session is not None:
collection = self.get_collection()
- params["annotations-user-data-idhash"] = self._get_userdata_idhash(data_adaptor)
- params["annotations-data-collection-is-read-only"] = False
- params["annotations-data-collection-name"] = collection
+ if current_app.auth.is_user_authenticated():
+ params["annotations-user-data-idhash"] = self._get_userdata_idhash(data_adaptor)
+ params["annotations-data-collection-is-read-only"] = False
+ params["annotations-data-collection-name"] = collection
parameters.update(params)
diff --git a/server/common/app_config.py b/server/common/app_config.py
index 7e8a3635..b3715ecc 100644
--- a/server/common/app_config.py
+++ b/server/common/app_config.py
@@ -1,21 +1,24 @@
-from server import display_version as cellxgene_display_version
-from flatten_dict import flatten, unflatten
-import os
-from os.path import splitext, basename, isdir
-import sys
-from urllib.parse import urlparse, quote_plus
-import yaml
import copy
+import os
+import sys
+import warnings
+from os.path import splitext, basename, isdir
+from urllib.parse import urlparse, quote_plus
+import yaml
+from flatten_dict import flatten, unflatten
+
+import server.compute.diffexp_cxg as diffexp_tiledb
+from server import display_version as cellxgene_display_version
+from server.auth.auth import AuthTypeFactory
+from server.common.annotations.hosted_tiledb import AnnotationsHostedTileDB
+from server.common.annotations.local_file_csv import AnnotationsLocalFile
+from server.common.data_locator import discover_s3_region_name
from server.common.default_config import get_default_config
from server.common.errors import ConfigurationError, DatasetAccessError, OntologyLoadFailure
+from server.common.utils.utils import custom_format_warning, find_available_port, is_port_available
from server.data_common.matrix_loader import MatrixDataLoader, MatrixDataCacheManager, MatrixDataType
-from server.common.utils import find_available_port, is_port_available
-import warnings
-from server.common.annotations import AnnotationsLocalFile
-from server.common.utils import custom_format_warning
-import server.compute.diffexp_cxg as diffexp_tiledb
-from server.common.data_locator import discover_s3_region_name
+from server.db.db_utils import DbUtils
DEFAULT_SERVER_PORT = 5005
# anything bigger than this will generate a special message
@@ -147,7 +150,6 @@ class AppConfig(object):
parameters is done"""
if messagefn is None:
-
def noop(message):
pass
@@ -194,6 +196,7 @@ class AppConfig(object):
server_config = self.server_config
dataset_config = data_adaptor.dataset_config
annotation = dataset_config.user_annotations
+ auth = server_config.auth
# FIXME The current set of config is not consistently presented:
# we have camalCase, hyphen-text, and underscore_text
@@ -240,6 +243,18 @@ class AppConfig(object):
"about_legal_privacy": dataset_config.app__about_legal_privacy,
}
+ # corpora dataset_props
+ # TODO/Note: putting info from the dataset into the /config is not ideal.
+ # However, it is definitely not part of /schema, and we do not have a top-level
+ # route for data properties. Consider creating one at some point.
+ corpora_props = data_adaptor.get_corpora_props()
+ if corpora_props and "default_embedding" in corpora_props:
+ default_embedding = corpora_props["default_embedding"]
+ if isinstance(default_embedding, str) and default_embedding.startswith("X_"):
+ default_embedding = default_embedding[2:] # drop X_ prefix
+ if default_embedding in data_adaptor.get_embedding_names():
+ parameters["default_embedding"] = default_embedding
+
data_adaptor.update_parameters(parameters)
if annotation:
annotation.update_parameters(parameters, data_adaptor)
@@ -252,11 +267,25 @@ class AppConfig(object):
config["library_versions"] = library_versions
config["links"] = links
config["parameters"] = parameters
+ config["corpora_props"] = corpora_props
config["limits"] = {
"column_request_max": server_config.limits__column_request_max,
"diffexp_cellcount_max": server_config.limits__diffexp_cellcount_max,
}
+ if dataset_config.app__authentication_enable and auth.is_valid_authentication_type():
+ config["authentication"] = {
+ "is_authenticated": auth.is_user_authenticated(),
+ "requires_client_login": auth.requires_client_login(),
+ "username": auth.get_user_name(),
+ "user_id": auth.get_user_id()
+ }
+ if auth.requires_client_login():
+ config["authentication"].update({
+ "login": auth.get_login_url(data_adaptor),
+ "logout": auth.get_logout_url(data_adaptor),
+ })
+
return c
@@ -366,6 +395,7 @@ class ServerConfig(BaseConfig):
def __init__(self, app_config, default_config):
dictval_cases = [
("app", "csp_directives"),
+ ("authentication", "params_oauth", "cookie"),
("adaptor", "cxg_adaptor", "tiledb_ctx"),
("multi_dataset", "dataroot"),
]
@@ -384,6 +414,15 @@ class ServerConfig(BaseConfig):
self.app__server_timing_headers = dc["app"]["server_timing_headers"]
self.app__csp_directives = dc["app"]["csp_directives"]
+ self.authentication__type = dc["authentication"]["type"]
+ self.authentication__params_oauth__api_base_url = dc["authentication"]["params_oauth"]["api_base_url"]
+ self.authentication__params_oauth__client_id = dc["authentication"]["params_oauth"]["client_id"]
+ self.authentication__params_oauth__client_secret = dc["authentication"]["params_oauth"]["client_secret"]
+ self.authentication__params_oauth__callback_base_url = \
+ dc["authentication"]["params_oauth"]["callback_base_url"]
+ self.authentication__params_oauth__session_cookie = dc["authentication"]["params_oauth"]["session_cookie"]
+ self.authentication__params_oauth__cookie = dc["authentication"]["params_oauth"]["cookie"]
+
self.multi_dataset__dataroot = dc["multi_dataset"]["dataroot"]
self.multi_dataset__index = dc["multi_dataset"]["index"]
self.multi_dataset__allowed_matrix_types = dc["multi_dataset"]["allowed_matrix_types"]
@@ -414,8 +453,13 @@ class ServerConfig(BaseConfig):
# The matrix data cache manager is created during the complete_config and stored here.
self.matrix_data_cache_manager = None
+ # The authentication object
+ self.auth = None
+
def complete_config(self, context):
self.handle_app(context)
+ self.handle_data_source(context)
+ self.handle_authentication(context)
self.handle_data_locator(context)
self.handle_adaptor(context) # may depend on data_locator
self.handle_single_dataset(context) # may depend on adaptor
@@ -484,6 +528,30 @@ class ServerConfig(BaseConfig):
elif not isinstance(v, str):
raise ConfigurationError("CSP directive value must be a string or list of strings.")
+ def handle_authentication(self, context):
+ self.check_attr("authentication__type", (type(None), str))
+
+ # oauth
+ ptypes = str if self.authentication__type == "oauth" else (type(None), str)
+ self.check_attr("authentication__params_oauth__api_base_url", ptypes)
+ self.check_attr("authentication__params_oauth__client_id", ptypes)
+ self.check_attr("authentication__params_oauth__client_secret", ptypes)
+ self.check_attr("authentication__params_oauth__callback_base_url", (type(None), str))
+ self.check_attr("authentication__params_oauth__session_cookie", bool)
+
+ if self.authentication__params_oauth__session_cookie:
+ self.check_attr("authentication__params_oauth__cookie", (type(None), dict))
+ else:
+ self.check_attr("authentication__params_oauth__cookie", dict)
+ # secret key: first, from CXG_OAUTH_CLIENT_SECRET environment variable
+ # second, from config file
+ self.authentication__params__oauth__client_secret = os.environ.get(
+ "CXG_OAUTH_CLIENT_SECRET", self.authentication__params_oauth__client_secret)
+
+ self.auth = AuthTypeFactory.create(self.authentication__type, self)
+ if self.auth is None:
+ raise ConfigurationError(f"Unknown authentication type: {self.authentication__type}")
+
def handle_data_locator(self, context):
self.check_attr("data_locator__s3__region_name", (type(None), bool, str))
if self.data_locator__s3__region_name is True:
@@ -504,12 +572,9 @@ class ServerConfig(BaseConfig):
region_name = None
self.data_locator__s3__region_name = region_name
- def handle_single_dataset(self, context):
+ def handle_data_source(self, context):
self.check_attr("single_dataset__datapath", (str, type(None)))
- self.check_attr("single_dataset__title", (str, type(None)))
- self.check_attr("single_dataset__about", (str, type(None)))
- self.check_attr("single_dataset__obs_names", (str, type(None)))
- self.check_attr("single_dataset__var_names", (str, type(None)))
+ self.check_attr("multi_dataset__dataroot", (type(None), dict, str))
if self.single_dataset__datapath is None:
if self.multi_dataset__dataroot is None:
@@ -520,6 +585,16 @@ class ServerConfig(BaseConfig):
if self.multi_dataset__dataroot is not None:
raise ConfigurationError("must supply only one of datapath or dataroot")
+ def handle_single_dataset(self, context):
+ self.check_attr("single_dataset__datapath", (str, type(None)))
+ self.check_attr("single_dataset__title", (str, type(None)))
+ self.check_attr("single_dataset__about", (str, type(None)))
+ self.check_attr("single_dataset__obs_names", (str, type(None)))
+ self.check_attr("single_dataset__var_names", (str, type(None)))
+
+ if self.single_dataset__datapath is None:
+ return
+
# create the matrix data cache manager:
if self.matrix_data_cache_manager is None:
self.matrix_data_cache_manager = MatrixDataCacheManager(max_cached=1, timelimit_s=None)
@@ -660,6 +735,7 @@ class DatasetConfig(BaseConfig):
self.app__inline_scripts = dc["app"]["inline_scripts"]
self.app__about_legal_tos = dc["app"]["about_legal_tos"]
self.app__about_legal_privacy = dc["app"]["about_legal_privacy"]
+ self.app__authentication_enable = dc["app"]["authentication_enable"]
self.presentation__max_categories = dc["presentation"]["max_categories"]
self.presentation__custom_colors = dc["presentation"]["custom_colors"]
@@ -670,6 +746,9 @@ class DatasetConfig(BaseConfig):
self.user_annotations__local_file_csv__file = dc["user_annotations"]["local_file_csv"]["file"]
self.user_annotations__ontology__enable = dc["user_annotations"]["ontology"]["enable"]
self.user_annotations__ontology__obo_location = dc["user_annotations"]["ontology"]["obo_location"]
+ self.user_annotations__hosted_tiledb_array__db_uri = dc["user_annotations"]["hosted_tiledb_array"]["db_uri"]
+ self.user_annotations__hosted_tiledb_array__hosted_file_directory = \
+ dc["user_annotations"]["hosted_tiledb_array"]["hosted_file_directory"] # noqa E501
self.embeddings__names = dc["embeddings"]["names"]
self.embeddings__enable_reembedding = dc["embeddings"]["enable_reembedding"]
@@ -696,6 +775,7 @@ class DatasetConfig(BaseConfig):
self.check_attr("app__inline_scripts", list)
self.check_attr("app__about_legal_tos", (type(None), str))
self.check_attr("app__about_legal_privacy", (type(None), str))
+ self.check_attr("app__authentication_enable", bool)
# scripts can be string (filename) or dict (attributes). Convert string to dict.
scripts = []
@@ -719,47 +799,62 @@ class DatasetConfig(BaseConfig):
self.check_attr("user_annotations__local_file_csv__file", (type(None), str))
self.check_attr("user_annotations__ontology__enable", bool)
self.check_attr("user_annotations__ontology__obo_location", (type(None), str))
+ self.check_attr("user_annotations__hosted_tiledb_array__db_uri", (type(None), str))
+ self.check_attr("user_annotations__hosted_tiledb_array__hosted_file_directory", (type(None), str))
if self.user_annotations__enable:
+ server_config = self.app_config.server_config
+ if not self.app__authentication_enable:
+ raise ConfigurationError("user annotations requires authentication to be enabled")
+ if not server_config.auth.is_valid_authentication_type():
+ auth_type = server_config.authentication__type
+ raise ConfigurationError(f"authentication method {auth_type} is not compatible with user annotations")
+
# TODO, replace this with a factory pattern once we have more than one way
# to do annotations. currently only local_file_csv
- if self.user_annotations__type != "local_file_csv":
- raise ConfigurationError('The only annotation type support is "local_file_csv"')
+ if self.user_annotations__type == "local_file_csv":
+ dirname = self.user_annotations__local_file_csv__directory
+ filename = self.user_annotations__local_file_csv__file
- dirname = self.user_annotations__local_file_csv__directory
- filename = self.user_annotations__local_file_csv__file
+ if filename is not None and dirname is not None:
+ raise ConfigurationError("'annotations-file' and 'annotations-dir' may not be used together.")
- if filename is not None and dirname is not None:
- raise ConfigurationError("'annotations-file' and 'annotations-dir' may not be used together.")
+ if filename is not None:
+ lf_name, lf_ext = splitext(filename)
+ if lf_ext and lf_ext != ".csv":
+ raise ConfigurationError(f"annotation file type must be .csv: {filename}")
- if filename is not None:
- lf_name, lf_ext = splitext(filename)
- if lf_ext and lf_ext != ".csv":
- raise ConfigurationError(f"annotation file type must be .csv: {filename}")
+ if dirname is not None and not isdir(dirname):
+ try:
+ os.mkdir(dirname)
+ except OSError:
+ raise ConfigurationError("Unable to create directory specified by --annotations-dir")
- if dirname is not None and not isdir(dirname):
- try:
- os.mkdir(dirname)
- except OSError:
- raise ConfigurationError("Unable to create directory specified by --annotations-dir")
+ self.user_annotations = AnnotationsLocalFile(dirname, filename)
- self.user_annotations = AnnotationsLocalFile(dirname, filename)
-
- # if the user has specified a fixed label file, go ahead and validate it
- # so that we can remove errors early in the process.
- server_config = self.app_config.server_config
- if server_config.single_dataset__datapath and self.user_annotations__local_file_csv__file:
- with server_config.matrix_data_cache_manager.data_adaptor(
- self.tag, server_config.single_dataset__datapath, self.app_config
- ) as data_adaptor:
- data_adaptor.check_new_labels(self.user_annotations.read_labels(data_adaptor))
-
- if self.user_annotations__ontology__enable or self.user_annotations__ontology__obo_location:
- try:
- self.user_annotations.load_ontology(self.user_annotations__ontology__obo_location)
- except OntologyLoadFailure as e:
- raise ConfigurationError("Unable to load ontology terms\n" + str(e))
+ # if the user has specified a fixed label file, go ahead and validate it
+ # so that we can remove errors early in the process.
+ server_config = self.app_config.server_config
+ if server_config.single_dataset__datapath and self.user_annotations__local_file_csv__file:
+ with server_config.matrix_data_cache_manager.data_adaptor(
+ self.tag, server_config.single_dataset__datapath, self.app_config
+ ) as data_adaptor:
+ data_adaptor.check_new_labels(self.user_annotations.read_labels(data_adaptor))
+ if self.user_annotations__ontology__enable or self.user_annotations__ontology__obo_location:
+ try:
+ self.user_annotations.load_ontology(self.user_annotations__ontology__obo_location)
+ except OntologyLoadFailure as e:
+ raise ConfigurationError("Unable to load ontology terms\n" + str(e))
+ elif self.user_annotations__type == "hosted_tiledb_array":
+ self.check_attr("user_annotations__hosted_tiledb_array__db_uri", str)
+ self.check_attr("user_annotations__hosted_tiledb_array__hosted_file_directory", str)
+ self.user_annotations = AnnotationsHostedTileDB(
+ directory_path=self.user_annotations__hosted_tiledb_array__hosted_file_directory,
+ db=DbUtils(self.user_annotations__hosted_tiledb_array__db_uri),
+ )
+ else:
+ raise ConfigurationError('The only annotation type support is "local_file_csv" or "hosted_tiledb_array')
else:
if self.user_annotations__type == "local_file_csv":
dirname = self.user_annotations__local_file_csv__directory
@@ -782,12 +877,15 @@ class DatasetConfig(BaseConfig):
self.check_attr("embeddings__names", list)
self.check_attr("embeddings__enable_reembedding", bool)
- if self.app_config.server_config.single_dataset__datapath:
+ server_config = self.app_config.server_config
+ if server_config.single_dataset__datapath:
if self.embeddings__enable_reembedding:
- matrix_data_loader = MatrixDataLoader(self.single_dataset__datapath, app_config=self.app_config)
- if matrix_data_loader.matrix_data_type() != MatrixDataType.H5AD:
+ matrix_data_loader = MatrixDataLoader(
+ server_config.single_dataset__datapath, app_config=self.app_config
+ )
+ if matrix_data_loader.matrix_data_type != MatrixDataType.H5AD:
raise ConfigurationError("'enable-reembedding is only supported with H5AD files.")
- if self.adaptor__anndata_adaptor__backed:
+ if server_config.adaptor__anndata_adaptor__backed:
raise ConfigurationError("enable-reembedding is not supported when run in --backed mode.")
def handle_diffexp(self, context):
@@ -798,7 +896,7 @@ class DatasetConfig(BaseConfig):
server_config = self.app_config.server_config
if server_config.single_dataset__datapath:
with server_config.matrix_data_cache_manager.data_adaptor(
- self.tag, server_config.single_dataset__datapath, self.app_config
+ self.tag, server_config.single_dataset__datapath, self.app_config
) as data_adaptor:
if self.diffexp__enable and data_adaptor.parameters.get("diffexp_may_be_slow", False):
context["messagefn"](
diff --git a/server/common/aws_secret_utils.py b/server/common/aws_secret_utils.py
new file mode 100644
index 00000000..47f7690b
--- /dev/null
+++ b/server/common/aws_secret_utils.py
@@ -0,0 +1,84 @@
+import logging
+import os
+import sys
+
+import boto3
+from flask import json
+
+from server.common.data_locator import discover_s3_region_name
+from server.common.errors import SecretKeyRetrievalError
+
+
+def handle_config_from_secret(app_config):
+ """Update configuration from the secret manager"""
+ secret_name = os.getenv("CXG_AWS_SECRET_NAME")
+ if not secret_name:
+ return
+
+ # need to find the secret manager region.
+ # 1. from CXG_AWS_SECRET_REGION_NAME
+ # 2. discover from dataroot location (if on s3)
+ # 3. discover from config file location (if on s3)
+ secret_region_name = os.getenv("CXG_AWS_SECRET_REGION_NAME")
+ if secret_region_name is None:
+ secret_region_name = discover_s3_region_name(app_config.multi_dataset__dataroot)
+ if not secret_region_name:
+ from server.eb.app import config_file
+ secret_region_name = discover_s3_region_name(config_file)
+ if not secret_region_name:
+ logging.error("Could not determine the AWS Secret Manager region")
+ sys.exit(1)
+
+ secrets = get_secret_key(secret_region_name, secret_name)
+
+ if not secrets:
+ return
+
+ server_attrs = (
+ ("flask_secret_key", "app__flask_secret_key"),
+ ("oauth_client_secret", "authentication__params_oauth__client_secret"),
+ )
+ default_dataset_attrs = (
+ ("db_uri", "user_annotations__hosted_tiledb_array__db_uri"),
+ )
+
+ # update server configuration attributes
+ for key, attr in server_attrs:
+ cur_val = getattr(app_config.server_config, attr)
+ if cur_val:
+ continue
+
+ # replace the attr with the secret if it is not set
+ val = secrets.get(key)
+ if val:
+ logging.info(f"set {attr} from secret")
+ app_config.update_server_config(**{attr : val})
+
+ # update default dataset configuration attributes
+ for key, attr in default_dataset_attrs:
+ cur_val = getattr(app_config.default_dataset_config, attr)
+ if cur_val:
+ continue
+
+ # replace the attr with the secret if it is not set
+ val = secrets.get(key)
+ if val:
+ logging.info(f"set {attr} from secret")
+ app_config.update_default_dataset_config(**{attr : val})
+
+
+def get_secret_key(region_name, secret_name):
+ session = boto3.session.Session()
+ client = session.client(service_name="secretsmanager", region_name=region_name)
+
+ try:
+ get_secret_value_response = client.get_secret_value(SecretId=secret_name)
+ if "SecretString" in get_secret_value_response:
+ var = get_secret_value_response["SecretString"]
+ secret = json.loads(var)
+ return secret
+ except Exception as e:
+ logging.critical(f"Caught exception during get_secret_key, {e}", exc_info=True)
+ raise SecretKeyRetrievalError
+
+ return None
diff --git a/server/common/corpora.py b/server/common/corpora.py
new file mode 100644
index 00000000..9f48873b
--- /dev/null
+++ b/server/common/corpora.py
@@ -0,0 +1,90 @@
+"""
+Corpora schema conventions support. Helper functions for reading.
+
+https://github.com/chanzuckerberg/corpora-data-portal/blob/main/backend/schema/corpora_schema.md
+
+https://github.com/chanzuckerberg/corpora-data-portal/blob/main/backend/schema/corpora_schema_h5ad_implementation.md
+"""
+import collections
+import json
+
+from server.cli.upgrade import validate_version_str
+
+
+def corpora_get_versions_from_anndata(adata):
+ """
+ Given an AnnData object, return:
+ * None - if not a Corpora object
+ * [ corpora_schema_version, corpora_encoding_version ] - if a Corpora object
+
+ Implements the identification protocol defined in the specification.
+ """
+
+ # per Corpora AnnData spec, this is a corpora file if the following is true
+ if "version" not in adata.uns_keys():
+ return None
+ version = adata.uns["version"]
+ if not isinstance(version, collections.abc.Mapping) or "corpora_schema_version" not in version:
+ return None
+
+ corpora_schema_version = version.get("corpora_schema_version")
+ corpora_encoding_version = version.get("corpora_encoding_version")
+
+ # TODO: spec says these must be SEMVER values, so check.
+ if validate_version_str(corpora_schema_version) and validate_version_str(corpora_encoding_version):
+ return [corpora_schema_version, corpora_encoding_version]
+
+
+def corpora_is_version_supported(corpora_schema_version, corpora_encoding_version):
+ return (
+ corpora_schema_version
+ and corpora_encoding_version
+ and corpora_schema_version.startswith("1.")
+ and corpora_encoding_version.startswith("0.1.")
+ )
+
+
+def corpora_get_props_from_anndata(adata):
+ """
+ Get Corpora dataset properties from an AnnData
+ """
+ versions = corpora_get_versions_from_anndata(adata)
+ if versions is None:
+ return None
+ [corpora_schema_version, corpora_encoding_version] = versions
+ version_is_supported = corpora_is_version_supported(corpora_schema_version, corpora_encoding_version)
+ if not version_is_supported:
+ raise ValueError("Unsupported Corpora schema version")
+
+ required_simple_fields = [
+ "version",
+ "title",
+ "layer_descriptions",
+ "organism",
+ "organism_ontology_term_id",
+ "project_name",
+ "project_description",
+ ]
+ # Spec says some values encoded as JSON due to the inability of AnnData to store complex types.
+ required_json_fields = ["contributors", "project_links"]
+ optional_simple_fields = ["preprint_doi", "publication_doi", "default_embedding", "default_field", "tags"]
+
+ corpora_props = {}
+ for key in required_simple_fields:
+ if key not in adata.uns:
+ raise KeyError(f"missing Corpora schema field {key}")
+ corpora_props[key] = adata.uns[key]
+
+ for key in required_json_fields:
+ if key not in adata.uns:
+ raise KeyError(f"missing Corpora schema field {key}")
+ try:
+ corpora_props[key] = json.loads(adata.uns[key])
+ except json.JSONDecodeError:
+ raise json.JSONDecodeError(f"Corpora schema field {key} is expected to be a valid JSON string")
+
+ for key in optional_simple_fields:
+ if key in adata.uns:
+ corpora_props[key] = adata.uns[key]
+
+ return corpora_props
diff --git a/server/common/default_config.py b/server/common/default_config.py
index 0b0c5118..f473dd9f 100644
--- a/server/common/default_config.py
+++ b/server/common/default_config.py
@@ -5,7 +5,7 @@ server:
app:
verbose: false
debug: false
- host: "127.0.0.1"
+ host: localhost
port : null
open_browser: false
force_https: false
@@ -14,6 +14,35 @@ server:
server_timing_headers: false
csp_directives: null
+ authentication:
+ # The authentication types may be "none", "session", "oauth"
+ # none: No authentication support, features like user_annotations must not be enabled.
+ # session: A session based userid is automatically generated. (no params needed)
+ # oauth: oauth2 is used for authentication; parameters are defined in params_oauth.
+ type: session
+
+ params_oauth:
+ # url to the auth server
+ api_base_url: null
+ # client_id of this app
+ client_id: null
+ # the client_secret known to the auth server and this app
+ client_secret: null
+ # cellxgene server location;
+ # the browser will be redirected to locations relative to this location during login and logout.
+ # A value of None, indicates the client and server are on the localhost. http://localhost: will be used.
+ callback_base_url: null
+
+ # if true, the jwt containing the id_token is stored in a session cookie
+ session_cookie: true
+
+ # if session_cookie is false, then a regular cookie will be used. In that case
+ # the cookie will be defined by a dictionary of parameters.
+ # The keys of the dictionary match the parameters of the flask set_cookie api
+ # (https://flask.palletsprojects.com/en/1.1.x/api/), and with the same meaning.
+ # legal keys: key, max_age, expires, path, domain, secure, httponly, and samesite.
+ cookie: null
+
multi_dataset:
# If dataroot is set, then cellxgene may serve multiple datasets. This parameter is not
# compatible with single_dataset/datapath.
@@ -132,6 +161,9 @@ dataset:
about_legal_tos: null
about_legal_privacy: null
+ # allow authentication support
+ authentication_enable: true
+
presentation:
max_categories: 1000
custom_colors: true
@@ -139,6 +171,9 @@ dataset:
user_annotations:
enable: true
type: local_file_csv
+ hosted_tiledb_array:
+ db_uri: null
+ hosted_file_directory: null
local_file_csv:
directory: null
file: null
diff --git a/server/common/errors.py b/server/common/errors.py
index d1dda2cf..5e8281e6 100644
--- a/server/common/errors.py
+++ b/server/common/errors.py
@@ -1,85 +1,57 @@
from http import HTTPStatus
-class RequestException(Exception):
+class CellxgeneException(Exception):
+ """Base class for cellxgene exceptions"""
+
+ def __init__(self, message):
+ self.message = message
+ super().__init__(message)
+
+
+class RequestException(CellxgeneException):
"""Baseclass for exceptions that can be raised from a request."""
# The default status code is 400 (Bad Request)
default_status_code = HTTPStatus.BAD_REQUEST
def __init__(self, message, status_code=None):
- Exception.__init__(self)
- self.message = message
+ super().__init__(message)
self.status_code = status_code or self.default_status_code
-class FilterError(RequestException):
- """Raised when filter is malformed"""
-
- pass
+def define_exception(name, doc):
+ globals()[name] = type(name, (CellxgeneException,), dict(__doc__=doc))
-class JSONEncodingValueError(RequestException):
- """Raised when data cannot be encoded into json"""
-
- pass
+def define_request_exception(name, doc, default_status_code=HTTPStatus.BAD_REQUEST):
+ globals()[name] = type(name, (RequestException,), dict(__doc__=doc, default_status_code=default_status_code))
-class MimeTypeError(RequestException):
- """Raised when incompatible MIME type selected"""
+define_request_exception("FilterError", "Raised when filter is malformed")
+define_request_exception("JSONEncodingValueError", "Raised when data cannot be encoded into json")
+define_request_exception("MimeTypeError", "Raised when incompatible MIME type selected")
+define_request_exception("DatasetAccessError", "Raised when file loaded into a DataAdaptor is misformatted")
+define_request_exception("DisabledFeatureError", "Raised when an attempt to use a disabled feature occurs")
+define_request_exception("AnnotationsError", "Raised when an attempt to use the annotations feature fails")
+define_request_exception(
+ "ComputeError",
+ "Raised when an error occurs during a compute algorithm (such as diffexp)",
+ HTTPStatus.INTERNAL_SERVER_ERROR,
+)
+define_request_exception("ExceedsLimitError", "Raised when an HTTP request exceeds a limit/quota")
+define_request_exception("ColorFormatException", "Raised when color helper functions encounter an unknown color format")
+define_request_exception(
+ "AuthenticationError",
+ "Raised when there is an authentication error",
+ default_status_code=HTTPStatus.UNAUTHORIZED)
- pass
+define_request_exception(
+ "AnnotationCategoryNameError",
+ "Raised when an annotation category name cant be saved",
+ default_status_code=HTTPStatus.UNPROCESSABLE_ENTITY)
-
-class DatasetAccessError(RequestException):
- """Raised when file loaded into a DataAdaptor is misformatted"""
-
- pass
-
-
-class DisabledFeatureError(RequestException):
- """Raised when an attempt to use a disabled feature occurs"""
-
- pass
-
-
-class AnnotationsError(RequestException):
- """Raised when an attempt to use the annotations feature fails"""
-
- pass
-
-
-class ComputeError(RequestException):
- """Raised when an error occurs during a compute algorithm (such as diffexp)"""
-
- default_status_code = HTTPStatus.INTERNAL_SERVER_ERROR
-
-
-class ExceedsLimitError(RequestException):
- """Raised when an HTTP request exceeds a limit/quota"""
-
- pass
-
-
-class ColorFormatException(RequestException):
- """Raised when color helper functions encounter an unknown color format"""
-
- pass
-
-
-class OntologyLoadFailure(Exception):
- """Raised when reading the ontology file fails"""
-
- pass
-
-
-class ConfigurationError(Exception):
- """Raised when checking configuration errors"""
-
- pass
-
-
-class PrepareError(Exception):
- """Raised when data is misprepared"""
-
- pass
+define_exception("OntologyLoadFailure", "Raised when reading the ontology file fails")
+define_exception("ConfigurationError", "Raised when checking configuration errors")
+define_exception("PrepareError", "Raised when data is misprepared")
+define_exception("SecretKeyRetrievalError", "Raised when get_secret_key from AWS fails")
diff --git a/server/common/rest.py b/server/common/rest.py
index a1e1a9ae..b2e30305 100644
--- a/server/common/rest.py
+++ b/server/common/rest.py
@@ -301,13 +301,9 @@ def layout_obs_get(request, data_adaptor):
def layout_obs_put(request, data_adaptor):
- if not data_adaptor.dataset_config.embedding__enable_reembedding:
+ if not data_adaptor.dataset_config.embeddings__enable_reembedding:
return abort(HTTPStatus.NOT_IMPLEMENTED)
- preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
- if preferred_mimetype != "application/octet-stream":
- return abort(HTTPStatus.NOT_ACCEPTABLE)
-
args = request.get_json()
filter = args["filter"] if args else None
if not filter:
@@ -315,17 +311,9 @@ def layout_obs_put(request, data_adaptor):
method = args["method"] if args else "umap"
try:
- schema, fbs = data_adaptor.compute_embedding(method, filter)
- return make_response(
- fbs,
- HTTPStatus.OK,
- {
- "Content-Type": "application/octet-stream",
- "CxG-Schema": json.dumps(schema),
- "Access-Control-Expose-Headers": "CxG-Schema",
- },
- )
+ schema = data_adaptor.compute_embedding(method, filter)
+ return make_response(jsonify(schema), HTTPStatus.OK, {"Content-Type": "application/json"})
except NotImplementedError as e:
- return abort_and_log(HTTPStatus.NOT_IMPLEMENTED, str(e), include_exc_info=True)
+ return abort_and_log(HTTPStatus.NOT_IMPLEMENTED, str(e))
except (ValueError, DisabledFeatureError, FilterError) as e:
return abort_and_log(HTTPStatus.BAD_REQUEST, str(e), include_exc_info=True)
diff --git a/server/test/test_datasets/pbmc3k.cxg/__tiledb_group.tdb b/server/common/utils/__init__.py
old mode 100755
new mode 100644
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/__tiledb_group.tdb
rename to server/common/utils/__init__.py
diff --git a/server/common/utils/matrix_utils.py b/server/common/utils/matrix_utils.py
new file mode 100644
index 00000000..3eeddc10
--- /dev/null
+++ b/server/common/utils/matrix_utils.py
@@ -0,0 +1,112 @@
+import logging
+
+import numpy as np
+from scipy.stats import mode
+
+
+def is_matrix_sparse(matrix: np.ndarray, sparse_threshold):
+ """
+ Returns whether `matrix` is sparse or not (i.e. dense). This is determined by figuring out whether the matrix has
+ a sparsity percentage below the sparse_threshold, returning the number of non-zeros encountered and number of
+ elements evaluated. This function may return before evaluating the whole matrix if it can be determined that matrix
+ is not sparse enough.
+ """
+
+ if sparse_threshold == 100.0:
+ return True
+ if sparse_threshold == 0.0:
+ return False
+
+ total_number_of_rows = matrix.shape[0]
+ total_number_of_columns = matrix.shape[1]
+ total_number_of_matrix_elements = total_number_of_rows * total_number_of_columns
+
+ # For efficiency, we count the number of non-zero elements in chunks of the matrix at a time until we hit the
+ # maximum number of non zero values allowed before the matrix is deemed "dense." This allows the function the
+ # quit early for large dense matrices.
+ row_stride = min(int(np.power(10, np.around(np.log10(1e9 / total_number_of_columns)))), 10_000)
+
+ maximum_number_of_non_zero_elements_in_matrix = int(
+ total_number_of_rows * total_number_of_columns * sparse_threshold / 100
+ )
+ number_of_non_zero_elements = 0
+
+ for start_row_index in range(0, total_number_of_rows, row_stride):
+ end_row_index = min(start_row_index + row_stride, total_number_of_rows)
+
+ matrix_subset = matrix[start_row_index:end_row_index, :]
+ if not isinstance(matrix_subset, np.ndarray):
+ matrix_subset = matrix_subset.toarray()
+
+ number_of_non_zero_elements += np.count_nonzero(matrix_subset)
+ if number_of_non_zero_elements > maximum_number_of_non_zero_elements_in_matrix:
+ if end_row_index != total_number_of_rows:
+ percentage_of_non_zero_elements = 100 * number_of_non_zero_elements / (
+ end_row_index * total_number_of_columns)
+ logging.info(
+ f"Matrix is not sparse. Percentage of non-zero elements (estimate): "
+ f"{percentage_of_non_zero_elements:6.2f}")
+ else:
+ percentage_of_non_zero_elements = 100 * number_of_non_zero_elements / total_number_of_matrix_elements
+ logging.info(
+ f"Matrix is not sparse. Percentage of non-zero elements (exact): "
+ f"{percentage_of_non_zero_elements:6.2f}")
+ return False
+
+ is_sparse = (100.0 * number_of_non_zero_elements / total_number_of_matrix_elements) < sparse_threshold
+ return is_sparse
+
+
+def get_column_shift_encode_for_matrix(matrix, sparse_threshold):
+ """
+ Returns a column shift if there is a column shift that allows the given matrix to be considered as sparse. Column
+ shift encoding works by taking the most common value in each column, then subtracting that value from each element
+ of the column. If each column mostly contains its most common value, then the resulting matrix can be very sparse.
+
+ This function determines if column shift encoding can be used to transform the matrix into a sparse matrix with a
+ sparsity below the sparse_threshold. If so, returns the array that stores this encoding. This function also returns
+ the number of non-zeros encountered and number of elements evaluated. This function may return before evaluating
+ the whole matrix if it can be determined that the matrix cannot benefit from column shift encoding.
+ """
+
+ total_number_of_rows = matrix.shape[0]
+ total_number_of_columns = matrix.shape[1]
+ total_number_of_matrix_elements = total_number_of_rows * total_number_of_columns
+
+ stride = max(1, 128_000_000 // total_number_of_rows)
+ column_shift = np.zeros(total_number_of_columns)
+
+ maximum_number_of_non_zero_elements_in_matrix = int(
+ total_number_of_rows * total_number_of_columns * sparse_threshold / 100
+ )
+ number_of_non_zero_elements = 0
+
+ for start_column_index in range(0, total_number_of_columns, stride):
+ end_column_index = min(start_column_index + stride, total_number_of_columns)
+
+ matrix_subset = matrix[:, start_column_index:end_column_index]
+ if not isinstance(matrix_subset, np.ndarray):
+ matrix_subset = matrix_subset.toarray()
+
+ matrix_subset_mode = mode(matrix_subset)
+
+ column_shift[start_column_index:end_column_index] = matrix_subset_mode.mode
+ number_of_non_zero_elements += total_number_of_rows * (end_column_index - start_column_index) - np.sum(
+ matrix_subset_mode.count
+ )
+
+ if number_of_non_zero_elements > maximum_number_of_non_zero_elements_in_matrix:
+ if end_column_index != total_number_of_columns:
+ logging.info(
+ "Matrix is not sparse even with column shift. Percentage of non-zero elements (estimate): %6.2f"
+ % (100 * number_of_non_zero_elements / end_column_index * total_number_of_rows)
+ )
+ else:
+ logging.info(
+ "Matrix is not sparse even with column shift. Percentage of non-zero elements (exact): %6.2f"
+ % (100 * number_of_non_zero_elements / total_number_of_matrix_elements)
+ )
+ return None
+
+ is_sparse = (100.0 * number_of_non_zero_elements / total_number_of_matrix_elements) < sparse_threshold
+ return column_shift if is_sparse else None
diff --git a/server/common/utils/sanitization_utils.py b/server/common/utils/sanitization_utils.py
new file mode 100644
index 00000000..af6f299f
--- /dev/null
+++ b/server/common/utils/sanitization_utils.py
@@ -0,0 +1,40 @@
+import re
+
+
+def sanitize_values_in_list(list_of_keys: list):
+ """
+ Returns a dictionary mapping of the old keys in the list of `list_of_keys` to its new, clean name that is both
+ safe and unique.
+ """
+
+ if not all([isinstance(key, str) for key in list_of_keys]):
+ raise Exception("List of keys to sanitize must contain all strings.")
+
+ # Mask out [~/.] and anything outside the ASCII range.
+ mask = re.compile(r"[^ -\-0-\[\]-\}]")
+ clean_keys_list = [mask.sub("_", key) for key in list_of_keys]
+
+ # Dedupe the clean keys list
+ deduped_clean_keys_list = []
+ for index, clean_key in enumerate(clean_keys_list):
+ total_occurrences_of_clean_key = clean_keys_list.count(clean_key)
+ total_occurrences_up_until_current_index = clean_keys_list[:index].count(clean_key)
+ deduped_clean_keys_list.append(
+ clean_key + "_" + str(total_occurrences_up_until_current_index + 1)
+ if total_occurrences_of_clean_key > 1
+ else clean_key
+ )
+
+ return dict(zip(list_of_keys, deduped_clean_keys_list))
+
+
+def sanitize_keys_in_dictionary(dict_to_sanitize: dict):
+ """
+ Clean and dedupe the keys in the given dictionary.
+ """
+
+ clean_keys = sanitize_values_in_list(dict_to_sanitize.keys())
+ for original_key, sanitized_key in clean_keys.items():
+ if original_key != sanitized_key:
+ dict_to_sanitize[sanitized_key] = dict_to_sanitize[original_key]
+ del dict_to_sanitize[original_key]
diff --git a/server/common/utils/type_conversion_utils.py b/server/common/utils/type_conversion_utils.py
new file mode 100644
index 00000000..8b467b47
--- /dev/null
+++ b/server/common/utils/type_conversion_utils.py
@@ -0,0 +1,93 @@
+import logging
+
+import numpy as np
+import pandas as pd
+
+
+def get_dtype_of_array(array: pd.Series):
+ return get_dtype_and_schema_of_array(array)[0]
+
+
+def get_schema_type_hint_of_array(array: pd.Series):
+ return get_dtype_and_schema_of_array(array)[1]
+
+
+def get_dtype_and_schema_of_array(array: pd.Series):
+ return (get_dtype_from_dtype(array.dtype, array_values=array),
+ get_schema_type_hint_from_dtype(array.dtype, array_values=array))
+
+
+def get_dtype_from_dtype(dtype, array_values=None):
+ """
+ Given a data type, finds the equivalent data type that the array should be encoded as. Notably, this is relevant
+ for 64 bit values which will get downcast to 32 bit.
+ """
+
+ dtype_name = dtype.name
+ dtype_kind = dtype.kind
+
+ if dtype == np.float32 or dtype == np.int32:
+ return dtype
+ if dtype_name == "bool":
+ return np.uint8
+ if dtype_name == "object" and dtype_kind == "O":
+ return np.unicode
+ if dtype_name == "category":
+ return get_dtype_from_dtype(dtype.categories.dtype, dtype.categories)
+
+ if can_cast_to_float32(dtype):
+ return np.float32
+ if can_cast_to_int32(dtype, array_values):
+ return np.int32
+
+ raise TypeError(f"Annotations of type {dtype} are unsupported.")
+
+
+def get_schema_type_hint_from_dtype(dtype, array_values=None):
+ """
+ Returns a dictionary that contains type hints about the data type given, especially if the data type is 64 bit
+ and will be downcast to 32 bit.
+ """
+
+ dtype_name = dtype.name
+ dtype_kind = dtype.kind
+
+ if dtype == np.float32 or dtype == np.int32:
+ return {"type": dtype_name}
+ if dtype_name == "bool":
+ return {"type": "boolean"}
+ if dtype_name == "object" and dtype_kind == "O":
+ return {"type": "string"}
+ if dtype_name == "category":
+ return {"type": "categorical", "categories": dtype.categories.tolist()}
+
+ if can_cast_to_float32(dtype):
+ return {"type": "float32"}
+ if can_cast_to_int32(dtype, array_values):
+ return {"type": "int32"}
+
+ raise TypeError(f"Annotations of type {dtype} are unsupported.")
+
+
+def can_cast_to_float32(dtype):
+ if dtype.kind == "f":
+ if not np.can_cast(dtype, np.float32):
+ logging.warning(f"Type {dtype.name} will be converted to 32 bit float and may lose precision.")
+ return True
+ return False
+
+
+def can_cast_to_int32(dtype, array_values=None):
+ """
+ A type can be cast to 32 bit, overriding the numpy `cast_cast` function if the values in the array that are of
+ the higher precision type has values that are entirely within the range of the downcast type.
+ """
+
+ if dtype.kind in ["i", "u"]:
+ if np.can_cast(dtype, np.int32):
+ return True
+ ii32 = np.iinfo(np.int32)
+ if not array_values.empty and (
+ array_values.min() >= ii32.min and array_values.max() <= ii32.max) or array_values.empty:
+ return True
+ return False
diff --git a/server/common/utils.py b/server/common/utils/utils.py
similarity index 69%
rename from server/common/utils.py
rename to server/common/utils/utils.py
index 7e876194..c6bb24f0 100644
--- a/server/common/utils.py
+++ b/server/common/utils/utils.py
@@ -5,12 +5,11 @@ import logging
import os
import pkgutil
import socket
-import warnings
-
-from flask import json
from urllib.parse import urlsplit, urljoin
+
import numpy as np
-import pandas as pd
+from flask import json
+
from server.common.errors import ConfigurationError
@@ -94,61 +93,6 @@ def jsonify_numpy(data):
return json.dumps(data, cls=Float32JSONEncoder, allow_nan=False)
-def dtype_to_schema(dtype):
- schema = {}
- if dtype == np.float32:
- schema["type"] = "float32"
- elif dtype == np.int32:
- schema["type"] = "int32"
- elif dtype == np.bool_:
- schema["type"] = "boolean"
- elif dtype == np.str:
- schema["type"] = "string"
- elif dtype == "category":
- schema["type"] = "categorical"
- schema["categories"] = dtype.categories.tolist()
- else:
- raise TypeError(f"Annotations of type {dtype} are unsupported.")
- return schema
-
-
-def can_cast_to_float32(array):
- if array.dtype.kind == "f":
- if not np.can_cast(array.dtype, np.float32):
- warnings.warn(f"Annotation {array.name} will be converted to 32 bit float and may lose precision.")
- return True
- return False
-
-
-def can_cast_to_int32(array):
- if array.dtype.kind in ["i", "u"]:
- if np.can_cast(array.dtype, np.int32):
- return True
- ii32 = np.iinfo(np.int32)
- if array.min() >= ii32.min and array.max() <= ii32.max:
- return True
- return False
-
-
-def series_to_schema(array):
- assert type(array) == pd.Series
- try:
- return dtype_to_schema(array.dtype)
- except TypeError:
- dtype = array.dtype
- data_kind = dtype.kind
- schema = {}
- if can_cast_to_float32(array):
- schema["type"] = "float32"
- elif can_cast_to_int32(array):
- schema["type"] = "int32"
- elif data_kind == "O" and dtype == "object":
- schema["type"] = "string"
- else:
- raise TypeError(f"Annotations of type {dtype} are unsupported.")
- return schema
-
-
def import_plugins(plugin_module):
"""
Load optional plugin modules from server.common.plugins
diff --git a/server/compute/scanpy.py b/server/compute/scanpy.py
index cebafabd..36c2e1a9 100644
--- a/server/compute/scanpy.py
+++ b/server/compute/scanpy.py
@@ -1,4 +1,5 @@
import importlib
+import numpy as np
"""
Wrapper for various scanpy modules. Will raise NotImplementedError if the scanpy
@@ -11,8 +12,8 @@ def get_scanpy_module():
sc = importlib.import_module("scanpy")
# Future: we could enforce versions here, eg, lookat sc.__version__
return sc
- except ModuleNotFoundError:
- raise NotImplementedError("Please install scanpy to enable UMAP re-embedding")
+ except ModuleNotFoundError as e:
+ raise NotImplementedError("Please install scanpy to enable UMAP re-embedding") from e
except Exception as e:
# will capture other ImportError corner cases
raise NotImplementedError() from e
@@ -46,4 +47,7 @@ def scanpy_umap(adata, obs_mask=None, pca_options={}, neighbors_options={}, umap
sc.pp.neighbors(adata, **neighbors_options)
sc.tl.umap(adata, **umap_options)
- return adata.obsm["X_umap"]
+ umap = adata.obsm["X_umap"]
+ result = np.full((obs_mask.shape[0], umap.shape[1]), np.NaN)
+ result[obs_mask] = umap
+ return result
diff --git a/server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__lock.tdb b/server/converters/__init__.py
old mode 100755
new mode 100644
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__lock.tdb
rename to server/converters/__init__.py
diff --git a/server/converters/cxgtool.py b/server/converters/cxgtool.py
index 0b8fb3a4..5116c4e6 100644
--- a/server/converters/cxgtool.py
+++ b/server/converters/cxgtool.py
@@ -1,70 +1,9 @@
"""
This program converts an [AnnData H5AD](https://anndata.readthedocs.io/en/stable/)
-into a cellxgene TileDB structure, aka a 'CXG'.
-
-The organization of the TileDB structure is:
-
- the.cxg TileDB Group
- ├─ obs TileDB array containing cell (row) attributes, one attribute per
- │ dataframe column, shape (n_obs,)
- ├─ var TileDB array containing gene (column) attributes, with one attribute per
- │ dataframe column, shape (n_var,)
- ├─ X Main count matrix as a 2D TileDB array, single unnamed numeric attribute
- ├─ X_col_shift TilebDB Array used in column shift encoding, shape (n_var,), dtype = X.dtype.
- │ Single unnamed numeric attribute. If this array is sparse, and X_col_shift exists,
- │ then all values in the i'th column were subtracted by X_col_shift[i].
- ├─ emb TileDB group, storing optional embeddings (group may be empty)
- │ └─ TileDB Array, single anon attribute, ND numeric array, shape (n_obs, N)
- └─ cxg_group_metadata Empty array used only to stash metadata about the overall object.
- └─ cxg_category_colors CXG colors object as described below:
- {
- "": {
- "": "",
- ...
- },
- ...
- }
- ...
-
-All arrays are defined to have a uint32 domain, zero based. All X counts and embedding
-coordinates are coerced to float32, which is ample precision for visualization purposes.
-Dataframe (metadata) types are generally preserved, or where that is not possible,
-converted to something with equal representative value in the cellxgene application
-(eg, categorical types are converted to string, bools to uint8, etc).
-
-The following objects are also decorated with auxiliary metadata using TileDB
-array metadata:
-
-* cxg_group_metadata: minimally, will contain a 'cxg_version' field, which
- is a semver string identifying the version number of the CXG layout.
- It may also contain 'cxg_parameters', a JSON-encoded parameter list
- describing CXG-wide dataset parameters.
-
-* obs, var: both contain an optional 'cxg_schema' field that is a json string,
- containing per-column (attribute) schema hinting. This is used where the TileDB
- native typing information is insufficient to reconstruct useful information
- such as categorical typing from Pandas DataFrames, and to communicate which column
- is the preferred human-readable index for obs & var.
-
-This file also embodies a number of empirically derived tiledb schema parameters,
-including the global data layout, spatial tile size, and the like. The CXG is
-self-describing in these areas, and the actual values (eg, tile size) are empirically
-derived from benchmarking. They may change in the future.
-
-cxgtool.py will extract color information stored in arrays in the 'uns' anndata
-property with the key "{category_name}_colors". For this to work, the following
-command must result in a mapping from category names to matplotlib-compatible colors:
-
-```
-dict(zip(adata.obs[cat].cat.categories, adata.uns[f"{cat}_colors"]))
-```
-
----
-
-TODO/ISSUES:
-* add sub-command structure to argparse, for future sub-commands
-* Possible future work: accept Loom files
+into a cellxgene TileDB structure, aka a [CXG](../../dev_docs/cxg.md).
+IF YOU UPDATE THIS FILE, IN ANY WAY THAT MODIFIES THE CXG FORMAT or CONTENTS,
+YOU MUST UPDATE THE CXG SPECIFICATION and VERSION NUMBER.
"""
import re
import anndata
@@ -76,11 +15,17 @@ import json
from scipy.stats import mode
from server.common.colors import convert_anndata_category_colors_to_cxg_category_colors
-from server.common.errors import ColorFormatException
+from server.common.errors import ColorFormatException, AnnotationCategoryNameError
+from server.common.corpora import (
+ corpora_get_props_from_anndata,
+ corpora_get_versions_from_anndata,
+ corpora_is_version_supported,
+)
-# the CXG container version number. Must be a semver string.
-CXG_VERSION = "0.1"
+# the CXG container version number. Must be a semver string (major.minor.patch)
+# DO NOT UPDATE THIS WITHOUT ALSO UPDATING THE CXG SPECIFICATION.
+CXG_VERSION = "0.2.0"
# log_level must have a default
log_level = 3
@@ -125,6 +70,12 @@ def main():
default=0.0, # force dense by default
help="The X array will be sparse if the percent of non-zeros falls below this value",
)
+ parser.add_argument(
+ "--disable-corpora",
+ action="store_true",
+ default=False,
+ help="Disable extraction and storing of Corpora schema information.",
+ )
args = parser.parse_args()
global log_level
@@ -136,25 +87,30 @@ def main():
basefname = splitext(basename(args.h5ad))[0]
out = args.out if args.out is not None else basefname
container = out if splitext(out)[1] == ".cxg" else out + ".cxg"
- title = args.title if args.title is not None else basefname
+
+ corpora_props = load_corpora_props(args, adata) if not args.disable_corpora else None
+ cxg_group_metadata = create_cxg_group_metadata(
+ adata,
+ basefname,
+ title=args.title,
+ about=args.about,
+ corpora_props=corpora_props,
+ extract_colors=not args.disable_custom_colors,
+ )
write_cxg(
adata,
container,
- title,
+ cxg_group_metadata=cxg_group_metadata,
var_names=args.var_names,
obs_names=args.obs_names,
- about=args.about,
- extract_colors=not args.disable_custom_colors,
sparse_threshold=args.sparse_threshold,
)
log(1, "done")
-def write_cxg(
- adata, container, title, var_names=None, obs_names=None, about=None, extract_colors=False, sparse_threshold=5.0
-):
+def write_cxg(adata, container, cxg_group_metadata, var_names=None, obs_names=None, sparse_threshold=5.0):
if not adata.var.index.is_unique:
raise ValueError("Variable index is not unique - unable to convert.")
if not adata.obs.index.is_unique:
@@ -179,19 +135,7 @@ def write_cxg(
log(1, f"\t...group created, with name {container}")
# dataset metadata
- metadata_dict = dict(cxg_version=CXG_VERSION, cxg_properties=json.dumps({"title": title, "about": about}))
- if extract_colors:
- try:
- metadata_dict["cxg_category_colors"] = json.dumps(
- convert_anndata_category_colors_to_cxg_category_colors(adata)
- )
- except ColorFormatException:
- log(
- 0,
- "Warning: failed to extract colors from h5ad file! "
- "Fix the h5ad file or rerun with --disable-custom-colors. See help for details.",
- )
- save_metadata(container, metadata_dict)
+ save_metadata(container, cxg_group_metadata)
log(1, "\t...dataset metadata saved")
# var/gene dataframe
@@ -347,20 +291,26 @@ def alias_index_col(df, df_name, index_col_name):
return (df, index_col_name)
+def generate_schema_hints_and_convert_value_types(df):
+ value = {}
+ schema_hints = {}
+ for k, v in df.items():
+ dtype, hints = cxg_type(v)
+ value[k] = v.to_numpy(dtype=dtype)
+ if hints:
+ schema_hints.update({k: hints})
+ return schema_hints, value
+
+
def save_dataframe(container, name, df, index_col_name, ctx):
A_name = f"{container}/{name}"
(df, index_col_name) = alias_index_col(df, name, index_col_name)
create_dataframe(A_name, df, ctx=ctx)
with tiledb.DenseArray(A_name, mode="w", ctx=ctx) as A:
- value = {}
- schema_hints = {}
- for k, v in df.items():
- dtype, hints = cxg_type(v)
- value[k] = v.to_numpy(dtype=dtype)
- if hints:
- schema_hints.update({k: hints})
-
+ schema_hints, value = generate_schema_hints_and_convert_value_types(df)
schema_hints.update({"index": index_col_name})
+ # convert all values in all cols to a numpy version of cxg datatypes,
+ # then store the contents in the tiledb array A
A[:] = value
A.meta["cxg_schema"] = json.dumps(schema_hints)
@@ -601,7 +551,60 @@ def save_metadata(container, metadata_dict):
A.meta[k] = v
-def sanitize_keys(keys):
+def load_corpora_props(args, adata):
+ versions = corpora_get_versions_from_anndata(adata)
+ if versions is None:
+ return None
+
+ [corpora_schema_version, corpora_encoding_version] = versions
+ corpora_props = corpora_get_props_from_anndata(adata)
+ version_is_supported = corpora_is_version_supported(corpora_schema_version, corpora_encoding_version)
+ if not version_is_supported or not corpora_props:
+ log(0, "ERROR: Unknown source file schema version is unsupported")
+ raise ValueError("Unsupported Corpora schema version")
+
+ log(1, "FYI, file appears to be encoded using Corpora schema standards...")
+ if args.title is not None or args.about is not None:
+ log(0, "Warning: explicit specification of --title or --about will override Corpora schema fields.")
+
+ return corpora_props
+
+
+def create_cxg_group_metadata(adata, basefname, title=None, about=None, corpora_props=None, extract_colors=True):
+
+ if corpora_props is not None:
+ # clobber encoding version to be OUR version, not the source H5AD encoding
+ corpora_props["version"].update({"corpora_encoding_version": CXG_VERSION})
+ corpora_project_links = corpora_props.get("project_links", [])
+ corpora_about_link = next(
+ (link for link in corpora_project_links if (link.get("link_type", None) == "SUMMARY")), {}
+ )
+ else:
+ corpora_about_link = {}
+
+ title = title or corpora_about_link.get("link_name", basefname)
+ about = about or corpora_about_link.get("link_url")
+
+ cxg_group_metadata = {"cxg_version": CXG_VERSION, "cxg_properties": json.dumps({"title": title, "about": about})}
+ if corpora_props is not None:
+ cxg_group_metadata.update({"corpora": json.dumps(corpora_props)})
+
+ if extract_colors:
+ try:
+ cxg_group_metadata["cxg_category_colors"] = json.dumps(
+ convert_anndata_category_colors_to_cxg_category_colors(adata)
+ )
+ except ColorFormatException:
+ log(
+ 0,
+ "Warning: failed to extract colors from h5ad file! "
+ "Fix the h5ad file or rerun with --disable-custom-colors. See help for details.",
+ )
+
+ return cxg_group_metadata
+
+
+def sanitize_keys(keys, update_keys=True):
"""
We need names to be safe to use as attribute names in tiledb. See:
TileDB-Inc/TileDB#1575
@@ -638,6 +641,8 @@ def sanitize_keys(keys):
for k, v, in clean_unique_keys.items():
if k != v:
+ if update_keys is False:
+ raise AnnotationCategoryNameError(f"{k} not a valid category name, please resubmit")
log(1, f"Renaming {k} to {v}")
return clean_unique_keys
diff --git a/server/data_anndata/anndata_adaptor.py b/server/data_anndata/anndata_adaptor.py
index 6022928e..3145a833 100644
--- a/server/data_anndata/anndata_adaptor.py
+++ b/server/data_anndata/anndata_adaptor.py
@@ -1,22 +1,22 @@
import warnings
-
-import numpy as np
-import pandas as pd
-from pandas.core.dtypes.dtypes import CategoricalDtype
-import anndata
-from scipy import sparse
-from packaging import version
from datetime import datetime
+
+import anndata
+import numpy as np
+from packaging import version
+from pandas.core.dtypes.dtypes import CategoricalDtype
+from scipy import sparse
from server_timing import Timing as ServerTiming
-from server.data_common.data_adaptor import DataAdaptor
-from server.data_common.fbs.matrix import encode_matrix_fbs
-from server.common.utils import series_to_schema
+import server.compute.diffexp_generic as diffexp_generic
from server.common.colors import convert_anndata_category_colors_to_cxg_category_colors
from server.common.constants import Axis, MAX_LAYOUTS
+from server.common.corpora import corpora_get_props_from_anndata
from server.common.errors import PrepareError, DatasetAccessError, FilterError
+from server.common.utils.type_conversion_utils import get_schema_type_hint_of_array
from server.compute.scanpy import scanpy_umap
-import server.compute.diffexp_generic as diffexp_generic
+from server.data_common.data_adaptor import DataAdaptor
+from server.data_common.fbs.matrix import encode_matrix_fbs
anndata_version = version.parse(str(anndata.__version__)).release
@@ -57,6 +57,9 @@ class AnndataAdaptor(DataAdaptor):
def open(data_locator, app_config, dataset_config=None):
return AnndataAdaptor(data_locator, app_config, dataset_config)
+ def get_corpora_props(self):
+ return corpora_get_props_from_anndata(self.data)
+
def get_name(self):
return "cellxgene anndata adaptor version"
@@ -134,7 +137,7 @@ class AnndataAdaptor(DataAdaptor):
curr_axis = getattr(self.data, str(ax))
for ann in curr_axis:
ann_schema = {"name": ann, "writable": False}
- ann_schema.update(series_to_schema(curr_axis[ann]))
+ ann_schema.update(get_schema_type_hint_of_array(curr_axis[ann]))
self.schema["annotations"][ax]["columns"].append(ann_schema)
for layout in self.get_embedding_names():
@@ -310,16 +313,15 @@ class AnndataAdaptor(DataAdaptor):
raise FilterError("Error parsing filter")
with ServerTiming.time("layout.compute"):
X_umap = scanpy_umap(self.data, obs_mask)
- normalized_layout = DataAdaptor.normalize_embedding(X_umap)
# Server picks reemedding name, which must not collide with any other
- # embedding name generated by this backed.
+ # embedding name generated by this backend.
name = f"reembed:{method}_{datetime.now().isoformat(timespec='milliseconds')}"
dims = [f"{name}_0", f"{name}_1"]
- df = pd.DataFrame(normalized_layout, columns=dims)
- fbs = encode_matrix_fbs(df, col_idx=df.columns, row_idx=None)
- schema = {"name": name, "type": "float32", "dims": dims}
- return (schema, fbs)
+ layout_schema = {"name": name, "type": "float32", "dims": dims}
+ self.schema["layout"]["obs"].append(layout_schema)
+ self.data.obsm[f"X_{name}"] = X_umap
+ return layout_schema
def compute_diffexp_ttest(self, maskA, maskB, top_n=None, lfc_cutoff=None):
if top_n is None:
diff --git a/server/data_common/data_adaptor.py b/server/data_common/data_adaptor.py
index eabec8d1..20dafdd0 100644
--- a/server/data_common/data_adaptor.py
+++ b/server/data_common/data_adaptor.py
@@ -1,14 +1,15 @@
from abc import ABCMeta, abstractmethod
-from server_timing import Timing as ServerTiming
-import numpy as np
-import pandas as pd
from os.path import basename, splitext
-from server.data_common.fbs.matrix import encode_matrix_fbs
+import numpy as np
+import pandas as pd
+from server_timing import Timing as ServerTiming
+
+from server.common.app_config import AppFeature, AppConfig
from server.common.constants import Axis
from server.common.errors import FilterError, JSONEncodingValueError, ExceedsLimitError
-from server.common.utils import jsonify_numpy
-from server.common.app_config import AppFeature, AppConfig
+from server.common.utils.utils import jsonify_numpy
+from server.data_common.fbs.matrix import encode_matrix_fbs
class DataAdaptor(metaclass=ABCMeta):
@@ -28,6 +29,11 @@ class DataAdaptor(metaclass=ABCMeta):
# parameters set by this data adaptor based on the data.
self.parameters = {}
+ self.uri_path = None
+
+ def set_uri_path(self, path):
+ # uri path to the dataset, e.g. /d/
+ self.uri_path = path
@staticmethod
@abstractmethod
@@ -66,8 +72,7 @@ class DataAdaptor(metaclass=ABCMeta):
@abstractmethod
def compute_embedding(self, method, filter):
- """compute a new embedding on the specified obs subset, and return a
- tuple of (schema, fbs)."""
+ """compute a new embedding on the specified obs subset, and return the embedding schema. """
pass
@abstractmethod
@@ -130,6 +135,9 @@ class DataAdaptor(metaclass=ABCMeta):
location = location[:-1]
return splitext(basename(location))[0]
+ def get_corpora_props(self):
+ return None
+
@abstractmethod
def get_schema(self):
"""
@@ -165,7 +173,7 @@ class DataAdaptor(metaclass=ABCMeta):
mask = np.zeros((count,), dtype=np.bool)
for i in filter:
if type(i) == list:
- mask[i[0] : i[1]] = True
+ mask[i[0]: i[1]] = True
else:
mask[i] = True
return mask
@@ -306,7 +314,7 @@ class DataAdaptor(metaclass=ABCMeta):
top_n = self.dataset_config.diffexp__top_n
if self.server_config.exceeds_limit(
- "diffexp_cellcount_max", np.count_nonzero(obs_mask_A) + np.count_nonzero(obs_mask_B)
+ "diffexp_cellcount_max", np.count_nonzero(obs_mask_A) + np.count_nonzero(obs_mask_B)
):
raise ExceedsLimitError("Diffexp request exceeds max cell count limit")
diff --git a/server/data_common/fbs/matrix.py b/server/data_common/fbs/matrix.py
index e10bc56a..213e3ce4 100644
--- a/server/data_common/fbs/matrix.py
+++ b/server/data_common/fbs/matrix.py
@@ -1,58 +1,24 @@
-import flatbuffers
-import numpy as np
-from scipy import sparse
-import pandas as pd
import json
+import numpy as np
+import pandas as pd
+from flatbuffers import Builder
+from scipy import sparse
+
import server.data_common.fbs.NetEncoding.Column as Column
-import server.data_common.fbs.NetEncoding.TypedArray as TypedArray
-import server.data_common.fbs.NetEncoding.Matrix as Matrix
-import server.data_common.fbs.NetEncoding.Int32Array as Int32Array
-import server.data_common.fbs.NetEncoding.Uint32Array as Uint32Array
import server.data_common.fbs.NetEncoding.Float32Array as Float32Array
import server.data_common.fbs.NetEncoding.Float64Array as Float64Array
+import server.data_common.fbs.NetEncoding.Int32Array as Int32Array
import server.data_common.fbs.NetEncoding.JSONEncodedArray as JSONEncodedArray
-
-
-# Placeholder until recent enhancements to flatbuffers Python
-# runtime are released, at which point we can use the default
-# version. This code is a port of the head. See:
-#
-# https://github.com/google/flatbuffers/pull/4829
-#
-def CreateNumpyVector(builder, x):
- """CreateNumpyVector writes a numpy array into the buffer."""
-
- if not isinstance(x, np.ndarray):
- raise TypeError(f"non-numpy-ndarray passed to CreateNumpyVector ({type(x)}")
-
- if x.dtype.kind not in ["b", "i", "u", "f"]:
- raise TypeError("numpy-ndarray holds elements of unsupported datatype")
-
- if x.ndim > 1:
- raise TypeError("multidimensional-ndarray passed to CreateNumpyVector")
-
- builder.StartVector(x.itemsize, x.size, x.dtype.alignment)
-
- # Ensure little endian byte ordering
- if x.dtype.str[0] == "<":
- x_little_endian = x
- else:
- x_little_endian = x.byteswap(inplace=False)
-
- # Calculate total length
- length = int(x_little_endian.itemsize * x_little_endian.size)
- builder.head = int(builder.Head() - length)
-
- # tobytes ensures c_contiguous ordering
- builder.Bytes[builder.Head() : builder.Head() + length] = x_little_endian.tobytes(order="C")
-
- return builder.EndVector(x.size)
+import server.data_common.fbs.NetEncoding.Matrix as Matrix
+import server.data_common.fbs.NetEncoding.TypedArray as TypedArray
+import server.data_common.fbs.NetEncoding.Uint32Array as Uint32Array
# Serialization helper
def serialize_column(builder, typed_arr):
""" Serialize NetEncoding.Column """
+
(u_type, u_value) = typed_arr
Column.ColumnStart(builder)
Column.ColumnAddUType(builder, u_type)
@@ -63,6 +29,7 @@ def serialize_column(builder, typed_arr):
# Serialization helper
def serialize_matrix(builder, n_rows, n_cols, columns, col_idx):
""" Serialize NetEncoding.Matrix """
+
Matrix.MatrixStart(builder)
Matrix.MatrixAddNRows(builder, n_rows)
Matrix.MatrixAddNCols(builder, n_cols)
@@ -77,9 +44,10 @@ def serialize_matrix(builder, n_rows, n_cols, columns, col_idx):
# Serialization helper
def serialize_typed_array(builder, source_array, encoding_info):
"""
- Serialize any of the various typed arrays, eg, Float32Array. Specific
- means of serialization and type conversion are provided by type_info.
+ Serialize any of the various typed arrays, eg, Float32Array. Specific means of serialization and type conversion
+ are provided by type_info.
"""
+
arr = source_array
(array_type, as_type) = encoding_info(source_array)
@@ -104,7 +72,8 @@ def serialize_typed_array(builder, source_array, encoding_info):
arr = arr[0]
elif arr.shape[1] == 1:
arr = arr.T[0]
- vec = CreateNumpyVector(builder, arr)
+
+ vec = builder.CreateNumpyVector(arr)
# serialize the typed array table
builder.StartObject(1)
@@ -113,38 +82,36 @@ def serialize_typed_array(builder, source_array, encoding_info):
return (array_type, array_value)
-column_encoding_type_map = {
- # array protocol string: ( array_type, as_type )
- np.dtype(np.float64).str: (TypedArray.TypedArray.Float32Array, np.float32),
- np.dtype(np.float32).str: (TypedArray.TypedArray.Float32Array, np.float32),
- np.dtype(np.float16).str: (TypedArray.TypedArray.Float32Array, np.float32),
- np.dtype(np.int8).str: (TypedArray.TypedArray.Int32Array, np.int32),
- np.dtype(np.int16).str: (TypedArray.TypedArray.Int32Array, np.int32),
- np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
- np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
- np.dtype(np.uint8).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
- np.dtype(np.uint16).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
- np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
- np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
-}
-column_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
-
-
def column_encoding(arr):
+ column_encoding_type_map = {
+ # array protocol string: ( array_type, as_type )
+ np.dtype(np.float64).str: (TypedArray.TypedArray.Float32Array, np.float32),
+ np.dtype(np.float32).str: (TypedArray.TypedArray.Float32Array, np.float32),
+ np.dtype(np.float16).str: (TypedArray.TypedArray.Float32Array, np.float32),
+ np.dtype(np.int8).str: (TypedArray.TypedArray.Int32Array, np.int32),
+ np.dtype(np.int16).str: (TypedArray.TypedArray.Int32Array, np.int32),
+ np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
+ np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
+ np.dtype(np.uint8).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
+ np.dtype(np.uint16).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
+ np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
+ np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
+ }
+ column_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
+
return column_encoding_type_map.get(arr.dtype.str, column_encoding_default)
-index_encoding_type_map = {
- # array protocol string: ( array_type, as_type )
- np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
- np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
- np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
- np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
-}
-index_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
-
-
def index_encoding(arr):
+ index_encoding_type_map = {
+ # array protocol string: ( array_type, as_type )
+ np.dtype(np.int32).str: (TypedArray.TypedArray.Int32Array, np.int32),
+ np.dtype(np.int64).str: (TypedArray.TypedArray.Int32Array, np.int32),
+ np.dtype(np.uint32).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
+ np.dtype(np.uint64).str: (TypedArray.TypedArray.Uint32Array, np.uint32),
+ }
+ index_encoding_default = (TypedArray.TypedArray.JSONEncodedArray, "json")
+
return index_encoding_type_map.get(arr.dtype.str, index_encoding_default)
@@ -165,8 +132,7 @@ def guess_at_mem_needed(matrix):
def encode_matrix_fbs(matrix, row_idx=None, col_idx=None):
"""
- Given a 2D DataFrame, ndarray or sparse equivalent, create and return a
- Matrix flatbuffer.
+ Given a 2D DataFrame, ndarray or sparse equivalent, create and return a Matrix flatbuffer.
:param matrix: 2D DataFrame, ndarray or sparse equivalent
:param row_idx: index for row dimension, Index or ndarray
@@ -183,7 +149,7 @@ def encode_matrix_fbs(matrix, row_idx=None, col_idx=None):
(n_rows, n_cols) = matrix.shape
# estimate size needed, so we don't unnecessarily realloc.
- builder = flatbuffers.Builder(guess_at_mem_needed(matrix))
+ builder = Builder(guess_at_mem_needed(matrix))
columns = []
for cidx in range(n_cols - 1, -1, -1):
@@ -239,9 +205,9 @@ def deserialize_typed_array(tarr):
def decode_matrix_fbs(fbs):
"""
- Given an FBS-encoded Matrix, return a Pandas DataFrame the contains the data
- and indices.
+ Given an FBS-encoded Matrix, return a Pandas DataFrame the contains the data and indices.
"""
+
matrix = Matrix.Matrix.GetRootAsMatrix(fbs, 0)
n_rows = matrix.NRows()
n_cols = matrix.NCols()
diff --git a/server/data_cxg/cxg_adaptor.py b/server/data_cxg/cxg_adaptor.py
index 2705f43b..db2fdd2a 100644
--- a/server/data_cxg/cxg_adaptor.py
+++ b/server/data_cxg/cxg_adaptor.py
@@ -1,9 +1,9 @@
import os
import json
import logging
-from server.common.utils import dtype_to_schema
+from server.common.utils.type_conversion_utils import get_schema_type_hint_from_dtype
from server.common.errors import DatasetAccessError, ConfigurationError
-from server.common.utils import path_join
+from server.common.utils.utils import path_join
from server.common.constants import Axis
from server.data_common.data_adaptor import DataAdaptor
from server.data_common.fbs.matrix import encode_matrix_fbs
@@ -74,6 +74,9 @@ class CxgAdaptor(DataAdaptor):
def get_title(self):
return self.title if self.title else super().get_title()
+ def get_corpora_props(self):
+ return self.corpora_props if self.corpora_props else super().get_corpora_props()
+
def get_name(self):
return "cellxgene cxg adaptor version"
@@ -144,26 +147,31 @@ class CxgAdaptor(DataAdaptor):
* version 0.1 -- metadata attache to cxg_group_metadata array.
Same as 0, except it adds group metadata.
"""
+ title = None
+ about = None
+ corpora_props = None
if self.has_array("cxg_group_metadata"):
# version >0
gmd = self.open_array("cxg_group_metadata")
cxg_version = gmd.meta["cxg_version"]
- if cxg_version == "0.1":
+ # version 0.1 used a malformed/shorthand semver string.
+ if cxg_version == "0.1" or cxg_version == "0.2.0":
cxg_properties = json.loads(gmd.meta["cxg_properties"])
title = cxg_properties.get("title", None)
about = cxg_properties.get("about", None)
+ if cxg_version == "0.2.0":
+ corpora_props = json.loads(gmd.meta["corpora"]) if "corpora" in gmd.meta else None
else:
# version 0
cxg_version = "0.0"
- title = None
- about = None
- if cxg_version not in ["0.0", "0.1"]:
+ if cxg_version not in ["0.0", "0.1", "0.2.0"]:
raise DatasetAccessError(f"cxg matrix is not valid: {self.url}")
self.title = title
self.about = about
self.cxg_version = cxg_version
+ self.corpora_props = corpora_props
@staticmethod
def _open_array(uri, tiledb_ctx):
@@ -381,7 +389,7 @@ class CxgAdaptor(DataAdaptor):
if schema["type"] == "categorical" and "categories" in type_hint:
schema["categories"] = type_hint["categories"]
else:
- schema.update(dtype_to_schema(attr.dtype))
+ schema.update(get_schema_type_hint_from_dtype(attr.dtype))
cols.append(schema)
annotations[ax] = dict(columns=cols)
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/__tiledb_group.tdb b/server/db/__init__.py
old mode 100755
new mode 100644
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/__tiledb_group.tdb
rename to server/db/__init__.py
diff --git a/server/db/cellxgene_orm.py b/server/db/cellxgene_orm.py
new file mode 100644
index 00000000..f2209b63
--- /dev/null
+++ b/server/db/cellxgene_orm.py
@@ -0,0 +1,63 @@
+import uuid
+
+from sqlalchemy import (
+ Column,
+ DateTime,
+ ForeignKey,
+ String,
+ func, JSON)
+from sqlalchemy.dialects.postgresql import UUID
+from sqlalchemy.ext.declarative import declarative_base
+from sqlalchemy.orm import relationship
+
+Base = declarative_base()
+
+
+class CellxGeneUser(Base):
+ """
+ A registered CellxGene user.
+ Links a user to their annotations
+ """
+
+ __tablename__ = "cxguser"
+
+ id = Column(String, primary_key=True)
+ created_at = Column(DateTime, nullable=False, server_default=func.now())
+ updated_at = Column(DateTime, nullable=False, server_default=func.now(), onupdate=func.now())
+
+ # Relationships
+ annotations = relationship("Annotation", back_populates="cxguser")
+
+
+class Annotation(Base):
+ """
+ An annotation is a link between a user, a dataset and tiledb dataframe. A user can have multiple annotations for a
+ dataset, the most recent annotation (based on created_at) will be the default returned when queried
+ """
+
+ __tablename__ = "annotation"
+
+ id = Column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4, unique=True, nullable=False)
+ tiledb_uri = Column(String)
+ user_id = Column(String, ForeignKey("cxguser.id"), nullable=False)
+ dataset_id = Column(UUID, ForeignKey("cxgdataset.id"), nullable=False)
+
+ created_at = Column(DateTime, nullable=False, server_default=func.now())
+ schema_hints = Column(JSON)
+ # Relationships
+ cxguser = relationship("CellxGeneUser", back_populates="annotations")
+ dataset = relationship("CellxGeneDataset", back_populates="annotations")
+
+
+class CellxGeneDataset(Base):
+ """
+ Datasets refer to datasets stored by cellxgene
+ """
+
+ __tablename__ = "cxgdataset"
+
+ id = Column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4, unique=True, nullable=False)
+ name = Column(String, unique=True, index=True)
+
+ created_at = Column(DateTime, nullable=False, server_default=func.now())
+ annotations = relationship("Annotation", back_populates="dataset")
diff --git a/server/db/create_db.py b/server/db/create_db.py
new file mode 100644
index 00000000..c0a7cf98
--- /dev/null
+++ b/server/db/create_db.py
@@ -0,0 +1,18 @@
+"""
+Drops and recreates all tables for local testing according to cellxgene_orm.py
+"""
+from sqlalchemy import create_engine
+
+from server.db.cellxgene_orm import Base
+
+
+def create_db(database_uri: str = "postgresql://postgres:test_pw@localhost:5432"):
+ engine = create_engine(database_uri)
+ print("Dropping tables")
+ Base.metadata.drop_all(engine)
+ print("Recreating tables")
+ Base.metadata.create_all(engine)
+
+
+if __name__ == "__main__":
+ create_db()
diff --git a/server/db/db_utils.py b/server/db/db_utils.py
new file mode 100644
index 00000000..1664303f
--- /dev/null
+++ b/server/db/db_utils.py
@@ -0,0 +1,73 @@
+import typing
+import uuid
+
+from sqlalchemy import create_engine
+from sqlalchemy.orm import sessionmaker
+
+from server.db.cellxgene_orm import Base, CellxGeneDataset, CellxGeneUser
+
+
+class DbUtils:
+ def __init__(self, database_uri: str = "postgresql://postgres:test_pw@localhost:5432"):
+ self.session = DBSessionMaker(database_uri).session()
+ self.engine = self.session.get_bind()
+
+ def get(self, table: Base, entity_id: typing.Union[str, typing.Tuple[str]]) -> typing.Union[Base, None]:
+ """
+ Query a table row by its primary key
+ :param table: SQLAlchemy Table to query
+ :param entity_id: Primary key of desired row
+ :return: SQLAlchemy Table object, None if not found
+ """
+ return self.session.query(table).get(entity_id)
+
+ def query(self, table_args: typing.List[Base], filter_args: typing.List[bool] = None) -> typing.List[Base]:
+ """
+ Query the database using the current DB session
+ :param table_args: List of SQLAlchemy Tables to query/join
+ :param filter_args: List of SQLAlchemy filter conditions
+ :return: List of SQLAlchemy query response objects
+ """
+ return (
+ self.session.query(*table_args).filter(*filter_args).all()
+ if filter_args
+ else self.session.query(*table_args).all()
+ )
+
+ def query_for_most_recent(self, table: Base, filter_args: typing.List[bool] = None) -> Base:
+ try:
+ return self.session.query(table).filter(*filter_args).order_by(table.created_at.desc()).limit(1).all()[0]
+ except IndexError:
+ return None
+
+ def get_or_create_dataset(self, dataset_name):
+ try:
+ dataset_id = self.query(
+ table_args=[CellxGeneDataset], filter_args=[CellxGeneDataset.name == dataset_name]
+ )[0].id
+ except IndexError:
+ dataset_id = uuid.uuid4()
+ dataset = CellxGeneDataset(id=dataset_id, name=dataset_name)
+ self.session.add(dataset)
+ self.session.commit()
+ return str(dataset_id)
+
+ def get_or_create_user(self, user_id):
+ try:
+ user_id = self.query(
+ table_args=[CellxGeneUser], filter_args=[CellxGeneUser.id == user_id]
+ )[0].id
+ except IndexError:
+ user = CellxGeneUser(id=user_id)
+ self.session.add(user)
+ self.session.commit()
+ return str(user_id)
+
+
+class DBSessionMaker:
+ def __init__(self, database_uri):
+ self.engine = create_engine(database_uri, connect_args={"connect_timeout": 5})
+ self.session_maker = sessionmaker(bind=self.engine)
+
+ def session(self, **kwargs):
+ return self.session_maker(**kwargs)
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/draw_graph_fr/__lock.tdb b/server/eb/.ebextensions/database.config
old mode 100755
new mode 100644
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/draw_graph_fr/__lock.tdb
rename to server/eb/.ebextensions/database.config
diff --git a/server/eb/README.md b/server/eb/README.md
index 5fa635df..2e7827c9 100644
--- a/server/eb/README.md
+++ b/server/eb/README.md
@@ -203,7 +203,7 @@ $ EB_INSTANCE=m5.large
$ CXG_DATAROOT=
$ CXG_CONFIG_FILE=
-# Potentially also set envvars for the sercret key.
+# Potentially also set envvars for the secret key.
$ eb create $EB_ENV --instance-type $EB_INSTANCE \
--envvars CXG_DATAROOT=$CXG_DATAROOT,CXG_CONFIG_FILE=$CXG_CONFIG_FILE
diff --git a/server/eb/app.py b/server/eb/app.py
index 438cd9bb..9aa86263 100644
--- a/server/eb/app.py
+++ b/server/eb/app.py
@@ -7,7 +7,9 @@ import base64
from flask import json
import logging
from flask_talisman import Talisman
-import boto3
+
+from server.common.aws_secret_utils import handle_config_from_secret
+from server.common.errors import SecretKeyRetrievalError
if os.path.isdir("/opt/python/log"):
@@ -19,9 +21,6 @@ if os.path.isdir("/opt/python/log"):
datefmt="%Y-%m-%d %H:%M:%S",
)
-# echo the logs to stdout. Useful for local testing
-logging.getLogger().addHandler(logging.StreamHandler(sys.stdout))
-
SERVERDIR = os.path.dirname(os.path.realpath(__file__))
sys.path.append(SERVERDIR)
@@ -34,37 +33,29 @@ except Exception:
sys.exit(1)
-def get_flask_secret_key(region_name, secret_name):
- session = boto3.session.Session()
- client = session.client(service_name="secretsmanager", region_name=region_name)
-
- try:
- get_secret_value_response = client.get_secret_value(SecretId=secret_name)
- if "SecretString" in get_secret_value_response:
- var = get_secret_value_response["SecretString"]
- secret = json.loads(var)
- return secret.get("flask_secret_key")
- except Exception:
- logging.critical("Caught exception during get_secret_key", exc_info=True)
- sys.exit(1)
-
- return None
-
-
class WSGIServer(Server):
def __init__(self, app_config):
super().__init__(app_config)
@staticmethod
def _before_adding_routes(app, app_config):
- script_hashes, style_hashes = WSGIServer.get_csp_hashes(app, app_config)
+ script_hashes = WSGIServer.get_csp_hashes(app, app_config)
server_config = app_config.server_config
+ # This hash should be in sync with the script within
+ # `client/configuration/webpack/obsoleteHTMLTemplate.html`
+
+ # It is _very_ difficult to generate the correct hash manually,
+ # consider forcing CSP to fail on the local server by intercepting the response via Requestly
+ # this should print the failing script's hash to console.
+ # See more here: https://github.com/chanzuckerberg/cellxgene/pull/1745
+ obsolete_browser_script_hash = ["'sha256-/rmgOi/skq9MpiZxPv6lPb1PNSN+Uf4NaUHO/IjyfwM='"]
csp = {
"default-src": ["'self'"],
"connect-src": ["'self'"],
- "script-src": ["'self'", "'unsafe-eval'", "'unsafe-inline'"] + script_hashes,
- "style-src": ["'self'", "'unsafe-inline'"] + style_hashes,
- "img-src": ["'self'", "data:"],
+ "script-src": ["'self'", "'unsafe-eval'"]
+ + obsolete_browser_script_hash + script_hashes,
+ "style-src": ["'self'", "'unsafe-inline'"],
+ "img-src": ["'self'", "https://cellxgene.cziscience.com", "data:"],
"object-src": ["'none'"],
"base-uri": ["'none'"],
"frame-ancestors": ["'none'"],
@@ -94,15 +85,13 @@ class WSGIServer(Server):
if not isinstance(csp_hashes, dict):
csp_hashes = {}
script_hashes = [f"'{hash}'" for hash in csp_hashes.get("script-hashes", [])]
- style_hashes = [f"'{hash}'" for hash in csp_hashes.get("style-hashes", [])]
-
- if len(script_hashes) == 0 or len(style_hashes) == 0:
+ if len(script_hashes) == 0:
logging.error("Content security policy hashes are missing, falling back to unsafe-inline policy")
- return (script_hashes, style_hashes)
+ return (script_hashes)
@staticmethod
- def compute_inline_scp_hashes(app, app_config):
+ def compute_inline_csp_hashes(app, app_config):
dataset_configs = [app_config.default_dataset_config] + list(app_config.dataroot_config.values())
hashes = []
for dataset_config in dataset_configs:
@@ -119,9 +108,9 @@ class WSGIServer(Server):
@staticmethod
def get_csp_hashes(app, app_config):
- script_hashes, style_hashes = WSGIServer.load_static_csp_hashes(app)
- script_hashes += WSGIServer.compute_inline_scp_hashes(app, app_config)
- return (script_hashes, style_hashes)
+ script_hashes = WSGIServer.load_static_csp_hashes(app)
+ script_hashes += WSGIServer.compute_inline_csp_hashes(app, app_config)
+ return script_hashes
try:
@@ -161,30 +150,17 @@ try:
logging.info("Configuration from CXG_DATAROOT")
app_config.update_server_config(multi_dataset__dataroot=dataroot)
- secret_name = os.getenv("CXG_AWS_SECRET_NAME")
- if secret_name:
- # need to find the secret manager region.
- # 1. from CXG_AWS_SECRET_REGION_NAME
- # 2. discover from dataroot location (if on s3)
- # 3. discover from config file location (if on s3)
- secret_region_name = os.getenv("CXG_AWS_SECRET_REGION_NAME")
- if secret_region_name is None:
- secret_region_name = discover_s3_region_name(app_config.multi_dataset__dataroot)
- if not secret_region_name:
- secret_region_name = discover_s3_region_name(config_file)
- if not secret_region_name:
- logging.error("Could not determine the AWS Secret Manager region")
- sys.exit(1)
-
- flask_secret_key = get_flask_secret_key(secret_region_name, secret_name)
- app_config.update_server_config(app__flask_secret_key=flask_secret_key)
+ # update from secret manager
+ try:
+ handle_config_from_secret(app_config)
+ except SecretKeyRetrievalError:
+ sys.exit(1)
# features are unsupported in the current hosted server
app_config.update_default_dataset_config(
user_annotations__enable=False, embeddings__enable_reembedding=False,
)
app_config.update_server_config(multi_dataset__allowed_matrix_types=["cxg"],)
-
app_config.complete_config(logging.info)
if not app_config.server_config.app__flask_secret_key:
diff --git a/server/requirements-dev.txt b/server/requirements-dev.txt
index 291ff9c1..7921926f 100644
--- a/server/requirements-dev.txt
+++ b/server/requirements-dev.txt
@@ -4,4 +4,6 @@ parameterized>=0.7.0
pytest>=3.6.3
twine>=1.12.1
codecov>=2.0.15
+scanpy>=1.4.6
+psycopg2==2.7.7
-r requirements.txt
diff --git a/server/requirements.txt b/server/requirements.txt
index d47e383f..ac08deed 100644
--- a/server/requirements.txt
+++ b/server/requirements.txt
@@ -8,9 +8,9 @@ Flask-Cors>=3.0.6
Flask-RESTful>=0.3.6
flask-server-timing>=0.1.2
flask-talisman>=0.7.0
-flatbuffers>=1.10.0
+flatbuffers>=1.11.0
flatten-dict>=0.2.0
-fsspec>=0.4.4
+fsspec>=0.4.4,<0.8.0
numba>=0.49.1
numpy>=1.16.0
packaging>=20.0
@@ -18,6 +18,7 @@ pandas>=0.24.2
PyYAML>=5.3
scipy>=1.3.0
requests>=2.22.0
+sqlalchemy>=1.3.18
tiledb>=0.5.9,>=0.6.2
s3fs>=0.4.2
gunicorn>=20.0.4
diff --git a/server/test/__init__.py b/server/test/__init__.py
index bc82be90..49e28ce5 100644
--- a/server/test/__init__.py
+++ b/server/test/__init__.py
@@ -1,35 +1,67 @@
+import os
import random
import shutil
import string
import tempfile
-import requests
import time
-import os
-from subprocess import Popen
-from os import path, popen
from contextlib import contextmanager
+from os import path, popen
+from subprocess import Popen
import pandas as pd
+import requests
-from server.common.annotations import AnnotationsLocalFile
-from server.common.data_locator import DataLocator
+from server.common.annotations.hosted_tiledb import AnnotationsHostedTileDB
+from server.common.annotations.local_file_csv import AnnotationsLocalFile
from server.common.app_config import AppConfig, DEFAULT_SERVER_PORT
-from server.common.utils import find_available_port
+from server.common.data_locator import DataLocator
+from server.common.utils.utils import find_available_port
from server.data_common.fbs.matrix import encode_matrix_fbs
from server.data_common.matrix_loader import MatrixDataLoader, MatrixDataType
-
+from server.db.db_utils import DbUtils
PROJECT_ROOT = popen("git rev-parse --show-toplevel").read().strip()
+FIXTURES_ROOT = PROJECT_ROOT + "/server/test/fixtures"
+
+
+def data_with_tmp_tiledb_annotations(ext: MatrixDataType):
+ tmp_dir = tempfile.mkdtemp()
+ fname = {
+ MatrixDataType.H5AD: f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad",
+ MatrixDataType.CXG: "test/fixtures/pbmc3k.cxg",
+ }[ext]
+ data_locator = DataLocator(fname)
+ config = AppConfig()
+ config.update_server_config(
+ multi_dataset__dataroot=data_locator.path, authentication__type="test"
+ )
+ config.update_default_dataset_config(
+ embeddings__names=["umap"],
+ presentation__max_categories=100,
+ diffexp__lfc_cutoff=0.01,
+ user_annotations__type="hosted_tiledb_array",
+ user_annotations__hosted_tiledb_array__db_uri="postgresql://postgres:test_pw@localhost:5432",
+ user_annotations__hosted_tiledb_array__hosted_file_directory=tmp_dir
+ )
+
+ config.complete_config()
+
+ data = MatrixDataLoader(data_locator.abspath()).open(config)
+ annotations = AnnotationsHostedTileDB(
+ tmp_dir,
+ DbUtils("postgresql://postgres:test_pw@localhost:5432")
+ )
+ return data, tmp_dir, annotations
def data_with_tmp_annotations(ext: MatrixDataType, annotations_fixture=False):
tmp_dir = tempfile.mkdtemp()
annotations_file = path.join(tmp_dir, "test_annotations.csv")
if annotations_fixture:
- shutil.copyfile(f"{PROJECT_ROOT}/server/test/test_datasets/pbmc3k-annotations.csv", annotations_file)
+ shutil.copyfile(f"{PROJECT_ROOT}/server/test/fixtures/pbmc3k-annotations.csv", annotations_file)
fname = {
MatrixDataType.H5AD: f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad",
- MatrixDataType.CXG: "test/test_datasets/pbmc3k.cxg",
+ MatrixDataType.CXG: "test/fixtures/pbmc3k.cxg",
}[ext]
data_locator = DataLocator(fname)
config = AppConfig()
@@ -39,6 +71,7 @@ def data_with_tmp_annotations(ext: MatrixDataType, annotations_fixture=False):
config.update_default_dataset_config(
embeddings__names=["umap"], presentation__max_categories=100, diffexp__lfc_cutoff=0.01,
)
+
config.complete_config()
data = MatrixDataLoader(data_locator.abspath()).open(config)
annotations = AnnotationsLocalFile(None, annotations_file)
@@ -86,15 +119,14 @@ def random_string(n):
return "".join(random.choice(string.ascii_letters) for _ in range(n))
-@contextmanager
-def test_server(command_line_args=[], app_config=None):
- """A context to run the cellxgene server.
+def start_test_server(command_line_args=[], app_config=None):
+ """
Command line arguments can be passed in, as well as an app_config.
This function is meant to be used like this, for example:
with test_server(...) as server:
- r = requests.get(f"{server}/...")
- // check r
+ r = requests.get(f"{server}/...")
+ // check r
where the server can be accessed within the context, and is terminated when
the context is exited.
@@ -104,7 +136,8 @@ def test_server(command_line_args=[], app_config=None):
yaml config file, which this server will read and parse.
"""
- port = int(os.environ.get("CXG_SERVER_PORT", DEFAULT_SERVER_PORT))
+ start = random.randint(DEFAULT_SERVER_PORT, 2 ** 16 - 1)
+ port = int(os.environ.get("CXG_SERVER_PORT", start))
port = find_available_port("localhost", port)
command = ["cellxgene", "--no-upgrade-check", "launch", "--verbose", "--port=%d" % port] + command_line_args
@@ -128,10 +161,25 @@ def test_server(command_line_args=[], app_config=None):
if tempdir:
tempdir.cleanup()
+ return ps, server
+
+
+def stop_test_server(ps):
+ try:
+ ps.terminate()
+ except ProcessLookupError:
+ pass
+
+
+@contextmanager
+def test_server(command_line_args=[], app_config=None):
+ """A context to run the cellxgene server."""
+
+ ps, server = start_test_server(command_line_args, app_config)
try:
yield server
finally:
try:
- ps.terminate()
+ stop_test_server(ps)
except ProcessLookupError:
pass
diff --git a/server/test/decode_fbs.py b/server/test/decode_fbs.py
deleted file mode 100644
index 5debe7ae..00000000
--- a/server/test/decode_fbs.py
+++ /dev/null
@@ -1,62 +0,0 @@
-"""
-Code to decode, for testing purposes, the flatbuffer encoded blobs.
-This code will need to be updated if fbs/matrix.fbs changes.
-
-For more information, see fbs/matrix.fbs and server/data_common/fbs/
-"""
-import json
-
-import server.data_common.fbs.NetEncoding.TypedArray as TypedArray
-import server.data_common.fbs.NetEncoding.Matrix as Matrix
-import server.data_common.fbs.NetEncoding.Int32Array as Int32Array
-import server.data_common.fbs.NetEncoding.Uint32Array as Uint32Array
-import server.data_common.fbs.NetEncoding.Float32Array as Float32Array
-import server.data_common.fbs.NetEncoding.Float64Array as Float64Array
-import server.data_common.fbs.NetEncoding.JSONEncodedArray as JSONEncodedArray
-
-
-def decode_typed_array(tarr):
- type_map = {
- TypedArray.TypedArray.Uint32Array: Uint32Array.Uint32Array,
- TypedArray.TypedArray.Int32Array: Int32Array.Int32Array,
- TypedArray.TypedArray.Float32Array: Float32Array.Float32Array,
- TypedArray.TypedArray.Float64Array: Float64Array.Float64Array,
- TypedArray.TypedArray.JSONEncodedArray: JSONEncodedArray.JSONEncodedArray,
- }
- (u_type, u) = tarr
- if u_type == TypedArray.TypedArray.NONE:
- return None
-
- TarType = type_map.get(u_type, None)
- assert TarType is not None
-
- arr = TarType()
- arr.Init(u.Bytes, u.Pos)
- narr = arr.DataAsNumpy()
- if u_type == TypedArray.TypedArray.JSONEncodedArray:
- narr = json.loads(narr.tostring().decode("utf-8"))
- return narr
-
-
-def decode_matrix_FBS(buf):
- """
- Given a FBS Matrix, return an decoded Python dict containing
- same info in native format.
-
- NOTE / TODO: row_idx not currently implemented
- """
- df = Matrix.Matrix.GetRootAsMatrix(buf, 0)
- n_rows = df.NRows()
- n_cols = df.NCols()
-
- columns_length = df.ColumnsLength()
-
- decoded_columns = []
- for col_idx in range(0, columns_length):
- col = df.Columns(col_idx)
- tarr = (col.UType(), col.U())
- decoded_columns.append(decode_typed_array(tarr))
-
- cidx = decode_typed_array((df.ColIndexType(), df.ColIndex()))
-
- return {"n_rows": n_rows, "n_cols": n_cols, "columns": decoded_columns, "col_idx": cidx, "row_idx": None}
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/pca/__lock.tdb b/server/test/fixtures/__init__.py
old mode 100755
new mode 100644
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/pca/__lock.tdb
rename to server/test/fixtures/__init__.py
diff --git a/server/test/fixtures/database/__init__.py b/server/test/fixtures/database/__init__.py
new file mode 100644
index 00000000..5a7fa984
--- /dev/null
+++ b/server/test/fixtures/database/__init__.py
@@ -0,0 +1,89 @@
+import string
+import random
+
+
+from sqlalchemy import func
+
+from server.db.cellxgene_orm import CellxGeneUser, CellxGeneDataset, Annotation, Base
+from server.db.create_db import create_db
+from server.db.db_utils import DbUtils
+
+
+class TestDatabase:
+ def __init__(self):
+ local_db_uri = "postgresql://postgres:test_pw@localhost:5432"
+ create_db(local_db_uri)
+ self.db = DbUtils(local_db_uri)
+ self._populate_test_data()
+ self._populate_test_data_many()
+
+ def _populate_test_data(self):
+ self._create_test_user()
+ self._create_test_dataset()
+ self._create_test_annotation()
+
+ def _populate_test_data_many(self):
+ self._create_test_users()
+ self._create_test_datasets()
+ self._create_test_annotations()
+
+ def _create_test_user(self):
+ user = CellxGeneUser(id="test_user_id")
+ user2 = CellxGeneUser(id='1234')
+ self.db.session.add(user)
+ self.db.session.add(user2)
+ self.db.session.commit()
+
+ def _create_test_dataset(self):
+ dataset = CellxGeneDataset(
+ name="test_dataset",
+ )
+ self.db.session.add(dataset)
+ self.db.session.commit()
+
+ def _create_test_annotation(self):
+ dataset = self.db.query([CellxGeneDataset],
+ [CellxGeneDataset.name == "test_dataset"],
+ )[0]
+ annotation = Annotation(
+ tiledb_uri="tiledb_uri",
+ user_id="test_user_id",
+ dataset_id=str(dataset.id)
+ )
+ self.db.session.add(annotation)
+ self.db.session.commit()
+
+ @staticmethod
+ def get_random_string():
+ letters = string.ascii_lowercase
+ return ''.join(random.choice(letters) for i in range(12))
+
+ def _create_test_users(self, user_count: int = 10):
+ users = []
+ for i in range(user_count):
+ users.append(CellxGeneUser(id=self.get_random_string()))
+ self.db.session.add_all(users)
+ self.db.session.commit()
+
+ def _create_test_datasets(self, dataset_count: int = 10):
+ datasets = []
+ for i in range(dataset_count):
+ datasets.append(CellxGeneDataset(name=self.get_random_string()))
+ self.db.session.add_all(datasets)
+ self.db.session.commit()
+
+ def order_by_random(self, table: Base):
+ return self.db.session.query(table).order_by(func.random()).first()
+
+ def _create_test_annotations(self, annotation_count: int = 10):
+ annotations = []
+ for i in range(annotation_count):
+ dataset = self.order_by_random(CellxGeneDataset)
+ user = self.order_by_random(CellxGeneUser)
+ annotations.append(Annotation(
+ tiledb_uri=self.get_random_string(),
+ user_id=user.id,
+ dataset_id=str(dataset.id)
+ ))
+ self.db.session.add_all(annotations)
+ self.db.session.commit()
diff --git a/server/test/test_datasets/fixtures.py b/server/test/fixtures/fixtures.py
similarity index 100%
rename from server/test/test_datasets/fixtures.py
rename to server/test/fixtures/fixtures.py
diff --git a/server/test/test_datasets/nan.h5ad b/server/test/fixtures/nan.h5ad
similarity index 100%
rename from server/test/test_datasets/nan.h5ad
rename to server/test/fixtures/nan.h5ad
diff --git a/server/test/test_datasets/pbmc3k-CSC-gz.h5ad b/server/test/fixtures/pbmc3k-CSC-gz.h5ad
similarity index 100%
rename from server/test/test_datasets/pbmc3k-CSC-gz.h5ad
rename to server/test/fixtures/pbmc3k-CSC-gz.h5ad
diff --git a/server/test/test_datasets/pbmc3k-CSR-gz.h5ad b/server/test/fixtures/pbmc3k-CSR-gz.h5ad
similarity index 100%
rename from server/test/test_datasets/pbmc3k-CSR-gz.h5ad
rename to server/test/fixtures/pbmc3k-CSR-gz.h5ad
diff --git a/server/test/test_datasets/pbmc3k-annotations.csv b/server/test/fixtures/pbmc3k-annotations.csv
similarity index 100%
rename from server/test/test_datasets/pbmc3k-annotations.csv
rename to server/test/fixtures/pbmc3k-annotations.csv
diff --git a/server/test/test_datasets/pbmc3k.cxg/X/__1587182255882_1587182255882_f7aa6ccb49a944f9ab9e25b11bbfab4f/__attr.tdb b/server/test/fixtures/pbmc3k.cxg/X/__1587182255882_1587182255882_f7aa6ccb49a944f9ab9e25b11bbfab4f/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/X/__1587182255882_1587182255882_f7aa6ccb49a944f9ab9e25b11bbfab4f/__attr.tdb
rename to server/test/fixtures/pbmc3k.cxg/X/__1587182255882_1587182255882_f7aa6ccb49a944f9ab9e25b11bbfab4f/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/X/__1587182255882_1587182255882_f7aa6ccb49a944f9ab9e25b11bbfab4f/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k.cxg/X/__1587182255882_1587182255882_f7aa6ccb49a944f9ab9e25b11bbfab4f/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/X/__1587182255882_1587182255882_f7aa6ccb49a944f9ab9e25b11bbfab4f/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k.cxg/X/__1587182255882_1587182255882_f7aa6ccb49a944f9ab9e25b11bbfab4f/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/X/__array_schema.tdb b/server/test/fixtures/pbmc3k.cxg/X/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/X/__array_schema.tdb
rename to server/test/fixtures/pbmc3k.cxg/X/__array_schema.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/tsne/__lock.tdb b/server/test/fixtures/pbmc3k.cxg/X/__lock.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/tsne/__lock.tdb
rename to server/test/fixtures/pbmc3k.cxg/X/__lock.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/__tiledb_group.tdb b/server/test/fixtures/pbmc3k.cxg/__tiledb_group.tdb
old mode 100644
new mode 100755
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/__tiledb_group.tdb
rename to server/test/fixtures/pbmc3k.cxg/__tiledb_group.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__1587182255763_1587182255763_88cd3d13926f4892af7230837bcc5178/__attr.tdb b/server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__1587182255763_1587182255763_88cd3d13926f4892af7230837bcc5178/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__1587182255763_1587182255763_88cd3d13926f4892af7230837bcc5178/__attr.tdb
rename to server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__1587182255763_1587182255763_88cd3d13926f4892af7230837bcc5178/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__1587182255763_1587182255763_88cd3d13926f4892af7230837bcc5178/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__1587182255763_1587182255763_88cd3d13926f4892af7230837bcc5178/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__1587182255763_1587182255763_88cd3d13926f4892af7230837bcc5178/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__1587182255763_1587182255763_88cd3d13926f4892af7230837bcc5178/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__array_schema.tdb b/server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__array_schema.tdb
rename to server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__array_schema.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/umap/__lock.tdb b/server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__lock.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/umap/__lock.tdb
rename to server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__lock.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__meta/__1587182255768_1587182255768_fa7617ae99f843929911e4bc7b03e3db b/server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__meta/__1587182255768_1587182255768_fa7617ae99f843929911e4bc7b03e3db
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/cxg_group_metadata/__meta/__1587182255768_1587182255768_fa7617ae99f843929911e4bc7b03e3db
rename to server/test/fixtures/pbmc3k.cxg/cxg_group_metadata/__meta/__1587182255768_1587182255768_fa7617ae99f843929911e4bc7b03e3db
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/__tiledb_group.tdb b/server/test/fixtures/pbmc3k.cxg/emb/__tiledb_group.tdb
old mode 100644
new mode 100755
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/__tiledb_group.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/__tiledb_group.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/draw_graph_fr/__1587182255875_1587182255875_e02a92b5d18f45e0a02031568f4633d4/__attr.tdb b/server/test/fixtures/pbmc3k.cxg/emb/draw_graph_fr/__1587182255875_1587182255875_e02a92b5d18f45e0a02031568f4633d4/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/draw_graph_fr/__1587182255875_1587182255875_e02a92b5d18f45e0a02031568f4633d4/__attr.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/draw_graph_fr/__1587182255875_1587182255875_e02a92b5d18f45e0a02031568f4633d4/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/draw_graph_fr/__1587182255875_1587182255875_e02a92b5d18f45e0a02031568f4633d4/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k.cxg/emb/draw_graph_fr/__1587182255875_1587182255875_e02a92b5d18f45e0a02031568f4633d4/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/draw_graph_fr/__1587182255875_1587182255875_e02a92b5d18f45e0a02031568f4633d4/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/draw_graph_fr/__1587182255875_1587182255875_e02a92b5d18f45e0a02031568f4633d4/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/draw_graph_fr/__array_schema.tdb b/server/test/fixtures/pbmc3k.cxg/emb/draw_graph_fr/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/draw_graph_fr/__array_schema.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/draw_graph_fr/__array_schema.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__lock.tdb b/server/test/fixtures/pbmc3k.cxg/emb/draw_graph_fr/__lock.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__lock.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/draw_graph_fr/__lock.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/pca/__1587182255827_1587182255827_8aeb4df5c2a74918b7aeeb6d34632e24/__attr.tdb b/server/test/fixtures/pbmc3k.cxg/emb/pca/__1587182255827_1587182255827_8aeb4df5c2a74918b7aeeb6d34632e24/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/pca/__1587182255827_1587182255827_8aeb4df5c2a74918b7aeeb6d34632e24/__attr.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/pca/__1587182255827_1587182255827_8aeb4df5c2a74918b7aeeb6d34632e24/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/pca/__1587182255827_1587182255827_8aeb4df5c2a74918b7aeeb6d34632e24/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k.cxg/emb/pca/__1587182255827_1587182255827_8aeb4df5c2a74918b7aeeb6d34632e24/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/pca/__1587182255827_1587182255827_8aeb4df5c2a74918b7aeeb6d34632e24/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/pca/__1587182255827_1587182255827_8aeb4df5c2a74918b7aeeb6d34632e24/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/pca/__array_schema.tdb b/server/test/fixtures/pbmc3k.cxg/emb/pca/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/pca/__array_schema.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/pca/__array_schema.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/var/__lock.tdb b/server/test/fixtures/pbmc3k.cxg/emb/pca/__lock.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/var/__lock.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/pca/__lock.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/tsne/__1587182255846_1587182255846_0b832c3ed3534b458e83c75726f2005f/__attr.tdb b/server/test/fixtures/pbmc3k.cxg/emb/tsne/__1587182255846_1587182255846_0b832c3ed3534b458e83c75726f2005f/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/tsne/__1587182255846_1587182255846_0b832c3ed3534b458e83c75726f2005f/__attr.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/tsne/__1587182255846_1587182255846_0b832c3ed3534b458e83c75726f2005f/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/tsne/__1587182255846_1587182255846_0b832c3ed3534b458e83c75726f2005f/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k.cxg/emb/tsne/__1587182255846_1587182255846_0b832c3ed3534b458e83c75726f2005f/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/tsne/__1587182255846_1587182255846_0b832c3ed3534b458e83c75726f2005f/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/tsne/__1587182255846_1587182255846_0b832c3ed3534b458e83c75726f2005f/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/tsne/__array_schema.tdb b/server/test/fixtures/pbmc3k.cxg/emb/tsne/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/tsne/__array_schema.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/tsne/__array_schema.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/X/__lock.tdb b/server/test/fixtures/pbmc3k.cxg/emb/tsne/__lock.tdb
old mode 100644
new mode 100755
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/X/__lock.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/tsne/__lock.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/umap/__1587182255867_1587182255867_44955ba2220c4ad59bbdbb6b26e21428/__attr.tdb b/server/test/fixtures/pbmc3k.cxg/emb/umap/__1587182255867_1587182255867_44955ba2220c4ad59bbdbb6b26e21428/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/umap/__1587182255867_1587182255867_44955ba2220c4ad59bbdbb6b26e21428/__attr.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/umap/__1587182255867_1587182255867_44955ba2220c4ad59bbdbb6b26e21428/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/umap/__1587182255867_1587182255867_44955ba2220c4ad59bbdbb6b26e21428/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k.cxg/emb/umap/__1587182255867_1587182255867_44955ba2220c4ad59bbdbb6b26e21428/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/umap/__1587182255867_1587182255867_44955ba2220c4ad59bbdbb6b26e21428/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/umap/__1587182255867_1587182255867_44955ba2220c4ad59bbdbb6b26e21428/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/emb/umap/__array_schema.tdb b/server/test/fixtures/pbmc3k.cxg/emb/umap/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/emb/umap/__array_schema.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/umap/__array_schema.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/draw_graph_fr/__lock.tdb b/server/test/fixtures/pbmc3k.cxg/emb/umap/__lock.tdb
old mode 100644
new mode 100755
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/draw_graph_fr/__lock.tdb
rename to server/test/fixtures/pbmc3k.cxg/emb/umap/__lock.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/louvain.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/louvain.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/louvain.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/louvain.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/louvain_var.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/louvain_var.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/louvain_var.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/louvain_var.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/n_counts.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/n_counts.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/n_counts.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/n_counts.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/n_genes.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/n_genes.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/n_genes.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/n_genes.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/name_0.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/name_0.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/name_0.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/name_0.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/name_0_var.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/name_0_var.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/name_0_var.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/name_0_var.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/percent_mito.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/percent_mito.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/percent_mito.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__1587182255797_1587182255797_d3d57575169a48eaa00d2f709d0a534c/percent_mito.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__array_schema.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__array_schema.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__array_schema.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/pca/__lock.tdb b/server/test/fixtures/pbmc3k.cxg/obs/__lock.tdb
old mode 100644
new mode 100755
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/pca/__lock.tdb
rename to server/test/fixtures/pbmc3k.cxg/obs/__lock.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/obs/__meta/__1587182255787_1587182255787_e19a340576a340db85c37b61b83dbe57 b/server/test/fixtures/pbmc3k.cxg/obs/__meta/__1587182255787_1587182255787_e19a340576a340db85c37b61b83dbe57
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/obs/__meta/__1587182255787_1587182255787_e19a340576a340db85c37b61b83dbe57
rename to server/test/fixtures/pbmc3k.cxg/obs/__meta/__1587182255787_1587182255787_e19a340576a340db85c37b61b83dbe57
diff --git a/server/test/test_datasets/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/n_cells.tdb b/server/test/fixtures/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/n_cells.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/n_cells.tdb
rename to server/test/fixtures/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/n_cells.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/name_0.tdb b/server/test/fixtures/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/name_0.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/name_0.tdb
rename to server/test/fixtures/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/name_0.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/name_0_var.tdb b/server/test/fixtures/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/name_0_var.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/name_0_var.tdb
rename to server/test/fixtures/pbmc3k.cxg/var/__1587182255773_1587182255773_12e07b1585a64ebe9d40098d67a8e8ad/name_0_var.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/var/__array_schema.tdb b/server/test/fixtures/pbmc3k.cxg/var/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/var/__array_schema.tdb
rename to server/test/fixtures/pbmc3k.cxg/var/__array_schema.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/tsne/__lock.tdb b/server/test/fixtures/pbmc3k.cxg/var/__lock.tdb
old mode 100644
new mode 100755
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/tsne/__lock.tdb
rename to server/test/fixtures/pbmc3k.cxg/var/__lock.tdb
diff --git a/server/test/test_datasets/pbmc3k.cxg/var/__meta/__1587182255771_1587182255771_5a3a50f0b360403eab0e427c429f574e b/server/test/fixtures/pbmc3k.cxg/var/__meta/__1587182255771_1587182255771_5a3a50f0b360403eab0e427c429f574e
similarity index 100%
rename from server/test/test_datasets/pbmc3k.cxg/var/__meta/__1587182255771_1587182255771_5a3a50f0b360403eab0e427c429f574e
rename to server/test/fixtures/pbmc3k.cxg/var/__meta/__1587182255771_1587182255771_5a3a50f0b360403eab0e427c429f574e
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__attr.tdb b/server/test/fixtures/pbmc3k_v0.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__attr.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k_v0.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/X/__array_schema.tdb b/server/test/fixtures/pbmc3k_v0.cxg/X/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/X/__array_schema.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/X/__array_schema.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/umap/__lock.tdb b/server/test/fixtures/pbmc3k_v0.cxg/X/__lock.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/umap/__lock.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/X/__lock.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__lock.tdb b/server/test/fixtures/pbmc3k_v0.cxg/__tiledb_group.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__lock.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/__tiledb_group.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/var/__lock.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/__tiledb_group.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/var/__lock.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/__tiledb_group.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__attr.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__attr.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/draw_graph_fr/__array_schema.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/draw_graph_fr/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/draw_graph_fr/__array_schema.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/draw_graph_fr/__array_schema.tdb
diff --git a/server/test/fixtures/pbmc3k_v0.cxg/emb/draw_graph_fr/__lock.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/draw_graph_fr/__lock.tdb
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__attr.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__attr.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/pca/__array_schema.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/pca/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/pca/__array_schema.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/pca/__array_schema.tdb
diff --git a/server/test/fixtures/pbmc3k_v0.cxg/emb/pca/__lock.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/pca/__lock.tdb
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__attr.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__attr.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/tsne/__array_schema.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/tsne/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/tsne/__array_schema.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/tsne/__array_schema.tdb
diff --git a/server/test/fixtures/pbmc3k_v0.cxg/emb/tsne/__lock.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/tsne/__lock.tdb
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__attr.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__attr.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__attr.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__attr.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/emb/umap/__array_schema.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/umap/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/emb/umap/__array_schema.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/emb/umap/__array_schema.tdb
diff --git a/server/test/fixtures/pbmc3k_v0.cxg/emb/umap/__lock.tdb b/server/test/fixtures/pbmc3k_v0.cxg/emb/umap/__lock.tdb
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain_var.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain_var.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain_var.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain_var.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_counts.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_counts.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_counts.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_counts.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_genes.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_genes.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_genes.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_genes.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0_var.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0_var.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0_var.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0_var.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/percent_mito.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/percent_mito.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/percent_mito.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/percent_mito.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__array_schema.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__array_schema.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__array_schema.tdb
diff --git a/server/test/fixtures/pbmc3k_v0.cxg/obs/__lock.tdb b/server/test/fixtures/pbmc3k_v0.cxg/obs/__lock.tdb
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/obs/__meta/__1576858534024_1576858534024_e856aef5b0e244b1a4b7dde70cd864d0 b/server/test/fixtures/pbmc3k_v0.cxg/obs/__meta/__1576858534024_1576858534024_e856aef5b0e244b1a4b7dde70cd864d0
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/obs/__meta/__1576858534024_1576858534024_e856aef5b0e244b1a4b7dde70cd864d0
rename to server/test/fixtures/pbmc3k_v0.cxg/obs/__meta/__1576858534024_1576858534024_e856aef5b0e244b1a4b7dde70cd864d0
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/__fragment_metadata.tdb b/server/test/fixtures/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/__fragment_metadata.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/__fragment_metadata.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/__fragment_metadata.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/n_cells.tdb b/server/test/fixtures/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/n_cells.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/n_cells.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/n_cells.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0.tdb b/server/test/fixtures/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0_var.tdb b/server/test/fixtures/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0_var.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0_var.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0_var.tdb
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/var/__array_schema.tdb b/server/test/fixtures/pbmc3k_v0.cxg/var/__array_schema.tdb
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/var/__array_schema.tdb
rename to server/test/fixtures/pbmc3k_v0.cxg/var/__array_schema.tdb
diff --git a/server/test/fixtures/pbmc3k_v0.cxg/var/__lock.tdb b/server/test/fixtures/pbmc3k_v0.cxg/var/__lock.tdb
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_datasets/pbmc3k_v0.cxg/var/__meta/__1576858533966_1576858533966_1362646d804b4982b35052a367633436 b/server/test/fixtures/pbmc3k_v0.cxg/var/__meta/__1576858533966_1576858533966_1362646d804b4982b35052a367633436
similarity index 100%
rename from server/test/test_datasets/pbmc3k_v0.cxg/var/__meta/__1576858533966_1576858533966_1362646d804b4982b35052a367633436
rename to server/test/fixtures/pbmc3k_v0.cxg/var/__meta/__1576858533966_1576858533966_1362646d804b4982b35052a367633436
diff --git a/server/test/schema.json b/server/test/fixtures/schema.json
similarity index 100%
rename from server/test/schema.json
rename to server/test/fixtures/schema.json
diff --git a/server/locust/README.md b/server/test/locust/README.md
similarity index 100%
rename from server/locust/README.md
rename to server/test/locust/README.md
diff --git a/server/test/locust/__init__.py b/server/test/locust/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/locust/config.py b/server/test/locust/config.py
similarity index 100%
rename from server/locust/config.py
rename to server/test/locust/config.py
diff --git a/server/locust/locustfile.py b/server/test/locust/locustfile.py
similarity index 99%
rename from server/locust/locustfile.py
rename to server/test/locust/locustfile.py
index 092f35f5..2e6a796f 100644
--- a/server/locust/locustfile.py
+++ b/server/test/locust/locustfile.py
@@ -4,7 +4,7 @@ import random
import json
from gevent.pool import Group
-import server.test.decode_fbs as decode_fbs
+import server.test.unit.decode_fbs as decode_fbs
from config import DataSets
diff --git a/server/locust/requirements-locust.txt b/server/test/locust/requirements-locust.txt
similarity index 100%
rename from server/locust/requirements-locust.txt
rename to server/test/locust/requirements-locust.txt
diff --git a/server/test/performance/__init__.py b/server/test/performance/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/create_test_matrix.py b/server/test/performance/create_test_matrix.py
similarity index 100%
rename from server/test/create_test_matrix.py
rename to server/test/performance/create_test_matrix.py
diff --git a/server/test/run_diffexp.py b/server/test/performance/run_diffexp.py
similarity index 100%
rename from server/test/run_diffexp.py
rename to server/test/performance/run_diffexp.py
diff --git a/server/test/test_database/__init__.py b/server/test/test_database/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_database/test_database.py b/server/test/test_database/test_database.py
new file mode 100644
index 00000000..4ecb4f27
--- /dev/null
+++ b/server/test/test_database/test_database.py
@@ -0,0 +1,60 @@
+import unittest
+from server.db.cellxgene_orm import CellxGeneUser, CellxGeneDataset, Annotation
+from server.db.db_utils import DbUtils
+from server.test.fixtures.database import TestDatabase
+
+
+class DatabaseTest(unittest.TestCase):
+ db = DbUtils("postgresql://postgres:test_pw@localhost:5432")
+
+ @classmethod
+ def setUpClass(cls) -> None:
+ TestDatabase()
+
+ @classmethod
+ def tearDownClass(cls) -> None:
+ del cls.db
+
+ def test_user_creation(self):
+ one_user = self.db.get(table=CellxGeneUser, entity_id='test_user_id')
+ self.assertEqual(one_user.id, 'test_user_id')
+ user_count = self.db.session.query(CellxGeneUser).count()
+ self.assertGreater(user_count, 10)
+
+ def test_dataset_creation(self):
+ one_dataset = self.db.query(table_args=[CellxGeneDataset],
+ filter_args=[CellxGeneDataset.name == 'test_dataset'])
+ self.assertEqual(one_dataset[0].name, 'test_dataset')
+ dataset_count = self.db.session.query(CellxGeneDataset).count()
+ self.assertGreater(dataset_count, 10)
+
+ def test_annotation_creation(self):
+ one_annotation = self.db.query(table_args=[Annotation], filter_args=[Annotation.tiledb_uri == 'tiledb_uri'])[0]
+ self.assertEqual(one_annotation.tiledb_uri, 'tiledb_uri')
+ annotation_count = self.db.session.query(Annotation).count()
+ self.assertGreater(annotation_count, 10)
+
+ def test_get_most_recent_annotation_for_user_dataset(self):
+ dataset_id = str(self.db.query(table_args=[CellxGeneDataset],
+ filter_args=[CellxGeneDataset.name == 'test_dataset'])[0].id)
+
+ # have to commit separately because created_at time written on the db server
+ self.db.session.add(Annotation(dataset_id=dataset_id, user_id='test_user_id', tiledb_uri='tiledb_uri_0'))
+ self.db.session.commit()
+
+ self.db.session.add(Annotation(dataset_id=dataset_id, user_id='test_user_id', tiledb_uri='tiledb_uri_1'))
+ self.db.session.commit()
+
+ self.db.session.add(Annotation(dataset_id=dataset_id, user_id='test_user_id', tiledb_uri='tiledb_uri_2'))
+ self.db.session.commit()
+
+ self.db.session.add(Annotation(dataset_id=dataset_id, user_id='test_user_id', tiledb_uri='tiledb_uri_3'))
+ self.db.session.commit()
+
+ self.db.session.add(Annotation(dataset_id=dataset_id, user_id='test_user_id', tiledb_uri='tiledb_uri_4'))
+ self.db.session.commit()
+
+ most_recent_annotation = self.db.query_for_most_recent(Annotation, [Annotation.dataset_id == dataset_id,
+ Annotation.user_id == 'test_user_id'])
+
+ self.assertEqual(most_recent_annotation.tiledb_uri, 'tiledb_uri_4')
diff --git a/server/test/unit/__init__.py b/server/test/unit/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/unit/auth/__init__.py b/server/test/unit/auth/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/unit/auth/test_auth.py b/server/test/unit/auth/test_auth.py
new file mode 100644
index 00000000..5d6eb748
--- /dev/null
+++ b/server/test/unit/auth/test_auth.py
@@ -0,0 +1,147 @@
+import unittest
+
+import requests
+
+from server.common.app_config import AppConfig
+from server.test import FIXTURES_ROOT, test_server
+
+
+class AuthTest(unittest.TestCase):
+ def setUp(self):
+ self.dataset_dataroot = FIXTURES_ROOT
+
+ def test_auth_none(self):
+ c = AppConfig()
+ c.update_server_config(
+ authentication__type=None, multi_dataset__dataroot=self.dataset_dataroot
+ )
+ c.update_default_dataset_config(user_annotations__enable=False)
+
+ c.complete_config()
+
+ with test_server(app_config=c) as server:
+ session = requests.Session()
+ r = session.get(f"{server}/d/pbmc3k.cxg/api/v0.2/config")
+ data_config = r.json()
+ assert "authentication" not in data_config["config"]
+
+ def test_auth_session(self):
+ c = AppConfig()
+ c.update_server_config(
+ authentication__type="session", multi_dataset__dataroot=self.dataset_dataroot
+ )
+ c.update_default_dataset_config(user_annotations__enable=True)
+ c.complete_config()
+
+ with test_server(app_config=c) as server:
+ session = requests.Session()
+ r = session.get(f"{server}/d/pbmc3k.cxg/api/v0.2/config")
+ data_config = r.json()
+ assert data_config["config"]["authentication"]["is_authenticated"]
+ assert not data_config["config"]["authentication"]["requires_client_login"]
+ assert data_config["config"]["authentication"]["username"] == "anonymous"
+
+ def test_auth_test(self):
+ c = AppConfig()
+ c.update_server_config(authentication__type="test")
+ c.update_server_config(
+ multi_dataset__dataroot=dict(
+ a1=dict(dataroot=self.dataset_dataroot, base_url="auth"),
+ a2=dict(dataroot=self.dataset_dataroot, base_url="no-auth"),
+ )
+ )
+
+ # specialize the configs
+ c.add_dataroot_config("a1", app__authentication_enable=True, user_annotations__enable=True)
+ c.add_dataroot_config("a2", app__authentication_enable=False, user_annotations__enable=False)
+
+ c.complete_config()
+
+ with test_server(app_config=c) as server:
+ session = requests.Session()
+
+ # auth datasets
+ r = session.get(f"{server}/auth/pbmc3k.cxg/api/v0.2/config")
+ data_config = r.json()
+ assert not data_config["config"]["authentication"]["is_authenticated"]
+ assert data_config["config"]["authentication"]["requires_client_login"]
+ assert data_config["config"]["authentication"]["username"] is None
+ assert data_config["config"]["parameters"]["annotations"]
+
+ login_uri = data_config["config"]["authentication"]["login"]
+ logout_uri = data_config["config"]["authentication"]["logout"]
+
+ assert login_uri == "/login?dataset=auth/pbmc3k.cxg"
+ assert logout_uri == "/logout?dataset=auth/pbmc3k.cxg"
+
+ r = session.get(f"{server}/{login_uri}")
+ # check that the login redirect worked
+ assert r.history[0].status_code == 302
+ assert r.url == f"{server}/auth/pbmc3k.cxg/"
+
+ r = session.get(f"{server}/auth/pbmc3k.cxg/api/v0.2/config")
+ data_config = r.json()
+ assert data_config["config"]["authentication"]["is_authenticated"]
+ assert data_config["config"]["authentication"]["username"] == "test_account"
+ assert data_config["config"]["parameters"]["annotations"]
+
+ r = session.get(f"{server}/{logout_uri}")
+ # check that the logout redirect worked
+ assert r.history[0].status_code == 302
+ assert r.url == f"{server}/auth/pbmc3k.cxg/"
+ r = session.get(f"{server}/auth/pbmc3k.cxg/api/v0.2/config")
+ data_config = r.json()
+ assert not data_config["config"]["authentication"]["is_authenticated"]
+ assert data_config["config"]["authentication"]["username"] is None
+ assert data_config["config"]["parameters"]["annotations"]
+
+ # no-auth datasets
+ r = session.get(f"{server}/no-auth/pbmc3k.cxg/api/v0.2/config")
+ data_config = r.json()
+ assert "authentication" not in data_config["config"]
+ assert not data_config["config"]["parameters"]["annotations"]
+
+ def test_auth_test_single(self):
+ c = AppConfig()
+ c.update_server_config(
+ authentication__type="test",
+ single_dataset__datapath=f"{self.dataset_dataroot}/pbmc3k.cxg")
+
+ c.complete_config()
+
+ with test_server(app_config=c) as server:
+ session = requests.Session()
+
+ r = session.get(f"{server}/api/v0.2/config")
+ data_config = r.json()
+ assert not data_config["config"]["authentication"]["is_authenticated"]
+ assert data_config["config"]["authentication"]["requires_client_login"]
+ assert data_config["config"]["authentication"]["username"] is None
+ assert data_config["config"]["parameters"]["annotations"]
+
+ login_uri = data_config["config"]["authentication"]["login"]
+ logout_uri = data_config["config"]["authentication"]["logout"]
+
+ assert login_uri == "/login"
+ assert logout_uri == "/logout"
+
+ r = session.get(f"{server}/{login_uri}")
+ # check that the login redirect worked
+ assert r.history[0].status_code == 302
+ assert r.url == f"{server}/"
+
+ r = session.get(f"{server}/api/v0.2/config")
+ data_config = r.json()
+ assert data_config["config"]["authentication"]["is_authenticated"]
+ assert data_config["config"]["authentication"]["username"] == "test_account"
+ assert data_config["config"]["parameters"]["annotations"]
+
+ r = session.get(f"{server}/{logout_uri}")
+ # check that the logout redirect worked
+ assert r.history[0].status_code == 302
+ assert r.url == f"{server}/"
+ r = session.get(f"{server}/api/v0.2/config")
+ data_config = r.json()
+ assert not data_config["config"]["authentication"]["is_authenticated"]
+ assert data_config["config"]["authentication"]["username"] is None
+ assert data_config["config"]["parameters"]["annotations"]
diff --git a/server/test/unit/cli/__init__.py b/server/test/unit/cli/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_cli_prepare.py b/server/test/unit/cli/test_prepare.py
similarity index 100%
rename from server/test/test_cli_prepare.py
rename to server/test/unit/cli/test_prepare.py
diff --git a/server/test/test_cli_upgrade.py b/server/test/unit/cli/test_upgrade.py
similarity index 100%
rename from server/test/test_cli_upgrade.py
rename to server/test/unit/cli/test_upgrade.py
diff --git a/server/test/unit/common/__init__.py b/server/test/unit/common/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_api.py b/server/test/unit/common/test_api.py
similarity index 85%
rename from server/test/test_api.py
rename to server/test/unit/common/test_api.py
index 4db3aea5..3e2c79b9 100644
--- a/server/test/test_api.py
+++ b/server/test/unit/common/test_api.py
@@ -2,28 +2,25 @@ import shutil
import time
import unittest
from http import HTTPStatus
-from subprocess import Popen
import pandas as pd
import requests
-import server.test.decode_fbs as decode_fbs
+import server.test.unit.decode_fbs as decode_fbs
from server.data_common.matrix_loader import MatrixDataType
-from server.test import data_with_tmp_annotations, make_fbs, PROJECT_ROOT
-from server.test.test_datasets.fixtures import pbmc3k_colors
-
+from server.test import (data_with_tmp_annotations, make_fbs, PROJECT_ROOT, FIXTURES_ROOT, start_test_server,
+ stop_test_server)
+from server.test.fixtures.fixtures import pbmc3k_colors
BAD_FILTER = {"filter": {"obs": {"annotation_value": [{"name": "xyz"}]}}}
+
# TODO (mweiden): remove ANNOTATIONS_ENABLED and Annotation subclasses when annotations are no longer experimental
class EndPoints(object):
ANNOTATIONS_ENABLED = True
- def setUp(self):
- self.session = requests.Session()
-
def test_initialize(self):
endpoint = "schema"
url = f"{self.URL_BASE}{endpoint}"
@@ -66,6 +63,34 @@ class EndPoints(object):
self.assertIsNone(df["row_idx"])
self.assertEqual(len(df["columns"]), df["n_cols"])
+ def test_put_layout_fbs(self):
+ # first check that re-embedding is turned on
+ result = self.session.get(f"{self.URL_BASE}config")
+ config_data = result.json()
+ re_embed = config_data["config"]["parameters"]["enable-reembedding"]
+ if not re_embed:
+ return
+ # attempt to reembed with umap over 100 cells.
+ endpoint = "layout/obs"
+ url = f"{self.URL_BASE}{endpoint}"
+ data = {}
+ data["filter"] = {}
+ data["filter"]["obs"] = {}
+ data["filter"]["obs"]["index"] = list(range(100))
+ data["method"] = "umap"
+ result = self.session.put(url, json=data)
+
+ self.assertEqual(result.status_code, HTTPStatus.OK)
+ result_data = result.json()
+ self.assertIsInstance(result_data, dict)
+ self.assertEqual(result_data["type"], "float32")
+ self.assertTrue(result_data["name"].startswith("reembed:umap_"))
+ self.assertIsInstance(result_data["dims"], list)
+ self.assertEqual(len(result_data["dims"]), 2)
+ dims = result_data["dims"]
+ self.assertTrue(dims[0].startswith("reembed:umap_") and dims[0].endswith("_0"))
+ self.assertTrue(dims[1].startswith("reembed:umap_") and dims[1].endswith("_1"))
+
def test_bad_filter(self):
endpoint = "data/var"
url = f"{self.URL_BASE}{endpoint}"
@@ -282,13 +307,13 @@ class EndPoints(object):
def test_static(self):
endpoint = "static"
file = "assets/favicon.ico"
- url = f"{self.LOCAL_URL}{endpoint}/{file}"
+ url = f"{self.server}/{endpoint}/{file}"
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.OK)
- @staticmethod
- def _setUpClass(child_class, start_command):
- child_class.ps = Popen(start_command)
+ def _setupClass(child_class, command_line):
+ child_class.ps, child_class.server = start_test_server(command_line)
+ child_class.URL_BASE = f"{child_class.server}/api/v0.2/"
child_class.session = requests.Session()
for i in range(90):
try:
@@ -297,13 +322,6 @@ class EndPoints(object):
except requests.exceptions.ConnectionError:
time.sleep(1)
- @staticmethod
- def _tearDownClass(child_class):
- try:
- child_class.ps.terminate()
- except ProcessLookupError:
- pass
-
class EndPointsAnnotations(EndPoints):
def test_get_schema_existing_writable(self):
@@ -359,31 +377,19 @@ class EndPointsAnnotations(EndPoints):
class EndPointsAnndata(unittest.TestCase, EndPoints):
"""Test Case for endpoints"""
- PORT = 5010
- LOCAL_URL = f"http://127.0.0.1:{PORT}/"
- VERSION = "v0.2"
- URL_BASE = f"{LOCAL_URL}api/{VERSION}/"
ANNOTATIONS_ENABLED = False
@classmethod
def setUpClass(cls):
- cls._setUpClass(
- cls,
- [
- "cellxgene",
- "--no-upgrade-check",
- "launch",
- f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad",
- "--disable-annotations",
- "--verbose",
- "--port",
- str(cls.PORT),
- ],
- )
+ cls._setupClass(cls, [
+ f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad",
+ "--disable-annotations",
+ "--experimental-enable-reembedding",
+ ])
@classmethod
def tearDownClass(cls):
- cls._tearDownClass(cls)
+ stop_test_server(cls.ps)
@property
def annotations_enabled(self):
@@ -393,98 +399,57 @@ class EndPointsAnndata(unittest.TestCase, EndPoints):
class EndPointsCxg(unittest.TestCase, EndPoints):
"""Test Case for endpoints"""
- PORT = 5011
- LOCAL_URL = f"http://127.0.0.1:{PORT}/"
- VERSION = "v0.2"
- URL_BASE = f"{LOCAL_URL}api/{VERSION}/"
ANNOTATIONS_ENABLED = False
@classmethod
def setUpClass(cls):
- cls._setUpClass(
- cls,
- [
- "cellxgene",
- "--no-upgrade-check",
- "launch",
- f"{PROJECT_ROOT}/server/test/test_datasets/pbmc3k.cxg",
- "--disable-annotations",
- "--verbose",
- "--port",
- str(cls.PORT),
- ],
- )
+ cls._setupClass(cls, [
+ f"{FIXTURES_ROOT}/pbmc3k.cxg",
+ "--disable-annotations",
+ ])
@classmethod
def tearDownClass(cls):
- cls._tearDownClass(cls)
+ stop_test_server(cls.ps)
class EndPointsAnndataAnnotations(unittest.TestCase, EndPointsAnnotations):
"""Test Case for endpoints"""
- PORT = 5012
- LOCAL_URL = f"http://127.0.0.1:{PORT}/"
- VERSION = "v0.2"
- URL_BASE = f"{LOCAL_URL}api/{VERSION}/"
ANNOTATIONS_ENABLED = True
- MATRIX_DATA_TYPE = MatrixDataType.H5AD
@classmethod
def setUpClass(cls):
cls.data, cls.tmp_dir, cls.annotations = data_with_tmp_annotations(
MatrixDataType.H5AD, annotations_fixture=True
)
- cls._setUpClass(
- cls,
- [
- "cellxgene",
- "--no-upgrade-check",
- "launch",
- "--annotations-file",
- cls.annotations.output_file,
- "--verbose",
- "--port",
- str(cls.PORT),
- cls.data.get_location(),
- ],
- )
+ cls._setupClass(cls, [
+ "--annotations-file",
+ cls.annotations.output_file,
+ cls.data.get_location(),
+ ])
@classmethod
def tearDownClass(cls):
shutil.rmtree(cls.tmp_dir)
- cls._tearDownClass(cls)
+ stop_test_server(cls.ps)
class EndPointsCxgAnnotations(unittest.TestCase, EndPointsAnnotations):
"""Test Case for endpoints"""
- PORT = 5013
- LOCAL_URL = f"http://127.0.0.1:{PORT}/"
- VERSION = "v0.2"
- URL_BASE = f"{LOCAL_URL}api/{VERSION}/"
ANNOTATIONS_ENABLED = True
- MATRIX_DATA_TYPE = MatrixDataType.CXG
@classmethod
def setUpClass(cls):
cls.data, cls.tmp_dir, cls.annotations = data_with_tmp_annotations(MatrixDataType.CXG, annotations_fixture=True)
- cls._setUpClass(
- cls,
- [
- "cellxgene",
- "--no-upgrade-check",
- "launch",
- "--annotations-file",
- cls.annotations.output_file,
- "--verbose",
- "--port",
- str(cls.PORT),
- cls.data.get_location(),
- ],
- )
+ cls._setupClass(cls, [
+ "--annotations-file",
+ cls.annotations.output_file,
+ cls.data.get_location(),
+ ])
@classmethod
def tearDownClass(cls):
shutil.rmtree(cls.tmp_dir)
- cls._tearDownClass(cls)
+ stop_test_server(cls.ps)
diff --git a/server/test/test_app_config.py b/server/test/unit/common/test_app_config.py
similarity index 69%
rename from server/test/test_app_config.py
rename to server/test/unit/common/test_app_config.py
index c55f927a..a299bf51 100644
--- a/server/test/test_app_config.py
+++ b/server/test/unit/common/test_app_config.py
@@ -1,12 +1,21 @@
+import os
import unittest
+from unittest import mock
+from unittest.mock import patch
+
+import requests
+
from server.common.app_config import AppConfig
from server.common.errors import ConfigurationError
-from server.test import PROJECT_ROOT, test_server
-import requests
+from server.test import PROJECT_ROOT, test_server, FIXTURES_ROOT
+
# NOTE, there are more tests that should be written for AppConfig.
# this is just a start.
+def mockenv(**envvars):
+ return mock.patch.dict(os.environ, envvars)
+
class AppConfigTest(unittest.TestCase):
def test_update(self):
@@ -52,8 +61,8 @@ class AppConfigTest(unittest.TestCase):
c.update_server_config(
multi_dataset__dataroot=dict(
s1=dict(dataroot=f"{PROJECT_ROOT}/example-dataset", base_url="set1/1/2"),
- s2=dict(dataroot=f"{PROJECT_ROOT}/server/test/test_datasets", base_url="set2"),
- s3=dict(dataroot=f"{PROJECT_ROOT}/server/test/test_datasets", base_url="set3"),
+ s2=dict(dataroot=f"{FIXTURES_ROOT}", base_url="set2"),
+ s3=dict(dataroot=f"{FIXTURES_ROOT}", base_url="set3"),
)
)
@@ -98,3 +107,29 @@ class AppConfigTest(unittest.TestCase):
r = session.get(f"{server}/health")
assert r.json()["status"] == "pass"
+
+ @mockenv(CXG_AWS_SECRET_NAME="TESTING", CXG_AWS_SECRET_REGION_NAME="TEST_REGION")
+ @patch('server.common.aws_secret_utils.get_secret_key')
+ def test_get_config_vars_from_aws_secrets(self, mock_get_secret_key):
+ mock_get_secret_key.return_value = {
+ "flask_secret_key": "mock_flask_secret",
+ "oauth_client_secret": "mock_oauth_secret",
+ "db_uri": "mock_db_uri"
+ }
+
+ config = AppConfig()
+
+ with self.assertLogs(level="INFO") as logger:
+ from server.common.aws_secret_utils import handle_config_from_secret
+ # should not throw error
+ # "AttributeError: 'XConfig' object has no attribute 'x'"
+ handle_config_from_secret(config)
+
+ # should log 3 lines (one for each var set from a secret)
+ self.assertEqual(len(logger.output), 3)
+ self.assertIn('INFO:root:set app__flask_secret_key from secret', logger.output[0])
+ self.assertIn('INFO:root:set authentication__params_oauth__client_secret from secret', logger.output[1])
+ self.assertIn('INFO:root:set user_annotations__hosted_tiledb_array__db_uri from secret', logger.output[2])
+ self.assertEqual(config.server_config.app__flask_secret_key, "mock_flask_secret")
+ self.assertEqual(config.server_config.authentication__params_oauth__client_secret, "mock_oauth_secret")
+ self.assertEqual(config.default_dataset_config.user_annotations__hosted_tiledb_array__db_uri, "mock_db_uri")
diff --git a/server/test/test_colors.py b/server/test/unit/common/test_colors.py
similarity index 96%
rename from server/test/test_colors.py
rename to server/test/unit/common/test_colors.py
index 2970587c..efef25ec 100644
--- a/server/test/test_colors.py
+++ b/server/test/unit/common/test_colors.py
@@ -4,7 +4,7 @@ import anndata
from server.common.colors import convert_color_to_hex_format, convert_anndata_category_colors_to_cxg_category_colors
from server.common.errors import ColorFormatException
from server.test import PROJECT_ROOT
-from server.test.test_datasets.fixtures import pbmc3k_colors
+from server.test.fixtures.fixtures import pbmc3k_colors
class ColorsTest(unittest.TestCase):
diff --git a/server/test/unit/common/test_corpora.py b/server/test/unit/common/test_corpora.py
new file mode 100644
index 00000000..f502244c
--- /dev/null
+++ b/server/test/unit/common/test_corpora.py
@@ -0,0 +1,146 @@
+import unittest
+import anndata
+import json
+import tempfile
+import shutil
+from http import HTTPStatus
+import requests
+
+from server.common.corpora import (
+ corpora_get_versions_from_anndata,
+ corpora_is_version_supported,
+ corpora_get_props_from_anndata,
+)
+from server.test import PROJECT_ROOT, start_test_server, stop_test_server
+
+VERSION = "v0.2"
+
+
+class CorporaAPITest(unittest.TestCase):
+ def test_corpora_get_versions_from_anndata(self):
+ adata = self._get_h5ad()
+
+ if "version" in adata.uns:
+ del adata.uns["version"]
+ self.assertIsNone(corpora_get_versions_from_anndata(adata))
+
+ # something bogus
+ adata.uns["version"] = 99
+ self.assertIsNone(corpora_get_versions_from_anndata(adata))
+
+ # something legit
+ adata.uns["version"] = {"corpora_schema_version": "0.0.0", "corpora_encoding_version": "9.9.9"}
+ self.assertEqual(corpora_get_versions_from_anndata(adata), ["0.0.0", "9.9.9"])
+
+ def test_corpora_is_version_supported(self):
+ self.assertTrue(corpora_is_version_supported("1.0.0", "0.1.0"))
+ self.assertFalse(corpora_is_version_supported("0.0.0", "0.1.0"))
+ self.assertFalse(corpora_is_version_supported("1.0.0", "0.0.0"))
+
+ def test_corpora_get_props_from_anndata(self):
+ adata = self._get_h5ad()
+
+ if "version" in adata.uns:
+ del adata.uns["version"]
+ self.assertIsNone(corpora_get_props_from_anndata(adata))
+
+ # something bogus
+ adata.uns["version"] = 99
+ self.assertIsNone(corpora_get_props_from_anndata(adata))
+
+ # unsupported version, but missing required values
+ adata.uns["version"] = {"corpora_schema_version": "99.0.0", "corpora_encoding_version": "32.1.0"}
+ with self.assertRaises(ValueError):
+ corpora_get_props_from_anndata(adata)
+
+ # legit version, but missing required values
+ adata.uns["version"] = {"corpora_schema_version": "1.0.0", "corpora_encoding_version": "0.1.0"}
+ with self.assertRaises(KeyError):
+ corpora_get_props_from_anndata(adata)
+
+ some_fields = {
+ "version": {"corpora_schema_version": "1.0.0", "corpora_encoding_version": "0.1.0"},
+ "title": "title",
+ "layer_descriptions": "layer_descriptions",
+ "organism": "organism",
+ "organism_ontology_term_id": "organism_ontology_term_id",
+ "project_name": "project_name",
+ "project_description": "project_description",
+ "contributors": json.dumps([{"contributors": "contributors"}]),
+ "project_links": json.dumps([{"link_name": "link_name", "link_url": "link_url", "link_type": "SUMMARY"}]),
+ }
+ for k in some_fields:
+ adata.uns[k] = some_fields[k]
+ some_fields["contributors"] = json.loads(some_fields["contributors"])
+ some_fields["project_links"] = json.loads(some_fields["project_links"])
+ self.assertEqual(corpora_get_props_from_anndata(adata), some_fields)
+
+ def _get_h5ad(self):
+ return anndata.read_h5ad(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad")
+
+
+class CorporaRESTAPITest(unittest.TestCase):
+ """ Confirm endpoints reflect Corpora-specific features """
+
+ @classmethod
+ def setCorporaFields(cls, path):
+ adata = anndata.read_h5ad(path)
+ corpora_props = {
+ "version": {
+ "corpora_schema_version": "1.0.0",
+ "corpora_encoding_version": "0.1.0"
+ },
+ "title": "PBMC3K",
+ "contributors": json.dumps([
+ {"name": "name"}
+ ]),
+ "layer_descriptions": {
+ "X": "raw counts"
+ },
+ "organism": "human",
+ "organism_ontology_term_id": "unknown",
+ "project_name": "test project",
+ "project_description": "test description",
+ "project_links": json.dumps([
+ {"link_name": "test link", "link_type": "SUMMARY", "link_url": "https://a.u.r.l/"}
+ ]),
+ "default_embedding": "X_tsne"
+ }
+ adata.uns.update(corpora_props)
+ adata.write(path)
+
+ @classmethod
+ def setUpClass(cls):
+ cls.tmp_dir = tempfile.TemporaryDirectory()
+ src = f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad"
+ dst = f"{cls.tmp_dir.name}/pbmc3k.h5ad"
+ shutil.copyfile(src, dst)
+ cls.setCorporaFields(dst)
+ cls.ps, cls.server = start_test_server([dst])
+
+ @classmethod
+ def tearDownClass(cls):
+ stop_test_server(cls.ps)
+ cls.tmp_dir.cleanup()
+
+ def setUp(self):
+ self.session = requests.Session()
+ self.url_base = f"{self.server}/api/{VERSION}/"
+
+ def test_config(self):
+ endpoint = "config"
+ url = f"{self.url_base}{endpoint}"
+ result = self.session.get(url)
+ self.assertEqual(result.status_code, HTTPStatus.OK)
+ self.assertEqual(result.headers["Content-Type"], "application/json")
+
+ result_data = result.json()
+ self.assertIsInstance(result_data["config"]["corpora_props"], dict)
+ self.assertIsInstance(result_data["config"]["parameters"], dict)
+
+ corpora_props = result_data["config"]["corpora_props"]
+ parameters = result_data["config"]["parameters"]
+
+ self.assertEqual(corpora_props["version"]["corpora_schema_version"], "1.0.0")
+ self.assertEqual(corpora_props["organism"], "human")
+ self.assertEqual(parameters["default_embedding"], "tsne")
diff --git a/server/test/test_nan_rest.py b/server/test/unit/common/test_nan_rest.py
similarity index 69%
rename from server/test/test_nan_rest.py
rename to server/test/unit/common/test_nan_rest.py
index 8e958eeb..1dd28a54 100644
--- a/server/test/test_nan_rest.py
+++ b/server/test/unit/common/test_nan_rest.py
@@ -1,17 +1,13 @@
from http import HTTPStatus
-from subprocess import Popen
import unittest
-import time
import math
+from server.test import start_test_server, stop_test_server
-import server.test.decode_fbs as decode_fbs
+import server.test.unit.decode_fbs as decode_fbs
import requests
-LOCAL_URL = "http://127.0.0.1:5006/"
VERSION = "v0.2"
-URL_BASE = f"{LOCAL_URL}api/{VERSION}/"
-
BAD_FILTER = {"filter": {"obs": {"annotation_value": [{"name": "xyz"}]}}}
@@ -20,33 +16,25 @@ class WithNaNs(unittest.TestCase):
@classmethod
def setUpClass(cls):
- cls.ps = Popen(["cellxgene", "launch", "test/test_datasets/nan.h5ad", "--verbose", "--port", "5006"])
- session = requests.Session()
- for i in range(90):
- try:
- session.get(f"{URL_BASE}schema")
- except requests.exceptions.ConnectionError:
- time.sleep(1)
+ cls.ps, cls.server = start_test_server(["test/fixtures/nan.h5ad"])
@classmethod
def tearDownClass(cls):
- try:
- cls.ps.terminate()
- except ProcessLookupError:
- pass
+ stop_test_server(cls.ps)
def setUp(self):
self.session = requests.Session()
+ self.url_base = f"{self.server}/api/{VERSION}/"
def test_initialize(self):
endpoint = "schema"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.url_base}{endpoint}"
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.OK)
def test_data(self):
endpoint = "data/var"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.url_base}{endpoint}"
filter = {"filter": {"var": {"index": [[0, 20]]}}}
result = self.session.put(url, json=filter)
self.assertEqual(result.status_code, HTTPStatus.OK)
@@ -56,7 +44,7 @@ class WithNaNs(unittest.TestCase):
def test_annotation_obs(self):
endpoint = "annotations/obs"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.url_base}{endpoint}"
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.headers["Content-Type"], "application/octet-stream")
@@ -65,7 +53,7 @@ class WithNaNs(unittest.TestCase):
def test_annotation_var(self):
endpoint = "annotations/var"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.url_base}{endpoint}"
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.headers["Content-Type"], "application/octet-stream")
diff --git a/server/test/test_filter.py b/server/test/unit/common/test_rest.py
similarity index 100%
rename from server/test/test_filter.py
rename to server/test/unit/common/test_rest.py
diff --git a/server/test/test_writable_annotation.py b/server/test/unit/common/test_writable_annotation.py
similarity index 52%
rename from server/test/test_writable_annotation.py
rename to server/test/unit/common/test_writable_annotation.py
index 6ca0419d..45e09de3 100644
--- a/server/test/test_writable_annotation.py
+++ b/server/test/unit/common/test_writable_annotation.py
@@ -1,15 +1,146 @@
import json
from os import path, listdir
import unittest
-import server.test.decode_fbs as decode_fbs
+from unittest.mock import MagicMock, patch
+
+import tiledb
+from flask import Flask
+
+import server.test.unit.decode_fbs as decode_fbs
import shutil
import numpy as np
import pandas as pd
from server.common.rest import schema_get_helper, annotations_put_fbs_helper
-from server.test import data_with_tmp_annotations, make_fbs
+from server.db.cellxgene_orm import CellxGeneDataset, Annotation
+from server.test import data_with_tmp_annotations, make_fbs, data_with_tmp_tiledb_annotations
from server.data_common.matrix_loader import MatrixDataType
+from server.common.errors import AnnotationCategoryNameError
+
+
+class auth(object):
+ def get_user_id():
+ return "1234"
+
+
+class WritableTileDBStoredAnnotationTest(unittest.TestCase):
+ def setUp(self):
+ self.user_id = '1234'
+ self.data, self.tmp_dir, self.annotations = data_with_tmp_tiledb_annotations(MatrixDataType.H5AD)
+ self.data.dataset_config.user_annotations = self.annotations
+ self.db = self.annotations.db
+ self.n_rows = self.data.get_shape()[0]
+ self.test_dict = {
+ "cat_A": pd.Series(["label_A"] * self.n_rows, dtype="category"),
+ "cat_B": pd.Series(["label_B"] * self.n_rows, dtype="category"),
+ }
+ self.fbs = make_fbs(self.test_dict)
+ self.df = pd.DataFrame(self.test_dict)
+ self.app = Flask('fake_app')
+ self.app.__setattr__("auth", auth)
+
+ def tearDown(self):
+ shutil.rmtree(self.tmp_dir)
+
+ def annotation_put_fbs(self, fbs):
+ annotations_put_fbs_helper(self.data, fbs)
+ res = json.dumps({"status": "OK"})
+ return res
+
+ def test_category_name_throws_errors_for_categories_that_cant_be_converted_to_filenames(self):
+ with self.app.test_request_context():
+ bad_category_names = make_fbs(
+ {
+ "cat_A": pd.Series(["label_A"] * self.n_rows, dtype="category"),
+ "cat/B": pd.Series(["label_B"] * self.n_rows, dtype="category"),
+ }
+ )
+ with self.assertRaises(AnnotationCategoryNameError):
+ self.annotation_put_fbs(bad_category_names)
+
+ def test_convert_to_pandas__converts_tiledb_to_pandas_df(self):
+ with self.app.test_request_context():
+ self.annotations.write_labels(self.df, self.data)
+ dataset_id = self.db.query([CellxGeneDataset], [CellxGeneDataset.name == self.data.get_location()])[0].id
+ annotation = self.db.query_for_most_recent(
+ Annotation,
+ [Annotation.user_id == self.user_id, Annotation.dataset_id == str(dataset_id)]
+ )
+ # retrieve tiledb array
+ df = tiledb.open(annotation.tiledb_uri)
+ self.assertEqual(type(df), tiledb.array.SparseArray)
+
+ # convert to pandas df
+ pandas_df = self.annotations.convert_to_pandas_df(df)
+ self.assertEqual(type(pandas_df), pd.DataFrame)
+
+ def test_write_labels_creates_a_dataset_if_it_doesnt_exist(self):
+ with self.app.test_request_context():
+
+ new_name = 'new_dataset/location'
+ self.data.get_location = MagicMock(return_value=new_name)
+ num_datasets = len(self.db.query([CellxGeneDataset]))
+ self.annotation_put_fbs(self.fbs)
+ more_datasets = len(self.db.query([CellxGeneDataset]))
+ self.assertGreater(more_datasets, num_datasets)
+
+ self.assertGreater(len(self.db.query([CellxGeneDataset], [CellxGeneDataset.name == new_name])), 0)
+
+ def test_write_labels_links_to_existing_dataset(self):
+ with self.app.test_request_context():
+ # add dataset to to db
+ self.annotation_put_fbs(self.fbs)
+
+ num_datasets = len(self.db.query([CellxGeneDataset]))
+
+ # create another annotation with the same dataset
+ self.annotation_put_fbs(self.fbs)
+
+ same_num_datasets = len(self.db.query([CellxGeneDataset]))
+
+ self.assertEqual(num_datasets, same_num_datasets)
+
+ def test_read_labels_returns_pandas_df(self):
+ with self.app.test_request_context():
+ self.annotation_put_fbs(self.fbs)
+ pandas_df = self.annotations.read_labels(self.data)
+ self.assertEqual(type(pandas_df), pd.DataFrame)
+
+ def test_read_labels_returns_df_matching_original(self):
+ with self.app.test_request_context():
+ self.annotation_put_fbs(self.fbs)
+ pandas_df = self.annotations.read_labels(self.data)
+
+ self.assertEqual(pandas_df.shape, (self.n_rows, 2))
+ self.assertEqual(set(pandas_df.columns), {"cat_A", "cat_B"})
+ self.assertTrue(self.data.original_obs_index.equals(pandas_df.index))
+ self.assertTrue(np.all(pandas_df["cat_A"] == ["label_A"] * self.n_rows))
+ self.assertTrue(np.all(pandas_df["cat_B"] == ["label_B"] * self.n_rows))
+
+ def test_error_checks(self):
+ # verify that the expected errors are generated
+ with self.app.test_request_context():
+ n_rows = self.data.get_shape()[0]
+ fbs_bad = make_fbs({"louvain": pd.Series(["undefined"] * n_rows, dtype="category")})
+
+ # ensure we catch attempt to overwrite non-writable data
+ with self.assertRaises(KeyError):
+ self.annotation_put_fbs(fbs_bad)
+
+ @patch('server.common.annotations.hosted_tiledb.current_app')
+ def test_write_labels_stores_df_as_tiledb_array(self, mock_user_id):
+ mock_user_id.auth.get_user_id.return_value = '1234'
+ self.annotations.write_labels(self.df, self.data)
+ # get uri
+ dataset_id = self.db.query([CellxGeneDataset], [CellxGeneDataset.name == self.data.get_location()])[0].id
+ annotation = self.db.query_for_most_recent(
+ Annotation,
+ [Annotation.user_id == '1234', Annotation.dataset_id == str(dataset_id)]
+ )
+
+ df = tiledb.open(annotation.tiledb_uri)
+ self.assertEqual(type(df), tiledb.array.SparseArray)
class WritableAnnotationTest(unittest.TestCase):
diff --git a/server/test/unit/common/utils/__init__.py b/server/test/unit/common/utils/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/unit/common/utils/test_matrix_utils.py b/server/test/unit/common/utils/test_matrix_utils.py
new file mode 100644
index 00000000..ffda1045
--- /dev/null
+++ b/server/test/unit/common/utils/test_matrix_utils.py
@@ -0,0 +1,67 @@
+import unittest
+
+import numpy as np
+
+from server.common.utils.matrix_utils import is_matrix_sparse, get_column_shift_encode_for_matrix
+
+
+class TestMatrixUtils(unittest.TestCase):
+
+ def test__is_matrix_sparse__zero_and_one_hundred_percent_threshold(self):
+ matrix = np.array([1, 2, 3])
+
+ self.assertFalse(is_matrix_sparse(matrix, 0))
+ self.assertTrue(is_matrix_sparse(matrix, 100))
+
+ def test__is_matrix_sparse__partially_populated_sparse_matrix_returns_true(self):
+ matrix = np.zeros([3, 4])
+ matrix[2][3] = 1.0
+ matrix[1][1] = 2.2
+
+ self.assertTrue(is_matrix_sparse(matrix, 50))
+
+ def test__is_matrix_sparse__partially_populated_dense_matrix_returns_false(self):
+ matrix = np.zeros([2, 2])
+ matrix[0][0] = 1.0
+ matrix[0][1] = 2.2
+ matrix[1][1] = 3.7
+
+ self.assertFalse(is_matrix_sparse(matrix, 50))
+
+ def test__is_matrix_sparse__giant_matrix_returns_false_early(self):
+ matrix = np.ones([20000, 20])
+
+ with self.assertLogs(level="INFO") as logger:
+ self.assertFalse(is_matrix_sparse(matrix, 1))
+
+ # Because the function returns early a log will output the _estimate_ instead of the _exact_ percentage of
+ # non-zero elements in the matrix.
+ self.assertIn("Percentage of non-zero elements (estimate)", logger.output[0])
+
+ def test__is_matrix_sparse_with_column_shift_encoding__regular_sparse_returns_true(self):
+ matrix = np.zeros([2, 2])
+ matrix[0][0] = 1.0
+
+ self.assertIsNotNone(get_column_shift_encode_for_matrix(matrix, 50))
+
+ def test__is_matrix_sparse_with_column_shift_encoding__column_shift_returns_same_value(self):
+ matrix = np.ones([2, 2])
+ expected_column_shift = [1, 1]
+
+ actual_column_shift = get_column_shift_encode_for_matrix(matrix, 50)
+ self.assertTrue((expected_column_shift == actual_column_shift).all())
+
+ def test__is_matrix_sparse_with_column_shift_encoding__impossible_column_shift_returns_none(self):
+ matrix = np.array([[1, 2], [3, 4]])
+
+ self.assertIsNone(get_column_shift_encode_for_matrix(matrix, 50))
+
+ def test__is_matrix_sparse_with_column_shift_encoding__giant_matrix_returns_false_early(self):
+ matrix = np.random.rand(20000, 20)
+
+ with self.assertLogs(level="INFO") as logger:
+ self.assertFalse(is_matrix_sparse(matrix, 1))
+
+ # Because the function returns early a log will output the _estimate_ instead of the _exact_ percentage of
+ # non-zero elements in the matrix.
+ self.assertIn("Percentage of non-zero elements (estimate)", logger.output[0])
diff --git a/server/test/unit/common/utils/test_sanitization_utils.py b/server/test/unit/common/utils/test_sanitization_utils.py
new file mode 100644
index 00000000..8ef04218
--- /dev/null
+++ b/server/test/unit/common/utils/test_sanitization_utils.py
@@ -0,0 +1,56 @@
+import unittest
+
+from server.common.utils.sanitization_utils import sanitize_values_in_list, sanitize_keys_in_dictionary
+
+
+class TestSanitizationUtils(unittest.TestCase):
+
+ def test__sanitize_values_in_list__not_strings_raises_exception(self):
+ keys_to_sanitize = [1, 2, 3]
+
+ with self.assertRaises(Exception) as exception_context:
+ sanitize_values_in_list(keys_to_sanitize)
+
+ self.assertIn("must contain all strings", str(exception_context.exception))
+
+ def test__sanitize_values_in_list__not_all_strings_raises_exception(self):
+ keys_to_sanitize = ["1", "2", 3]
+
+ with self.assertRaises(Exception) as exception_context:
+ sanitize_values_in_list(keys_to_sanitize)
+
+ self.assertIn("must contain all strings", str(exception_context.exception))
+
+ def test__sanitize_values_in_list__replace_non_ascii_character_with_underscore(self):
+ keys_to_sanitize = ["abc.", "~abc", "a~b/c"]
+ expected_sanitized_keys_dict = dict(zip(keys_to_sanitize, ["abc_", "_abc", "a_b_c"]))
+
+ actual_sanitized_keys_dict = sanitize_values_in_list(keys_to_sanitize)
+
+ self.assertEqual(expected_sanitized_keys_dict, actual_sanitized_keys_dict)
+
+ def test__sanitize_keys_in_dictionary__replace_non_ascii_character_with_underscore(self):
+ dictionary_to_sanitize = {"abc.": 3, "~abc": 4, "a~b/c": 5}
+ expected_sanitized_dict = {"abc_": 3, "_abc": 4, "a_b_c": 5}
+
+ actual_sanitized_dict = dictionary_to_sanitize
+ sanitize_keys_in_dictionary(actual_sanitized_dict)
+
+ self.assertEqual(expected_sanitized_dict, actual_sanitized_dict)
+
+ def test__sanitize_keys_in_dictionary__non_string_key_raises_exception(self):
+ dictionary_to_sanitize = {4: 3, "~abc": 4, "a~b/c": 5}
+
+ with self.assertRaises(Exception) as exception_context:
+ sanitize_keys_in_dictionary(dictionary_to_sanitize)
+
+ self.assertIn("must contain all strings", str(exception_context.exception))
+
+ def test__sanitize_keys_in_dictionary__replace_only_some_keys(self):
+ dictionary_to_sanitize = {"abc": 3, "~abc": 4, "a~b/c": 5}
+ expected_sanitized_dict = {"abc": 3, "_abc": 4, "a_b_c": 5}
+
+ actual_sanitized_dict = dictionary_to_sanitize
+ sanitize_keys_in_dictionary(actual_sanitized_dict)
+
+ self.assertEqual(expected_sanitized_dict, actual_sanitized_dict)
diff --git a/server/test/unit/common/utils/test_type_conversion_utils.py b/server/test/unit/common/utils/test_type_conversion_utils.py
new file mode 100644
index 00000000..a5bb0395
--- /dev/null
+++ b/server/test/unit/common/utils/test_type_conversion_utils.py
@@ -0,0 +1,121 @@
+import unittest
+from unittest.mock import patch
+
+import numpy as np
+from pandas import Series
+
+from server.common.utils.type_conversion_utils import can_cast_to_float32, can_cast_to_int32, get_dtype_of_array, \
+ get_schema_type_hint_of_array
+
+
+class TestTypeConversionUtils(unittest.TestCase):
+
+ def test__can_cast_to_float32__string_is_false(self):
+ array_to_convert = Series(data=["1", "2", "3"], dtype=str)
+
+ can_cast = can_cast_to_float32(array_to_convert.dtype)
+
+ self.assertFalse(can_cast)
+
+ def test__can_cast_to_float32__int_is_true_warning_outputted(self):
+ array_to_convert = Series(data=[1, 2, 3], dtype=np.dtype(np.float64))
+
+ with self.assertLogs(level="WARN") as logger:
+ can_cast = can_cast_to_float32(array_to_convert.dtype)
+ self.assertIn("may lose precision", logger.output[0])
+
+ self.assertTrue(can_cast)
+
+ @patch("logging.warning")
+ def test__can_cast_to_float64__int_is_false(self, mock_log_warning):
+ array_to_convert = Series(data=[1, 2, 3], dtype=np.dtype(np.float32))
+
+ can_cast = can_cast_to_float32(array_to_convert.dtype)
+
+ self.assertTrue(can_cast)
+ assert not mock_log_warning.called
+
+ def test__can_cast_to_int32__string_is_false(self):
+ array_to_convert = Series(data=["1", "2", "3"], dtype=str)
+
+ can_cast = can_cast_to_int32(array_to_convert.dtype, array_to_convert)
+
+ self.assertFalse(can_cast)
+
+ def test__can_cast_to_int32__int64_is_true(self):
+ array_to_convert = Series(data=["1", "2", "3"], dtype=np.dtype(np.int64))
+
+ can_cast = can_cast_to_int32(array_to_convert.dtype, array_to_convert)
+
+ self.assertTrue(can_cast)
+
+ def test__can_cast_to_int32__int16_is_true(self):
+ array_to_convert = Series(data=["1", "2", "3"], dtype=np.dtype(np.int16))
+
+ can_cast = can_cast_to_int32(array_to_convert.dtype, array_to_convert)
+
+ self.assertTrue(can_cast)
+
+ def test__can_cast_to_int32__int64_with_large_value_is_false(self):
+ array_to_convert = Series(data=["3000000000", "2", "3"], dtype=np.dtype(np.int64))
+
+ can_cast = can_cast_to_int32(array_to_convert.dtype, array_to_convert)
+
+ self.assertFalse(can_cast)
+
+ def test__get_dtype_of_array__supported_dtypes_return_as_expected(self):
+ types = [np.float32, np.int32, np.bool_, str]
+ expected_dtypes = [np.float32, np.int32, np.uint8, np.unicode]
+
+ for test_type_index in range(len(types)):
+ with self.subTest(f"Testing get_dtype_of_array with type {types[test_type_index].__name__}",
+ i=test_type_index):
+ array = Series(data=[], dtype=types[test_type_index])
+ self.assertEqual(get_dtype_of_array(array), expected_dtypes[test_type_index])
+
+ def test__get_schema_type_hint_of_array__supported_dtypes_return_as_expected(self):
+ types = [np.float32, np.int32, np.bool_, str]
+ expected_schema_hints = [{"type": "float32"}, {"type": "int32"}, {"type": "boolean"}, {"type": "string"}]
+
+ for test_type_index in range(len(types)):
+ with self.subTest(f"Testing get_schema_type_hint_of_array with type {types[test_type_index].__name__}",
+ i=test_type_index):
+ array = Series(data=[], dtype=types[test_type_index])
+ self.assertEqual(get_schema_type_hint_of_array(array), expected_schema_hints[test_type_index])
+
+ def test__get_dtype_of_array__categories_return_as_expected(self):
+ array = Series(data=["a", "b", "c"], dtype="category")
+ expected_dtype = np.unicode
+
+ actual_dtype = get_dtype_of_array(array)
+
+ self.assertEqual(expected_dtype, actual_dtype)
+
+ def test__get_schema_type_hint_of_array__categories_return_as_expected(self):
+ array = Series(data=["a", "b", "b"], dtype="category")
+ expected_schema_hint = {"type": "categorical", "categories": ["a", "b"]}
+
+ actual_schema_hint = get_schema_type_hint_of_array(array)
+
+ self.assertEqual(expected_schema_hint, actual_schema_hint)
+
+ def test__get_dtype_of_array__castable_dtypes_return_as_expected(self):
+ types = [np.float64, np.int64]
+ expected_dtypes = [np.float32, np.int32]
+
+ for test_type_index in range(len(types)):
+ with self.subTest(f"Testing get_dtype_of_array with castable type {types[test_type_index].__name__}",
+ i=test_type_index):
+ array = Series(data=[], dtype=types[test_type_index])
+ self.assertEqual(get_dtype_of_array(array), expected_dtypes[test_type_index])
+
+ def test__get_schema_type_hint_of_array__castable_dtypes_return_as_expected(self):
+ types = [np.float64, np.int64]
+ expected_schema_hints = [{"type": "float32"}, {"type": "int32"}]
+
+ for test_type_index in range(len(types)):
+ with self.subTest(
+ f"Testing get_schema_type_hint_of_array with castable type {types[test_type_index].__name__}",
+ i=test_type_index):
+ array = Series(data=[], dtype=types[test_type_index])
+ self.assertEqual(get_schema_type_hint_of_array(array), expected_schema_hints[test_type_index])
diff --git a/server/test/test_plugins.py b/server/test/unit/common/utils/test_utils.py
similarity index 95%
rename from server/test/test_plugins.py
rename to server/test/unit/common/utils/test_utils.py
index b5f2a5b5..7d5b6272 100644
--- a/server/test/test_plugins.py
+++ b/server/test/unit/common/utils/test_utils.py
@@ -2,7 +2,7 @@ import os
import shutil
import unittest
-from server.common.utils import import_plugins
+from server.common.utils.utils import import_plugins
from server.test import PROJECT_ROOT, random_string
diff --git a/server/test/unit/compute/__init__.py b/server/test/unit/compute/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_diffexp.py b/server/test/unit/compute/test_diffexp_cxg.py
similarity index 88%
rename from server/test/test_diffexp.py
rename to server/test/unit/compute/test_diffexp_cxg.py
index abe1fc57..7a6aabdd 100644
--- a/server/test/test_diffexp.py
+++ b/server/test/unit/compute/test_diffexp_cxg.py
@@ -1,10 +1,10 @@
import unittest
from server.data_common.matrix_loader import MatrixDataLoader
-from server.test import PROJECT_ROOT, app_config
+from server.test import PROJECT_ROOT, app_config, FIXTURES_ROOT
import server.compute.diffexp_cxg as diffexp_cxg
import server.compute.diffexp_generic as diffexp_generic
-from server.converters.cxgtool import write_cxg
-from server.test.create_test_matrix import create_test_h5ad
+from server.converters.cxgtool import write_cxg, create_cxg_group_metadata
+from server.test.performance.create_test_matrix import create_test_h5ad
from server.data_common.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
import numpy as np
import tempfile
@@ -16,8 +16,7 @@ class DiffExpTest(unittest.TestCase):
adaptor types and different algorithms."""
def load_dataset(self, path, extra_server_config={}, extra_dataset_config={}):
- config = app_config(path, extra_server_config=extra_server_config,
- extra_dataset_config=extra_dataset_config)
+ config = app_config(path, extra_server_config=extra_server_config, extra_dataset_config=extra_dataset_config)
loader = MatrixDataLoader(path)
adaptor = loader.open(config)
return adaptor
@@ -69,7 +68,7 @@ class DiffExpTest(unittest.TestCase):
def test_cxg_default(self):
"""Test a cxg adaptor with its default diffexp algorithm (diffexp_cxg)"""
- adaptor = self.load_dataset(f"{PROJECT_ROOT}/server/test/test_datasets/pbmc3k.cxg")
+ adaptor = self.load_dataset(f"{FIXTURES_ROOT}/pbmc3k.cxg")
maskA = self.get_mask(adaptor, 1, 10)
maskB = self.get_mask(adaptor, 2, 10)
@@ -83,7 +82,7 @@ class DiffExpTest(unittest.TestCase):
def test_cxg_generic(self):
"""Test a cxg adaptor with the generic adaptor"""
- adaptor = self.load_dataset(f"{PROJECT_ROOT}/server/test/test_datasets/pbmc3k.cxg")
+ adaptor = self.load_dataset(f"{FIXTURES_ROOT}/pbmc3k.cxg")
maskA = self.get_mask(adaptor, 1, 10)
maskB = self.get_mask(adaptor, 2, 10)
# run it directly
@@ -105,13 +104,15 @@ class DiffExpTest(unittest.TestCase):
adata = adaptor_anndata.data
sparsename = os.path.join(dirname, "sparse.cxg")
- write_cxg(adata=adata, container=sparsename, title="sparse", sparse_threshold=11)
+ cxg_group_metadata = create_cxg_group_metadata(adata=adata, basefname="sparse.h5ad", title="sparse",)
+ write_cxg(adata=adata, container=sparsename, cxg_group_metadata=cxg_group_metadata, sparse_threshold=11)
adaptor_sparse = self.load_dataset(sparsename)
assert adaptor_sparse.open_array("X").schema.sparse
assert adaptor_sparse.has_array("X_col_shift") == apply_col_shift
densename = os.path.join(dirname, "dense.cxg")
- write_cxg(adata=adata, container=densename, title="dense", sparse_threshold=0)
+ cxg_group_metadata = create_cxg_group_metadata(adata=adata, basefname="dense.h5ad", title="dense",)
+ write_cxg(adata=adata, container=densename, cxg_group_metadata=cxg_group_metadata, sparse_threshold=0)
adaptor_dense = self.load_dataset(densename)
assert not adaptor_dense.open_array("X").schema.sparse
assert not adaptor_dense.has_array("X_col_shift")
diff --git a/server/test/unit/converters/__init__.py b/server/test/unit/converters/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_cxgtool.py b/server/test/unit/converters/test_cxgtool.py
similarity index 75%
rename from server/test/test_cxgtool.py
rename to server/test/unit/converters/test_cxgtool.py
index f7daf5db..46e26b48 100644
--- a/server/test/test_cxgtool.py
+++ b/server/test/unit/converters/test_cxgtool.py
@@ -4,10 +4,10 @@ import unittest
import anndata
from server.common.data_locator import DataLocator
-from server.converters.cxgtool import write_cxg
+from server.converters.cxgtool import write_cxg, create_cxg_group_metadata
from server.data_cxg.cxg_adaptor import CxgAdaptor
from server.test import PROJECT_ROOT, app_config, random_string
-from server.test.test_datasets.fixtures import pbmc3k_colors
+from server.test.fixtures.fixtures import pbmc3k_colors
class TestCxgAdaptor(unittest.TestCase):
@@ -33,6 +33,9 @@ class TestCxgAdaptor(unittest.TestCase):
data_locator = f"/tmp/test_{rand_str}.cxg"
self.fixtures.append(data_locator)
source_h5ad = anndata.read_h5ad(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad")
- write_cxg(adata=source_h5ad, container=data_locator, title="pbmc3k", **kwargs)
+ cxg_group_metadata = create_cxg_group_metadata(
+ adata=source_h5ad, basefname="pbmc3k.h5ad", title="pbmc3k", **kwargs
+ )
+ write_cxg(adata=source_h5ad, container=data_locator, cxg_group_metadata=cxg_group_metadata)
config = app_config(data_locator)
return CxgAdaptor(DataLocator(data_locator), config)
diff --git a/server/test/unit/data_anndata/__init__.py b/server/test/unit/data_anndata/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_anndata_adaptor.py b/server/test/unit/data_anndata/test_anndata_adaptor.py
similarity index 92%
rename from server/test/test_anndata_adaptor.py
rename to server/test/unit/data_anndata/test_anndata_adaptor.py
index 3afe2ea3..d4a3a778 100644
--- a/server/test/test_anndata_adaptor.py
+++ b/server/test/unit/data_anndata/test_anndata_adaptor.py
@@ -1,20 +1,19 @@
import json
-from os import path
-import pytest
+import sys
import time
import unittest
-import sys
-import server.test.decode_fbs as decode_fbs
-from parameterized import parameterized_class
import numpy as np
import pandas as pd
+import pytest
+from parameterized import parameterized_class
+import server.test.unit.decode_fbs as decode_fbs
from server.common.data_locator import DataLocator
from server.common.errors import FilterError
from server.data_anndata.anndata_adaptor import AnndataAdaptor
-from server.test import PROJECT_ROOT, app_config
-from server.test.test_datasets.fixtures import pbmc3k_colors
+from server.test import PROJECT_ROOT, app_config, FIXTURES_ROOT
+from server.test.fixtures.fixtures import pbmc3k_colors
"""
Test the anndata adaptor using the pbmc3k data set.
@@ -25,11 +24,11 @@ Test the anndata adaptor using the pbmc3k data set.
("data_locator", "backed"),
[
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", False),
- (f"{PROJECT_ROOT}/server/test/test_datasets/pbmc3k-CSC-gz.h5ad", False),
- (f"{PROJECT_ROOT}/server/test/test_datasets/pbmc3k-CSR-gz.h5ad", False),
+ (f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", False),
+ (f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", False),
(f"{PROJECT_ROOT}/example-dataset/pbmc3k.h5ad", True),
- (f"{PROJECT_ROOT}/server/test/test_datasets/pbmc3k-CSC-gz.h5ad", True),
- (f"{PROJECT_ROOT}/server/test/test_datasets/pbmc3k-CSR-gz.h5ad", True),
+ (f"{FIXTURES_ROOT}/pbmc3k-CSC-gz.h5ad", True),
+ (f"{FIXTURES_ROOT}/pbmc3k-CSR-gz.h5ad", True),
],
)
class AdaptorTest(unittest.TestCase):
@@ -84,7 +83,7 @@ class AdaptorTest(unittest.TestCase):
self.assertEqual(self.data.get_colors(), pbmc3k_colors)
def test_get_schema(self):
- with open(path.join(path.dirname(__file__), "schema.json")) as fh:
+ with open(f"{FIXTURES_ROOT}/schema.json") as fh:
schema = json.load(fh)
self.assertDictEqual(self.data.get_schema(), schema)
@@ -237,14 +236,14 @@ class AdaptorTest(unittest.TestCase):
self.data.compute_embedding("umap", filter)
return
- (schema, fbs) = self.data.compute_embedding("umap", filter)
+ schema = self.data.compute_embedding("umap", filter)
self.assertIsInstance(schema["name"], str)
name = schema["name"]
self.assertEqual(schema["type"], "float32")
self.assertEqual(schema["dims"], [f"{name}_0", f"{name}_1"])
- emb = decode_fbs.decode_matrix_FBS(fbs)
- self.assertEqual(emb["n_rows"], 100)
- self.assertEqual(emb["n_cols"], 2)
- self.assertEqual(emb["col_idx"], [f"{name}_0", f"{name}_1"])
+ emb = self.data.data.obsm[f"X_{name}"]
+ self.assertEqual(emb.shape, (2638, 2))
+ self.assertTrue(np.isfinite(emb[0:100]).all())
+ self.assertTrue(np.isnan(emb[100:]).all())
diff --git a/server/test/test_anndata_adaptor_data_load.py b/server/test/unit/data_anndata/test_anndata_adaptor_data_load.py
similarity index 92%
rename from server/test/test_anndata_adaptor_data_load.py
rename to server/test/unit/data_anndata/test_anndata_adaptor_data_load.py
index e5c01a03..35ccb886 100644
--- a/server/test/test_anndata_adaptor_data_load.py
+++ b/server/test/unit/data_anndata/test_anndata_adaptor_data_load.py
@@ -43,13 +43,10 @@ class DataLocatorAdaptorTest(unittest.TestCase):
def get_basic_config(self):
config = AppConfig()
config.update_server_config(
- single_dataset__obs_names=None,
- single_dataset__var_names=None,
+ single_dataset__obs_names=None, single_dataset__var_names=None,
)
config.update_default_dataset_config(
- embeddings__names=["umap"],
- presentation__max_categories=100,
- diffexp__lfc_cutoff=0.01,
+ embeddings__names=["umap"], presentation__max_categories=100, diffexp__lfc_cutoff=0.01,
)
return config
diff --git a/server/test/test_nan_anndata_adaptor.py b/server/test/unit/data_anndata/test_nan_anndata_adaptor.py
similarity index 90%
rename from server/test/test_nan_anndata_adaptor.py
rename to server/test/unit/data_anndata/test_nan_anndata_adaptor.py
index 6471dd8b..48c4e020 100644
--- a/server/test/test_nan_anndata_adaptor.py
+++ b/server/test/unit/data_anndata/test_nan_anndata_adaptor.py
@@ -1,19 +1,19 @@
-import pytest
+import math
import unittest
import warnings
-import math
-import server.test.decode_fbs as decode_fbs
+import pytest
-from server.data_anndata.anndata_adaptor import AnndataAdaptor
-from server.common.errors import FilterError
+import server.test.unit.decode_fbs as decode_fbs
from server.common.data_locator import DataLocator
-from server.test import PROJECT_ROOT, app_config
+from server.common.errors import FilterError
+from server.data_anndata.anndata_adaptor import AnndataAdaptor
+from server.test import app_config, FIXTURES_ROOT
class NaNTest(unittest.TestCase):
def setUp(self):
- self.data_locator = DataLocator(f"{PROJECT_ROOT}/server/test/test_datasets/nan.h5ad")
+ self.data_locator = DataLocator(f"{FIXTURES_ROOT}/nan.h5ad")
self.config = app_config(self.data_locator.path)
with warnings.catch_warnings():
@@ -22,8 +22,9 @@ class NaNTest(unittest.TestCase):
self.data._create_schema()
def test_load(self):
- with self.assertWarns(UserWarning):
+ with self.assertLogs(level="WARN") as logger:
self.data = AnndataAdaptor(self.data_locator, self.config)
+ self.assertTrue(logger.output)
def test_init(self):
self.assertEqual(self.data.cell_count, 100)
diff --git a/server/test/unit/data_common/__init__.py b/server/test/unit/data_common/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/unit/data_common/fbs/__init__.py b/server/test/unit/data_common/fbs/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_fbs.py b/server/test/unit/data_common/fbs/test_matrix.py
similarity index 98%
rename from server/test/test_fbs.py
rename to server/test/unit/data_common/fbs/test_matrix.py
index 59ad989a..1e23cb6b 100644
--- a/server/test/test_fbs.py
+++ b/server/test/unit/data_common/fbs/test_matrix.py
@@ -3,7 +3,7 @@ import pandas as pd
import numpy as np
from scipy import sparse
-import server.test.decode_fbs as decode_fbs
+import server.test.unit.decode_fbs as decode_fbs
from server.data_common.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
diff --git a/server/test/test_matrixcache.py b/server/test/unit/data_common/test_matrix_loader.py
similarity index 96%
rename from server/test/test_matrixcache.py
rename to server/test/unit/data_common/test_matrix_loader.py
index 6c68cfc3..d365b001 100644
--- a/server/test/test_matrixcache.py
+++ b/server/test/unit/data_common/test_matrix_loader.py
@@ -1,13 +1,13 @@
+import os
+import shutil
+import tempfile
+import time
import unittest
-from server.data_common.matrix_loader import MatrixDataCacheManager
+
from server.common.app_config import AppConfig
from server.common.errors import DatasetAccessError
-import tempfile
-import shutil
-import os
-import time
-
-from server.test import PROJECT_ROOT
+from server.data_common.matrix_loader import MatrixDataCacheManager
+from server.test import FIXTURES_ROOT
class MatrixCacheTest(unittest.TestCase):
@@ -15,7 +15,7 @@ class MatrixCacheTest(unittest.TestCase):
pass
def make_temporay_datasets(self, dirname, num):
- source = f"{PROJECT_ROOT}/server/test/test_datasets/pbmc3k.cxg"
+ source = f"{FIXTURES_ROOT}/pbmc3k.cxg"
for i in range(num):
target = os.path.join(dirname, str(i) + ".cxg")
shutil.copytree(source, target)
@@ -38,7 +38,7 @@ class MatrixCacheTest(unittest.TestCase):
result = {}
for k, v in datasets.items():
# filter out the dirname and the .cxg from the name
- newk = int(k[1][len(dirname) + 1 : -4])
+ newk = int(k[1][len(dirname) + 1: -4])
result[newk] = v
return result
diff --git a/server/test/unit/data_cxg/__init__.py b/server/test/unit/data_cxg/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_cxg_adaptor.py b/server/test/unit/data_cxg/test_cxg_adaptor.py
similarity index 74%
rename from server/test/test_cxg_adaptor.py
rename to server/test/unit/data_cxg/test_cxg_adaptor.py
index 205ce24b..3544138c 100644
--- a/server/test/test_cxg_adaptor.py
+++ b/server/test/unit/data_cxg/test_cxg_adaptor.py
@@ -2,8 +2,8 @@ import unittest
from server.common.data_locator import DataLocator
from server.data_cxg.cxg_adaptor import CxgAdaptor
-from server.test import PROJECT_ROOT, app_config
-from server.test.test_datasets.fixtures import pbmc3k_colors
+from server.test import FIXTURES_ROOT, app_config
+from server.test.fixtures.fixtures import pbmc3k_colors
class TestCxgAdaptor(unittest.TestCase):
@@ -14,6 +14,6 @@ class TestCxgAdaptor(unittest.TestCase):
self.assertDictEqual(data.get_colors(), dict())
def get_data(self, fixture):
- data_locator = f"{PROJECT_ROOT}/server/test/test_datasets/{fixture}"
+ data_locator = f"{FIXTURES_ROOT}/{fixture}"
config = app_config(data_locator)
return CxgAdaptor(DataLocator(data_locator), config)
diff --git a/server/test/unit/decode_fbs.py b/server/test/unit/decode_fbs.py
new file mode 100644
index 00000000..6fb1fcec
--- /dev/null
+++ b/server/test/unit/decode_fbs.py
@@ -0,0 +1,31 @@
+"""
+Code to decode, for testing purposes, the flatbuffer encoded blobs.
+
+This code will need to be updated if fbs/matrix.fbs changes. For more information, see fbs/matrix.fbs and
+server/data_common/fbs/
+"""
+
+import server.data_common.fbs.NetEncoding.Matrix as Matrix
+from server.data_common.fbs.matrix import deserialize_typed_array
+
+
+def decode_matrix_FBS(buf):
+ """
+ Given a FBS Matrix, return an decoded Python dict containing same info in native format.
+ NOTE / TODO: row_idx not currently implemented
+ """
+ df = Matrix.Matrix.GetRootAsMatrix(buf, 0)
+ n_rows = df.NRows()
+ n_cols = df.NCols()
+
+ columns_length = df.ColumnsLength()
+
+ decoded_columns = []
+ for col_idx in range(0, columns_length):
+ col = df.Columns(col_idx)
+ tarr = (col.UType(), col.U())
+ decoded_columns.append(deserialize_typed_array(tarr))
+
+ cidx = deserialize_typed_array((df.ColIndexType(), df.ColIndex()))
+
+ return {"n_rows": n_rows, "n_cols": n_cols, "columns": decoded_columns, "col_idx": cidx, "row_idx": None}
diff --git a/server/test/unit/eb/__init__.py b/server/test/unit/eb/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/test/test_eb.py b/server/test/unit/eb/test_eb.py
similarity index 90%
rename from server/test/test_eb.py
rename to server/test/unit/eb/test_eb.py
index f1e1ef9d..b43fda24 100644
--- a/server/test/test_eb.py
+++ b/server/test/unit/eb/test_eb.py
@@ -2,7 +2,7 @@ import unittest
import tempfile
import requests
import subprocess
-from server.test import PROJECT_ROOT
+from server.test import PROJECT_ROOT, FIXTURES_ROOT
from server.common.app_config import AppConfig
from contextlib import contextmanager
import time
@@ -37,7 +37,7 @@ class Elastic_Beanstalk_Test(unittest.TestCase):
c = AppConfig()
# test that eb works
c.update_server_config(
- multi_dataset__dataroot=f"{PROJECT_ROOT}/server/test/test_datasets", app__flask_secret_key="open sesame"
+ multi_dataset__dataroot=f"{FIXTURES_ROOT}", app__flask_secret_key="open sesame"
)
c.complete_config()
diff --git a/setup.py b/setup.py
index 1ef0c6cd..56a2bd28 100644
--- a/setup.py
+++ b/setup.py
@@ -11,7 +11,7 @@ with open("server/requirements-prepare.txt") as fh:
setup(
name="cellxgene",
- version="0.15.0",
+ version="0.16.0",
packages=find_packages(),
url="https://github.com/chanzuckerberg/cellxgene",
license="MIT",