diff --git a/.gitignore b/.gitignore
index a2a4695d..3a4cbc2e 100644
--- a/.gitignore
+++ b/.gitignore
@@ -17,12 +17,12 @@ venv/
cellxgene/
# client build
-server/app/web/static/css/
-server/app/web/static/img/
-server/app/web/static/media/
-server/app/web/static/fonts/
-server/app/web/static/js/
-server/app/web/templates/index\.html
+server/common/web/static/css/
+server/common/web/static/img/
+server/common/web/static/media/
+server/common/web/static/fonts/
+server/common/web/static/js/
+server/common/web/templates/index\.html
# Jupyter Notebook
.ipynb_checkpoints
diff --git a/MANIFEST.in b/MANIFEST.in
index 6ed38441..7d059a83 100644
--- a/MANIFEST.in
+++ b/MANIFEST.in
@@ -1,5 +1,5 @@
-recursive-include server/app/web/templates *
-recursive-include server/app/web/static *
+recursive-include server/common/web/templates *
+recursive-include server/common/web/static *
include server/requirements.txt
include server/requirements-prepare.txt
diff --git a/Makefile b/Makefile
index bc02135a..e4af098c 100644
--- a/Makefile
+++ b/Makefile
@@ -34,24 +34,24 @@ build-client:
build-cli: build-client
git ls-files server/ | cpio -pdm $(BUILDDIR)
cp -r client/build/ $(CLIENTBUILD)
- mkdir -p $(SERVERBUILD)/app/web/static/img
- mkdir -p $(SERVERBUILD)/app/web/templates/
- cp $(CLIENTBUILD)/index.html $(SERVERBUILD)/app/web/templates/
- cp -r $(CLIENTBUILD)/static $(SERVERBUILD)/app/web/
- cp $(CLIENTBUILD)/favicon.png $(SERVERBUILD)/app/web/static/img
- cp $(CLIENTBUILD)/service-worker.js $(SERVERBUILD)/app/web/static/js/
+ mkdir -p $(SERVERBUILD)/common/web/static/img
+ mkdir -p $(SERVERBUILD)/common/web/templates/
+ cp $(CLIENTBUILD)/index.html $(SERVERBUILD)/common/web/templates/
+ cp -r $(CLIENTBUILD)/static $(SERVERBUILD)/common/web/
+ cp $(CLIENTBUILD)/favicon.png $(SERVERBUILD)/common/web/static/img
+ cp $(CLIENTBUILD)/service-worker.js $(SERVERBUILD)/common/web/static/js/
cp MANIFEST.in README.md setup.cfg setup.py $(BUILDDIR)
# If you are actively developing in the server folder use this, dirties the source tree
.PHONY: build-for-server-dev
build-for-server-dev: clean-server build-client
- mkdir -p server/app/web/static/img
- mkdir -p server/app/web/static/js
- mkdir -p server/app/web/templates/
- cp client/build/index.html server/app/web/templates/
- cp -r client/build/static server/app/web/
- cp client/build/favicon.png server/app/web/static/img
- cp client/build/service-worker.js server/app/web/static/js/
+ mkdir -p server/common/web/static/img
+ mkdir -p server/common/web/static/js
+ mkdir -p server/common/web/templates/
+ cp client/build/index.html server/common/web/templates/
+ cp -r client/build/static server/common/web/
+ cp client/build/favicon.png server/common/web/static/img
+ cp client/build/service-worker.js server/common/web/static/js/
# TESTING
@@ -137,10 +137,6 @@ dev-env-client:
dev-env-server:
pip install -r server/requirements-dev.txt
-.PHONY: gui-env
-gui-env: dev-env
- pip install -r server/requirements-gui.txt
-
# give PART=[major, minor, part] as param to make bump
.PHONY: bump
bump:
@@ -195,30 +191,3 @@ install-dist: uninstall
uninstall:
pip uninstall -y cellxgene || :
-
-# GUI
-
-.PHONY: build-assets
-build-assets:
- pyside2-rcc server/gui/cellxgene.qrc -o server/gui/cellxgene_rc.py
-
-.PHONY: gui-spec-osx
-gui-spec-osx: clean-lite gui-env
- pip install -e .[gui]
- pyi-makespec -D -w --additional-hooks-dir server/gui/ -n cellxgene --add-binary='/System/Library/Frameworks/Tk.framework/Tk':'tk' --add-binary='/System/Library/Frameworks/Tcl.framework/Tcl':'tcl' --add-data server/app/web/templates/:server/app/web/templates/ --add-data server/app/web/static/:server/app/web/static/ --icon server/gui/images/cxg_icons.icns server/gui/main.py
- mv cellxgene.spec cellxgene-osx.spec
-
-.PHONY: gui-spec-windows
-gui-spec-windows: clean-lite dev-env
- pip install -e .[gui]
- pyi-makespec -D -w --additional-hooks-dir server/gui/ -n cellxgene --add-data server/app/web/templates;server/app/web/templates --add-data server/app/web/static;server/app/web/static --icon server/gui/images/icon.ico server/gui/main.py
- mv cellxgene.spec cellxgene-windows.spec
-
-.PHONY: gui-build-osx
-gui-build-osx: clean-lite
- pyinstaller --clean cellxgene-osx.spec
-
-.PHONY: gui-build-windows
-gui-build-windows: clean-lite
- pyinstaller --clean cellxgene-windows.spec
-
diff --git a/cellxgene-osx.spec b/cellxgene-osx.spec
deleted file mode 100644
index 7bac23c4..00000000
--- a/cellxgene-osx.spec
+++ /dev/null
@@ -1,41 +0,0 @@
-# -*- mode: python ; coding: utf-8 -*-
-
-block_cipher = None
-
-
-a = Analysis(['server/gui/main.py'],
- pathex=['/Users/charlotteweaver/Documents/Git/cellxgene'],
- binaries=[('/System/Library/Frameworks/Tk.framework/Tk', 'tk'), ('/System/Library/Frameworks/Tcl.framework/Tcl', 'tcl')],
- datas=[('server/app/web/templates/', 'server/app/web/templates/'), ('server/app/web/static/', 'server/app/web/static/')],
- hiddenimports=['sklearn', 'sklearn.utils._cython_blas', 'sklearn.neighbors.typedefs', 'sklearn.neighbors.quad_tree', 'sklearn.tree', 'sklearn.tree._utils'],
- hookspath=['server/gui/'],
- runtime_hooks=[],
- excludes=[],
- win_no_prefer_redirects=False,
- win_private_assemblies=False,
- cipher=block_cipher,
- noarchive=False)
-pyz = PYZ(a.pure, a.zipped_data,
- cipher=block_cipher)
-exe = EXE(pyz,
- a.scripts,
- [],
- exclude_binaries=True,
- name='cellxgene',
- debug=False,
- bootloader_ignore_signals=False,
- strip=False,
- upx=True,
- console=False , icon='server/gui/images/cxg_icons.icns')
-coll = COLLECT(exe,
- a.binaries,
- a.zipfiles,
- a.datas,
- strip=False,
- upx=True,
- upx_exclude=[],
- name='cellxgene')
-app = BUNDLE(coll,
- name='cellxgene.app',
- icon='server/gui/images/cxg_icons.icns',
- bundle_identifier=None)
diff --git a/cellxgene-windows.spec b/cellxgene-windows.spec
deleted file mode 100644
index 2bd7054a..00000000
--- a/cellxgene-windows.spec
+++ /dev/null
@@ -1,36 +0,0 @@
-# -*- mode: python -*-
-
-block_cipher = None
-
-
-a = Analysis(['server\\gui\\main.py'],
- pathex=['C:\\Users\\Charlotte\\Documents\\git\\cellxgene'],
- binaries=[],
- datas=[('server/app/web/templates/', 'server/app/web/templates'), ('server/app/web/static/', 'server/app/web/static')],
- hiddenimports=[],
- hookspath=['server/gui/'],
- runtime_hooks=[],
- excludes=[],
- win_no_prefer_redirects=False,
- win_private_assemblies=False,
- cipher=block_cipher,
- noarchive=False)
-pyz = PYZ(a.pure, a.zipped_data,
- cipher=block_cipher)
-exe = EXE(pyz,
- a.scripts,
- [],
- exclude_binaries=True,
- name='cellxgene',
- debug=False,
- bootloader_ignore_signals=False,
- strip=False,
- upx=True,
- console=False , icon='server\\gui\\images\\icon.ico')
-coll = COLLECT(exe,
- a.binaries,
- a.zipfiles,
- a.datas,
- strip=False,
- upx=True,
- name='cellxgene')
diff --git a/dev_docs/developer_scripts.md b/dev_docs/developer_scripts.md
index dd938233..fbb2ddd0 100644
--- a/dev_docs/developer_scripts.md
+++ b/dev_docs/developer_scripts.md
@@ -43,7 +43,6 @@ Installs requirements files
```
dev-env - installs requirements and requirments-dev (for building code)
-gui-env - installs requirements and requirments-dev and requirments-gui (for building native app)
```
### install commands
@@ -59,17 +58,6 @@ install-dist - installs from local dist folder
uninstall - uninstalls cellxgene
```
-### gui
-
-Commands for building the native app
-
-```
-build-assets - builds the image and icon assets for the gui to pull from
-gui-spec-osx - creates the initial spec file for osx, do not run unless you are starting from scratch, one time only
-gui-spec-windows - creates the initial spec file for windows, do not run unless you are starting from scratch, one time only
-gui-build-osx - builds the app from the osx spec file
-gui-build-windows - builds the app from the windows spec file
-```
## Client Makefile
The following phony `make` targets in `client/Makefile` are convenience methods for getting you up and developing.
diff --git a/example-dataset/pbmc3k.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__attr.tdb b/example-dataset/pbmc3k.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__attr.tdb
new file mode 100644
index 00000000..68f06ed1
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__attr.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__fragment_metadata.tdb b/example-dataset/pbmc3k.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__fragment_metadata.tdb
new file mode 100644
index 00000000..511a453a
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/X/__1576858534264_1576858534264_4f12045b32ea45a490bdad087bac4dc3/__fragment_metadata.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/X/__array_schema.tdb b/example-dataset/pbmc3k.cxg/X/__array_schema.tdb
new file mode 100644
index 00000000..b5c8fd3a
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/X/__array_schema.tdb differ
diff --git a/server/app/driver/__init__.py b/example-dataset/pbmc3k.cxg/X/__lock.tdb
similarity index 100%
rename from server/app/driver/__init__.py
rename to example-dataset/pbmc3k.cxg/X/__lock.tdb
diff --git a/server/app/rest_api/__init__.py b/example-dataset/pbmc3k.cxg/__tiledb_group.tdb
similarity index 100%
rename from server/app/rest_api/__init__.py
rename to example-dataset/pbmc3k.cxg/__tiledb_group.tdb
diff --git a/server/app/scanpy_engine/__init__.py b/example-dataset/pbmc3k.cxg/emb/__tiledb_group.tdb
similarity index 100%
rename from server/app/scanpy_engine/__init__.py
rename to example-dataset/pbmc3k.cxg/emb/__tiledb_group.tdb
diff --git a/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__attr.tdb b/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__attr.tdb
new file mode 100644
index 00000000..28d844a9
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__attr.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__fragment_metadata.tdb b/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__fragment_metadata.tdb
new file mode 100644
index 00000000..6faeac81
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__1576858534229_1576858534229_391cdd6b87b649dea76842dfa59ed0d9/__fragment_metadata.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__array_schema.tdb b/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__array_schema.tdb
new file mode 100644
index 00000000..b914fda9
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__array_schema.tdb differ
diff --git a/server/app/util/__init__.py b/example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__lock.tdb
similarity index 100%
rename from server/app/util/__init__.py
rename to example-dataset/pbmc3k.cxg/emb/draw_graph_fr/__lock.tdb
diff --git a/example-dataset/pbmc3k.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__attr.tdb b/example-dataset/pbmc3k.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__attr.tdb
new file mode 100644
index 00000000..b7f3d4d5
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__attr.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__fragment_metadata.tdb b/example-dataset/pbmc3k.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__fragment_metadata.tdb
new file mode 100644
index 00000000..167e76b2
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/pca/__1576858534121_1576858534121_454903804a694b3b8ccdae56065664ba/__fragment_metadata.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/emb/pca/__array_schema.tdb b/example-dataset/pbmc3k.cxg/emb/pca/__array_schema.tdb
new file mode 100644
index 00000000..2ad7d0e0
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/pca/__array_schema.tdb differ
diff --git a/server/app/util/fbs/NetEncoding/__init__.py b/example-dataset/pbmc3k.cxg/emb/pca/__lock.tdb
similarity index 100%
rename from server/app/util/fbs/NetEncoding/__init__.py
rename to example-dataset/pbmc3k.cxg/emb/pca/__lock.tdb
diff --git a/example-dataset/pbmc3k.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__attr.tdb b/example-dataset/pbmc3k.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__attr.tdb
new file mode 100644
index 00000000..5f3adbea
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__attr.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__fragment_metadata.tdb b/example-dataset/pbmc3k.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__fragment_metadata.tdb
new file mode 100644
index 00000000..5b449a93
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/tsne/__1576858534161_1576858534161_aa4803e7e7be4b23bea35f5d62296f14/__fragment_metadata.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/emb/tsne/__array_schema.tdb b/example-dataset/pbmc3k.cxg/emb/tsne/__array_schema.tdb
new file mode 100644
index 00000000..b914fda9
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/tsne/__array_schema.tdb differ
diff --git a/server/app/util/fbs/__init__.py b/example-dataset/pbmc3k.cxg/emb/tsne/__lock.tdb
similarity index 100%
rename from server/app/util/fbs/__init__.py
rename to example-dataset/pbmc3k.cxg/emb/tsne/__lock.tdb
diff --git a/example-dataset/pbmc3k.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__attr.tdb b/example-dataset/pbmc3k.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__attr.tdb
new file mode 100644
index 00000000..b80fb865
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__attr.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__fragment_metadata.tdb b/example-dataset/pbmc3k.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__fragment_metadata.tdb
new file mode 100644
index 00000000..167e8680
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/umap/__1576858534193_1576858534193_67d97bcdd3d1486985f5974b133cb496/__fragment_metadata.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/emb/umap/__array_schema.tdb b/example-dataset/pbmc3k.cxg/emb/umap/__array_schema.tdb
new file mode 100644
index 00000000..b914fda9
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/emb/umap/__array_schema.tdb differ
diff --git a/server/app/web/__init__.py b/example-dataset/pbmc3k.cxg/emb/umap/__lock.tdb
similarity index 100%
rename from server/app/web/__init__.py
rename to example-dataset/pbmc3k.cxg/emb/umap/__lock.tdb
diff --git a/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/__fragment_metadata.tdb b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/__fragment_metadata.tdb
new file mode 100644
index 00000000..cddac6b7
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/__fragment_metadata.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain.tdb b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain.tdb
new file mode 100644
index 00000000..2b3b91d9
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain_var.tdb b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain_var.tdb
new file mode 100644
index 00000000..56196286
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/louvain_var.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_counts.tdb b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_counts.tdb
new file mode 100644
index 00000000..8141af5d
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_counts.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_genes.tdb b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_genes.tdb
new file mode 100644
index 00000000..51219d8e
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/n_genes.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0.tdb b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0.tdb
new file mode 100644
index 00000000..e431a413
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0_var.tdb b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0_var.tdb
new file mode 100644
index 00000000..be821f91
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/name_0_var.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/percent_mito.tdb b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/percent_mito.tdb
new file mode 100644
index 00000000..22773bef
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__1576858534031_1576858534031_1641d0129fe64c78b2d0a6a684ce47ba/percent_mito.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/obs/__array_schema.tdb b/example-dataset/pbmc3k.cxg/obs/__array_schema.tdb
new file mode 100644
index 00000000..54eae0c3
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__array_schema.tdb differ
diff --git a/server/gui/__init__.py b/example-dataset/pbmc3k.cxg/obs/__lock.tdb
similarity index 100%
rename from server/gui/__init__.py
rename to example-dataset/pbmc3k.cxg/obs/__lock.tdb
diff --git a/example-dataset/pbmc3k.cxg/obs/__meta/__1576858534024_1576858534024_e856aef5b0e244b1a4b7dde70cd864d0 b/example-dataset/pbmc3k.cxg/obs/__meta/__1576858534024_1576858534024_e856aef5b0e244b1a4b7dde70cd864d0
new file mode 100644
index 00000000..88398dfe
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/obs/__meta/__1576858534024_1576858534024_e856aef5b0e244b1a4b7dde70cd864d0 differ
diff --git a/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/__fragment_metadata.tdb b/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/__fragment_metadata.tdb
new file mode 100644
index 00000000..05ed94e2
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/__fragment_metadata.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/n_cells.tdb b/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/n_cells.tdb
new file mode 100644
index 00000000..733701eb
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/n_cells.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0.tdb b/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0.tdb
new file mode 100644
index 00000000..7ce34978
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0_var.tdb b/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0_var.tdb
new file mode 100644
index 00000000..7fa293f2
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/var/__1576858533970_1576858533970_d241b2e750eb425a9ee23ed1de686c2a/name_0_var.tdb differ
diff --git a/example-dataset/pbmc3k.cxg/var/__array_schema.tdb b/example-dataset/pbmc3k.cxg/var/__array_schema.tdb
new file mode 100644
index 00000000..48812401
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/var/__array_schema.tdb differ
diff --git a/server/utils/__init__.py b/example-dataset/pbmc3k.cxg/var/__lock.tdb
similarity index 100%
rename from server/utils/__init__.py
rename to example-dataset/pbmc3k.cxg/var/__lock.tdb
diff --git a/example-dataset/pbmc3k.cxg/var/__meta/__1576858533966_1576858533966_1362646d804b4982b35052a367633436 b/example-dataset/pbmc3k.cxg/var/__meta/__1576858533966_1576858533966_1362646d804b4982b35052a367633436
new file mode 100644
index 00000000..09f3a459
Binary files /dev/null and b/example-dataset/pbmc3k.cxg/var/__meta/__1576858533966_1576858533966_1362646d804b4982b35052a367633436 differ
diff --git a/pyproject.toml b/pyproject.toml
index a30d4e82..fcfaa647 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -19,7 +19,6 @@ exclude = '''
| dist
| server/app/util/fbs/NetEncoding
)/
- | server/gui/cellxgene_rc.py
)
'''
diff --git a/server/Makefile b/server/Makefile
index 6f727878..aa8c9860 100644
--- a/server/Makefile
+++ b/server/Makefile
@@ -2,8 +2,8 @@ include ../common.mk
.PHONY: clean
clean:
- rm -f app/web/templates/index.html
- rm -rf app/web/static
+ rm -f common/web/templates/index.html
+ rm -rf common/web/static
.PHONY: unit-test
unit-test:
diff --git a/server/app/app.py b/server/app/app.py
index 466cc822..6b42a5de 100644
--- a/server/app/app.py
+++ b/server/app/app.py
@@ -1,26 +1,196 @@
import os
import datetime
-from flask import Flask
+from flask import Flask, redirect, current_app, make_response, render_template
+from flask import Blueprint, request, send_from_directory
from flask_caching import Cache
from flask_compress import Compress
from flask_cors import CORS
+from flask_restful import Api, Resource
-from server.app.rest_api.rest import get_api_resources
-from server.app.util.utils import Float32JSONEncoder
-from server.app.web import webapp
+from http import HTTPStatus
+
+import server.common.rest as common_rest
+from server.common.errors import DatasetAccessError
+from server.common.utils import path_join, Float32JSONEncoder
+from server.common.data_locator import DataLocator
+from server.data_common.matrix_loader import MatrixDataLoader, MatrixDataType
+
+from functools import wraps
+
+webbp = Blueprint("webapp", "server.common.web", template_folder="templates")
+
+
+@webbp.route("/")
+def dataset_index(dataset=None):
+ config = current_app.app_config
+ if dataset is None:
+ if config.datapath:
+ location = config.datapath
+ else:
+ return dataroot_index()
+ else:
+ location = path_join(config.dataroot, dataset)
+
+ scripts = config.scripts
+
+ try:
+ cache_manager = current_app.matrix_data_cache_manager
+ with cache_manager.data_adaptor(location, config) as data_adaptor:
+ dataset_title = config.get_title(data_adaptor)
+ return render_template("index.html", datasetTitle=dataset_title, SCRIPTS=scripts)
+ except DatasetAccessError as e:
+ return make_response(f"Invalid dataset {dataset}: {str(e)}", HTTPStatus.BAD_REQUEST)
+
+
+@webbp.route("/favicon.png")
+def favicon():
+ return send_from_directory(os.path.join(webbp.root_path, "static/img/"), "favicon.png")
+
+
+def get_data_adaptor(dataset=None):
+ config = current_app.app_config
+
+ if dataset is None:
+ datapath = config.datapath
+ else:
+ datapath = path_join(config.dataroot, dataset)
+ # path_join returns a normalized path. Therefore it is
+ # sufficient to check that the datapath starts with the
+ # dataroot to determine that the datapath is under the dataroot.
+ if not datapath.startswith(config.dataroot):
+ raise DatasetAccessError("Invalid dataset {dataset}")
+
+ if datapath is None:
+ return make_response("Dataset must be supplied", HTTPStatus.BAD_REQUEST)
+
+ cache_manager = current_app.matrix_data_cache_manager
+ return cache_manager.data_adaptor(datapath, config)
+
+
+def rest_get_data_adaptor(func):
+ @wraps(func)
+ def wrapped_function(self, dataset=None):
+ try:
+ with get_data_adaptor(dataset) as data_adaptor:
+ return func(self, data_adaptor)
+ except DatasetAccessError as e:
+ return make_response(f"Invalid dataset {dataset}: {str(e)}", HTTPStatus.BAD_REQUEST)
+
+ return wrapped_function
+
+
+def static_redirect(dataset, therest):
+ """ redirect all static requests to the standard location """
+ return redirect(f'/static/{therest}', code=301)
+
+
+def favicon_redirect(dataset):
+ """ redirect favicon to static dir """
+ return redirect('/static/favicon.png', code=301)
+
+
+def dataroot_index():
+ # FIXME with a splash screen that includes a listing of all the datasets.
+ # or perhaps a login screen if this is a hosted environment
+ data = "
Welcome to cellxgene "
+
+ # the following is just for demo purposes...
+ try:
+ config = current_app.app_config
+ locator = DataLocator(config.dataroot)
+ datasets = []
+ for fname in locator.ls():
+ location = path_join(config.dataroot, fname)
+ matrix_data_loader = MatrixDataLoader(location)
+ if matrix_data_loader.etype != MatrixDataType.UNKNOWN:
+ datasets.append(fname)
+
+ data += " Select one of these datasets... "
+ data += ""
+ datasets.sort()
+ for dataset in datasets:
+ data += f"{dataset} "
+ data += " "
+ except Exception:
+ pass
+
+ return make_response(data)
+
+
+class SchemaAPI(Resource):
+ @rest_get_data_adaptor
+ def get(self, data_adaptor):
+ return common_rest.schema_get(data_adaptor, current_app.annotations)
+
+
+class ConfigAPI(Resource):
+ @rest_get_data_adaptor
+ def get(self, data_adaptor):
+ return common_rest.config_get(
+ current_app.app_config, data_adaptor, current_app.annotations)
+
+
+class AnnotationsObsAPI(Resource):
+ @rest_get_data_adaptor
+ def get(self, data_adaptor):
+ return common_rest.annotations_obs_get(
+ request, data_adaptor, current_app.annotations)
+
+ @rest_get_data_adaptor
+ def put(self, data_adaptor):
+ return common_rest.annotations_obs_put(
+ request, data_adaptor, current_app.annotations)
+
+
+class AnnotationsVarAPI(Resource):
+ @rest_get_data_adaptor
+ def get(self, data_adaptor):
+ return common_rest.annotations_var_get(request, data_adaptor, current_app.annotations)
+
+
+class DataVarAPI(Resource):
+ @rest_get_data_adaptor
+ def put(self, data_adaptor):
+ return common_rest.data_var_put(request, data_adaptor)
+
+
+class DiffExpObsAPI(Resource):
+ @rest_get_data_adaptor
+ def post(self, data_adaptor):
+ return common_rest.diffexp_obs_post(request, data_adaptor)
+
+
+class LayoutObsAPI(Resource):
+ @rest_get_data_adaptor
+ def get(self, data_adaptor):
+ return common_rest.layout_obs_get(request, data_adaptor)
+
+
+def get_api_resources(bp_api):
+ api = Api(bp_api)
+ # Initialization routes
+ api.add_resource(SchemaAPI, "/schema")
+ api.add_resource(ConfigAPI, "/config")
+ # Data routes
+ api.add_resource(AnnotationsObsAPI, "/annotations/obs")
+ api.add_resource(AnnotationsVarAPI, "/annotations/var")
+ api.add_resource(DataVarAPI, "/data/var")
+ # Computation routes
+ api.add_resource(DiffExpObsAPI, "/diffexp/obs")
+ api.add_resource(LayoutObsAPI, "/layout/obs")
+ return api
class Server:
- def __init__(self):
- self.data = None
- self.cache = Cache(config={"CACHE_TYPE": "simple", "CACHE_DEFAULT_TIMEOUT": 860_000})
- self.app = None
+ def __init__(self, matrix_data_cache_manager, annotations, app_config):
- def create_app(self):
- self.app = Flask(__name__, static_folder="web/static")
+ self.app = Flask(__name__, static_folder="../common/web/static")
self.app.json_encoder = Float32JSONEncoder
+
+ self.cache = Cache(config={"CACHE_TYPE": "simple", "CACHE_DEFAULT_TIMEOUT": 860_000})
self.cache.init_app(self.app)
+
Compress(self.app)
CORS(self.app, supports_credentials=True)
@@ -30,13 +200,26 @@ class Server:
# Config
SECRET_KEY = os.environ.get("CXG_SECRET_KEY", default="SparkleAndShine")
self.app.config.update(SECRET_KEY=SECRET_KEY)
- self.app.config.update(SCRIPTS=[])
- resources = get_api_resources()
- self.app.register_blueprint(webapp.bp)
- self.app.register_blueprint(resources.blueprint)
- self.app.add_url_rule("/", endpoint="index")
+ self.app.register_blueprint(webbp)
- def attach_data(self, data, title="Demo", about=""):
- self.app.config.update(DATASET_TITLE=title, ABOUT_DATASET=about)
- self.app.data = data
+ api_version = "/api/v0.2"
+ if app_config.datapath:
+ bp_api = Blueprint("api", __name__, url_prefix=api_version)
+ resources = get_api_resources(bp_api)
+ self.app.register_blueprint(resources.blueprint)
+
+ else:
+ # NOTE: These routes only allow the dataset to be in the directory
+ # of the dataroot, and not a subdirectory. We may want to change
+ # the route format at some point
+ bp_api = Blueprint("api_dataset", __name__, url_prefix="/" + api_version)
+ resources = get_api_resources(bp_api)
+ self.app.register_blueprint(resources.blueprint)
+ self.app.add_url_rule("//", 'dataset_index', dataset_index)
+ self.app.add_url_rule("//static/", "static_redirect", static_redirect)
+ self.app.add_url_rule("//favicon.png", "favicon_redirect", favicon_redirect)
+
+ self.app.matrix_data_cache_manager = matrix_data_cache_manager
+ self.app.annotations = annotations
+ self.app.app_config = app_config
diff --git a/server/app/driver/driver.py b/server/app/driver/driver.py
deleted file mode 100644
index 38bdf484..00000000
--- a/server/app/driver/driver.py
+++ /dev/null
@@ -1,113 +0,0 @@
-from abc import ABCMeta, abstractmethod
-
-"""
-Sort order for methods
-1. Initialize
-2. Helper
-3. Filter
-4. Data & Metadata
-5. Computation
-"""
-
-
-class CXGDriver(metaclass=ABCMeta):
- def __init__(self, data_locator=None, args={}):
- self.config = self._get_default_config()
- self.config.update(args)
- if data_locator:
- self._load_data(data_locator)
- self.data_locator = data_locator
- else:
- self.data = None
-
- def update(self, data_locator=None, args={}):
- self.config.update(args)
- if data_locator:
- self._load_data(data_locator)
- self.data_locator = data_locator
-
- @staticmethod
- def _get_default_config():
- return {
- "layout": None,
- "max_category_items": None,
- "diffexp_lfc_cutoff": None,
- "disable_diffexp": False,
- "diffexp_may_be_slow": False,
- }
-
- @abstractmethod
- def get_config_parameters(self, uid=None):
- """
- return a dict of properties that will be used to set the engine-specific
- "parameters" info for client-side configuration.
-
- See rest.py /config route for use
- """
- pass
-
- @property
- def features(self):
- features = {
- "cluster": {"available": False},
- "layout": {"obs": {"available": False}, "var": {"available": False}},
- "diffexp": {"available": True, "interactiveLimit": 50000},
- }
- # TODO - Interactive limit should be generated from the actual available methods see GH issue #94
- if self.config["layout"]:
- # TODO handle "var" when gene layout becomes available
- features["layout"]["obs"] = {"available": True, "interactiveLimit": 50000}
- return features
-
- @abstractmethod
- def get_schema(self):
- """
- Return current schema
- """
- pass
-
- @abstractmethod
- def _load_data(self, data_locator):
- pass
-
- @abstractmethod
- def annotation_to_fbs_matrix(self, axis, field=None, uid=None):
- """
- Gets annotation value for each observation
- :param axis: string obs or var
- :param fields: list of keys for annotation to return, returns all annotation values if not set.
- :return: flatbuffer: in fbs/matrix.fbs encoding
- """
- pass
-
- @abstractmethod
- def annotation_put_fbs(self, axis, fbs, uid=None):
- """
- Put/save FBS as user-defined labels
- """
- pass
-
- @abstractmethod
- def data_frame_to_fbs_matrix(self, filter, axis):
- pass
-
- @abstractmethod
- def diffexp_topN(self, obsFilter1, obsFilter2, top_n=None, interactive_limit=None):
- """
- Computes the top N differentially expressed variables between two observation sets. If mode
- is "TOP_N", then stats for the top N
- dataframes
- contain a subset of variables, then statistics for all variables will be returned, otherwise
- only the top N vars will be returned.
- :param obsFilter1: filter: dictionary with filter params for first set of observations
- :param obsFilter2: filter: dictionary with filter params for second set of observations
- :param top_n: Limit results to top N (Top var mode only)
- :param interactive_limit: -- don't compute if total # genes in dataframes are larger than this
- :return: top N genes and corresponding stats
- """
- pass
-
- @abstractmethod
- def layout_to_fbs_matrix(self, filter):
- """ same as layout, except returns a flatbuffer """
- pass
diff --git a/server/app/rest_api/rest.py b/server/app/rest_api/rest.py
deleted file mode 100644
index 55f0d32d..00000000
--- a/server/app/rest_api/rest.py
+++ /dev/null
@@ -1,240 +0,0 @@
-from http import HTTPStatus
-import warnings
-from uuid import uuid4
-import re
-
-from flask import Blueprint, current_app, jsonify, make_response, request, session
-from flask_restful import Api, Resource
-from server import __version__ as cellxgene_version
-from anndata import __version__ as anndata_version
-
-from server.app.util.constants import Axis, DiffExpMode, JSON_NaN_to_num_warning_msg, CXGUID, CXG_ANNO_COLLECTION
-from server.app.util.errors import (
- FilterError,
- InteractiveError,
- JSONEncodingValueError,
- PrepareError,
- DisabledFeatureError,
-)
-
-
-class SchemaAPI(Resource):
- def get(self):
- cxguid = get_userid(session)
- anno_collection = get_anno_collection(session)
- return make_response(
- jsonify({"schema": current_app.data.get_schema(uid=cxguid, collection=anno_collection)}), HTTPStatus.OK
- )
-
-
-class ConfigAPI(Resource):
- def get(self):
- cxguid = get_userid(session)
- anno_collection = get_anno_collection(session)
- config = {
- "config": {
- "features": [
- {"method": "POST", "path": "/cluster/", **current_app.data.features["cluster"]},
- {"method": "POST", "path": "/layout/obs", **current_app.data.features["layout"]["obs"]},
- {"method": "POST", "path": "/layout/var", **current_app.data.features["layout"]["var"]},
- {"method": "POST", "path": "/diffexp/", **current_app.data.features["diffexp"]},
- ],
- "displayNames": {
- "engine": f"cellxgene Scanpy engine version ",
- "dataset": current_app.config["DATASET_TITLE"],
- },
- "links": {"about-dataset": current_app.config["ABOUT_DATASET"]},
- "parameters": {**current_app.data.get_config_parameters(uid=cxguid, collection=anno_collection)},
- "library_versions": {"cellxgene": cellxgene_version, "anndata": str(anndata_version)},
- }
- }
-
- return make_response(jsonify(config), HTTPStatus.OK)
-
-
-class AnnotationsObsAPI(Resource):
- def get(self):
- fields = request.args.getlist("annotation-name", None)
- preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
- cxguid = get_userid(session)
- anno_collection = get_anno_collection(session)
- try:
- if preferred_mimetype == "application/octet-stream":
- fbs = current_app.data.annotation_to_fbs_matrix("obs", fields, uid=cxguid, collection=anno_collection)
- return make_response(fbs, HTTPStatus.OK, {"Content-Type": "application/octet-stream"})
- else:
- return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
- except KeyError:
- return make_response(f"Error bad key in {fields}", HTTPStatus.BAD_REQUEST)
- except ValueError as e:
- return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
-
- def put(self):
- cxguid = get_userid(session)
- anno_collection = request.args.get("annotation-collection-name", default=None)
- if anno_collection is not None:
- if not is_safe_collection_name(anno_collection):
- return make_response(f"Error, bad annotation collection name", HTTPStatus.BAD_REQUEST)
- set_anno_collection(session, anno_collection)
- else:
- anno_collection = get_anno_collection(session)
-
- try:
- fbs = request.get_data()
- res = current_app.data.annotation_put_fbs("obs", fbs, uid=cxguid, collection=anno_collection)
- return make_response(res, HTTPStatus.OK, {"Content-Type": "application/json"})
- except (ValueError, DisabledFeatureError, KeyError) as e:
- return make_response(str(e), HTTPStatus.BAD_REQUEST)
- except Exception as e:
- return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
-
-
-class AnnotationsVarAPI(Resource):
- def get(self):
- fields = request.args.getlist("annotation-name", None)
- preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
- try:
- if preferred_mimetype == "application/octet-stream":
- return make_response(
- current_app.data.annotation_to_fbs_matrix("var", fields),
- HTTPStatus.OK,
- {"Content-Type": "application/octet-stream"},
- )
- else:
- return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
- except KeyError:
- return make_response(f"Error bad key in {fields}", HTTPStatus.BAD_REQUEST)
- except ValueError as e:
- return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
-
-
-class DataVarAPI(Resource):
- def put(self):
- preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
- try:
- if preferred_mimetype == "application/octet-stream":
- filter_json = request.get_json()
- filter = filter_json["filter"] if filter_json else None
- return make_response(
- current_app.data.data_frame_to_fbs_matrix(filter, axis=Axis.VAR),
- HTTPStatus.OK,
- {"Content-Type": "application/octet-stream"},
- )
- else:
- return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
- except FilterError as e:
- return make_response(e.message, HTTPStatus.BAD_REQUEST)
- except ValueError as e:
- return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
-
-
-class DiffExpObsAPI(Resource):
- def post(self):
- args = request.get_json()
- # confirm mode is present and legal
- try:
- mode = DiffExpMode(args["mode"])
- except KeyError:
- return make_response("Error: mode is required", HTTPStatus.BAD_REQUEST)
- except ValueError:
- return make_response(f"Error: invalid mode option {args['mode']}", HTTPStatus.BAD_REQUEST)
- # Validate filters
- if mode == DiffExpMode.VAR_FILTER or "varFilter" in args:
- # not NOT_IMPLEMENTED
- return make_response("mode=varfilter not implemented", HTTPStatus.NOT_IMPLEMENTED)
- if mode == DiffExpMode.TOP_N and "count" not in args:
- return make_response("mode=topN requires a count parameter", HTTPStatus.BAD_REQUEST)
-
- if "set1" not in args:
- return make_response("set1 is required.", HTTPStatus.BAD_REQUEST)
- if Axis.VAR in args["set1"]["filter"]:
- return make_response("Var filter not allowed for set1", HTTPStatus.BAD_REQUEST)
- # set2
- if "set2" not in args:
- return make_response("Set2 as inverse of set1 is not implemented", HTTPStatus.NOT_IMPLEMENTED)
- if Axis.VAR in args["set2"]["filter"]:
- return make_response("Var filter not allowed for set2", HTTPStatus.BAD_REQUEST)
-
- set1_filter = args["set1"]["filter"]
- set2_filter = args.get("set2", {"filter": {}})["filter"]
-
- # TODO: implement varfilter mode
-
- # mode=topN
- count = args.get("count", None)
- try:
- diffexp = current_app.data.diffexp_topN(
- set1_filter, set2_filter, count, current_app.data.features["diffexp"]["interactiveLimit"],
- )
- return make_response(diffexp, HTTPStatus.OK, {"Content-Type": "application/json"})
- except (ValueError, FilterError) as e:
- return make_response(e.message, HTTPStatus.BAD_REQUEST)
- except InteractiveError:
- return make_response("Non-interactive request", HTTPStatus.FORBIDDEN)
- except JSONEncodingValueError as e:
- # JSON encoding failure, usually due to bad data
- warnings.warn(JSON_NaN_to_num_warning_msg)
- return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
- except ValueError as e:
- return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
-
-
-class LayoutObsAPI(Resource):
- def get(self):
- preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
- try:
- if preferred_mimetype == "application/octet-stream":
- return make_response(
- current_app.data.layout_to_fbs_matrix(), HTTPStatus.OK, {"Content-Type": "application/octet-stream"}
- )
- else:
- return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
- except PrepareError as e:
- return make_response(e.message, HTTPStatus.INTERNAL_SERVER_ERROR)
- except ValueError as e:
- return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
-
-
-def get_userid(ss):
- if CXGUID not in ss:
- ss[CXGUID] = uuid4().hex
- ss.permanent = True
- return ss[CXGUID]
-
-
-def get_anno_collection(ss):
- collection = ss[CXG_ANNO_COLLECTION] if CXG_ANNO_COLLECTION in ss else None
- return collection
-
-
-def set_anno_collection(ss, name):
- ss[CXG_ANNO_COLLECTION] = name
- ss.permanent = True
-
-
-def is_safe_collection_name(name):
- """
- return true if this is a safe collection name
-
- this is ultra convervative. If we want to allow full legal file name syntax,
- we could look at modules like `pathvalidate`
- """
- if name is None:
- return False
- return re.match(r"^[\w\-]+$", name) is not None
-
-
-def get_api_resources():
- bp = Blueprint("api", __name__, url_prefix="/api/v0.2")
- api = Api(bp)
- # Initialization routes
- api.add_resource(SchemaAPI, "/schema")
- api.add_resource(ConfigAPI, "/config")
- # Data routes
- api.add_resource(AnnotationsObsAPI, "/annotations/obs")
- api.add_resource(AnnotationsVarAPI, "/annotations/var")
- api.add_resource(DataVarAPI, "/data/var")
- # Computation routes
- api.add_resource(DiffExpObsAPI, "/diffexp/obs")
- api.add_resource(LayoutObsAPI, "/layout/obs")
- return api
diff --git a/server/app/scanpy_engine/labels.py b/server/app/scanpy_engine/labels.py
deleted file mode 100644
index 55f94283..00000000
--- a/server/app/scanpy_engine/labels.py
+++ /dev/null
@@ -1,60 +0,0 @@
-"""
-Helpers for user annotations
-"""
-import os
-import os.path
-from datetime import datetime
-import pandas as pd
-
-
-def read_labels(fname):
- if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
- return pd.read_csv(fname, dtype="category", index_col=0, header=0, comment="#", keep_default_na=False)
- else:
- return pd.DataFrame()
-
-
-def write_labels(fname, df, header=None, backup_dir=None):
- if backup_dir is not None:
- backup(fname, backup_dir)
- if not df.empty:
- with open(fname, "w", newline="") as f:
- if header is not None:
- f.write(header)
- df.to_csv(f)
- else:
- open(fname, "w").close()
-
-
-def backup(fname, backup_dir, max_backups=9):
- """
- save N backups of file to backup_dir.
- 1. fname -> backup_dir/fname-TIME
- 2. delete excess files in backup_dir
- """
-
- # Make sure there is work to do
- if not os.path.exists(fname):
- return
-
- # Ensure backup_dir exists
- if not os.path.exists(backup_dir):
- os.mkdir(backup_dir)
-
- # Save current file to backup_dir
- fname_base = os.path.basename(fname)
- fname_base_root, fname_base_ext = os.path.splitext(fname_base)
- # don't use ISO standard time format, as it contains characters illegal on some filesytems.
- nowish = datetime.now().strftime("%Y-%m-%dT%H-%M-%S")
- backup_fname = os.path.join(backup_dir, f"{fname_base_root}-{nowish}{fname_base_ext}")
- if os.path.exists(backup_fname):
- os.remove(backup_fname)
- os.rename(fname, backup_fname)
-
- # prune the backup_dir to max number of backup files, keeping the most recent backups
- backups = list(filter(lambda s: s.startswith(fname_base_root), os.listdir(backup_dir)))
- excess_count = len(backups) - max_backups
- if excess_count > 0:
- backups.sort()
- for bu in backups[0:excess_count]:
- os.remove(os.path.join(backup_dir, bu))
diff --git a/server/app/scanpy_engine/scanpy_engine.py b/server/app/scanpy_engine/scanpy_engine.py
deleted file mode 100644
index 63f03830..00000000
--- a/server/app/scanpy_engine/scanpy_engine.py
+++ /dev/null
@@ -1,658 +0,0 @@
-import warnings
-import copy
-import threading
-from datetime import datetime
-import os.path
-from hashlib import blake2b
-import base64
-from packaging import version
-
-import numpy as np
-import pandas
-from pandas.core.dtypes.dtypes import CategoricalDtype
-import anndata
-from scipy import sparse
-
-from server import __version__ as cellxgene_version
-from server.app.driver.driver import CXGDriver
-from server.app.util.constants import Axis, DEFAULT_TOP_N, MAX_LAYOUTS
-from server.app.util.errors import (
- FilterError,
- JSONEncodingValueError,
- PrepareError,
- ScanpyFileError,
- DisabledFeatureError,
-)
-from server.app.util.utils import jsonify_scanpy, requires_data
-from server.app.scanpy_engine.diffexp import diffexp_ttest
-from server.app.util.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
-from server.app.scanpy_engine.labels import read_labels, write_labels
-
-
-anndata_version = version.parse(str(anndata.__version__)).release
-
-
-def anndata_version_is_pre_070():
- major = anndata_version[0]
- minor = anndata_version[1] if len(anndata_version) > 1 else 0
- return major == 0 and minor < 7
-
-
-def has_method(o, name):
- """ return True if `o` has callable method `name` """
- op = getattr(o, name, None)
- return op is not None and callable(op)
-
-
-class ScanpyEngine(CXGDriver):
- def __init__(self, data_locator=None, args={}):
- super().__init__(data_locator, args)
- # lock used to protect label file write ops
- self.label_lock = threading.RLock()
- if self.data:
- self._validate_and_initialize()
-
- def update(self, data_locator=None, args={}):
- super().__init__(data_locator, args)
- if self.data:
- self._validate_and_initialize()
-
- @staticmethod
- def _get_default_config():
- return {
- "layout": [],
- "max_category_items": 100,
- "obs_names": None,
- "var_names": None,
- "diffexp_lfc_cutoff": 0.01,
- "annotations": False,
- "annotations_file": None,
- "annotations_output_dir": None,
- "annotations_cell_ontology_enabled": False,
- "annotations_cell_ontology_obopath": None,
- "annotations_cell_ontology_terms": None,
- "backed": False,
- "disable_diffexp": False,
- "diffexp_may_be_slow": False,
- }
-
- def get_config_parameters(self, uid=None, collection=None):
- params = {
- "max-category-items": self.config["max_category_items"],
- "disable-diffexp": self.config["disable_diffexp"],
- "diffexp-may-be-slow": self.config["diffexp_may_be_slow"],
- "annotations": self.config["annotations"],
- "annotations_cell_ontology_enabled": self.config["annotations_cell_ontology_enabled"],
- "annotations_cell_ontology_terms": self.config["annotations_cell_ontology_terms"],
- }
- if self.config["annotations"]:
- if uid is not None:
- params.update({"annotations-user-data-idhash": self.get_userdata_idhash(uid)})
- if self.config["annotations_file"] is not None:
- # user has hard-wired the name of the annotation data collection
- fname = os.path.basename(self.config["annotations_file"])
- collection_fname = os.path.splitext(fname)[0]
- params.update(
- {
- "annotations-data-collection-is-read-only": True,
- "annotations-data-collection-name": collection_fname,
- }
- )
- elif collection is not None:
- params.update(
- {"annotations-data-collection-is-read-only": False, "annotations-data-collection-name": collection}
- )
- return params
-
- @staticmethod
- def _create_unique_column_name(df, col_name_prefix):
- """ given the columns of a dataframe, and a name prefix, return a column name which
- does not exist in the dataframe, AND which is prefixed by `prefix`
-
- The approach is to append a numeric suffix, starting at zero and increasing by
- one, until an unused name is found (eg, prefix_0, prefix_1, ...).
- """
- suffix = 0
- while f"{col_name_prefix}{suffix}" in df:
- suffix += 1
- return f"{col_name_prefix}{suffix}"
-
- def _alias_annotation_names(self):
- """
- The front-end relies on the existance of a unique, human-readable
- index for obs & var (eg, var is typically gene name, obs the cell name).
- The user can specify these via the --obs-names and --var-names config.
- If they are not specified, use the existing index to create them, giving
- the resulting column a unique name (eg, "name").
-
- In both cases, enforce that the result is unique, and communicate the
- index column name to the front-end via the obs_names and var_names config
- (which is incorporated into the schema).
- """
- self.original_obs_index = self.data.obs.index
-
- for (ax_name, config_name) in ((Axis.OBS, "obs_names"), (Axis.VAR, "var_names")):
- name = self.config[config_name]
- df_axis = getattr(self.data, str(ax_name))
- if name is None:
- # Default: create unique names from index
- if not df_axis.index.is_unique:
- raise KeyError(
- f"Values in {ax_name}.index must be unique. "
- "Please prepare data to contain unique index values, or specify an "
- "alternative with --{ax_name}-name."
- )
- name = self._create_unique_column_name(df_axis.columns, "name_")
- self.config[config_name] = name
- # reset index to simple range; alias name to point at the
- # previously specified index.
- df_axis.rename_axis(name, inplace=True)
- df_axis.reset_index(inplace=True)
- elif name in df_axis.columns:
- # User has specified alternative column for unique names, and it exists
- if not df_axis[name].is_unique:
- raise KeyError(
- f"Values in {ax_name}.{name} must be unique. " "Please prepare data to contain unique values."
- )
- df_axis.reset_index(drop=True, inplace=True)
- else:
- # user specified a non-existent column name
- raise KeyError(f"Annotation name {name}, specified in --{ax_name}-name does not exist.")
-
- @staticmethod
- def _can_cast_to_float32(ann):
- if ann.dtype.kind == "f":
- if not np.can_cast(ann.dtype, np.float32):
- warnings.warn(f"Annotation {ann.name} will be converted to 32 bit float and may lose precision.")
- return True
- return False
-
- @staticmethod
- def _can_cast_to_int32(ann):
- if ann.dtype.kind in ["i", "u"]:
- if np.can_cast(ann.dtype, np.int32):
- return True
- ii32 = np.iinfo(np.int32)
- if ann.min() >= ii32.min and ann.max() <= ii32.max:
- return True
- return False
-
- @staticmethod
- def _get_col_type(col):
- dtype = col.dtype
- data_kind = dtype.kind
- schema = {}
-
- if ScanpyEngine._can_cast_to_float32(col):
- schema["type"] = "float32"
- elif ScanpyEngine._can_cast_to_int32(col):
- schema["type"] = "int32"
- elif dtype == np.bool_:
- schema["type"] = "boolean"
- elif data_kind == "O" and dtype == "object":
- schema["type"] = "string"
- elif data_kind == "O" and dtype == "category":
- schema["type"] = "categorical"
- schema["categories"] = dtype.categories.tolist()
- else:
- raise TypeError(f"Annotations of type {dtype} are unsupported by cellxgene.")
- return schema
-
- @requires_data
- def _create_schema(self):
- self.schema = {
- "dataframe": {"nObs": self.cell_count, "nVar": self.gene_count, "type": str(self.data.X.dtype)},
- "annotations": {
- "obs": {"index": self.config["obs_names"], "columns": []},
- "var": {"index": self.config["var_names"], "columns": []},
- },
- "layout": {"obs": []},
- }
- for ax in Axis:
- curr_axis = getattr(self.data, str(ax))
- for ann in curr_axis:
- ann_schema = {"name": ann, "writable": False}
- ann_schema.update(self._get_col_type(curr_axis[ann]))
- self.schema["annotations"][ax]["columns"].append(ann_schema)
-
- for layout in self.config["layout"]:
- layout_schema = {"name": layout, "type": "float32", "dims": [f"{layout}_0", f"{layout}_1"]}
- self.schema["layout"]["obs"].append(layout_schema)
-
- @requires_data
- def get_schema(self, uid=None, collection=None):
- schema = self.schema # base schema
- # add label obs annotations as needed
- labels = read_labels(self.get_anno_fname(uid, collection))
- if labels is not None and not labels.empty:
- schema = copy.deepcopy(schema)
- for col in labels.columns:
- col_schema = {
- "name": col,
- "writable": True,
- }
- col_schema.update(self._get_col_type(labels[col]))
- schema["annotations"]["obs"]["columns"].append(col_schema)
- return schema
-
- def get_userdata_idhash(self, uid):
- """
- Return a short hash that weakly identifies the user and dataset.
- Used to create safe annotations output file names.
- """
- id = (uid + self.data_locator.abspath()).encode()
- idhash = base64.b32encode(blake2b(id, digest_size=5).digest()).decode("utf-8")
- return idhash
-
- def get_anno_fname(self, uid=None, collection=None):
- """ return the current annotation file name """
- if not self.config["annotations"]:
- return None
-
- if self.config["annotations_file"] is not None:
- return self.config["annotations_file"]
-
- # we need to generate a file name, which we can only do if we have a UID and collection name
- if uid is None or collection is None:
- return None
- idhash = self.get_userdata_idhash(uid)
- return os.path.join(self.get_anno_output_dir(), f"{collection}-{idhash}.csv")
-
- def get_anno_output_dir(self):
- """ return the current annotation output directory """
- if not self.config["annotations"]:
- return None
-
- if self.config["annotations_output_dir"]:
- return self.config["annotations_output_dir"]
-
- if self.config["annotations_file"]:
- return os.path.dirname(os.path.abspath(self.config["annotations_file"]))
-
- return os.getcwd()
-
- def get_anno_backup_dir(self, uid, collection=None):
- """ return the current annotation backup directory """
- if not self.config["annotations"]:
- return None
-
- fname = self.get_anno_fname(uid, collection)
- root, ext = os.path.splitext(fname)
- return f"{root}-backups"
-
- def _load_data(self, data_locator):
- # as of AnnData 0.6.19, backed mode performs initial load fast, but at the
- # cost of significantly slower access to X data.
- try:
- # there is no guarantee data_locator indicates a local file. The AnnData
- # API will only consume local file objects. If we get a non-local object,
- # make a copy in tmp, and delete it after we load into memory.
- with data_locator.local_handle() as lh:
- # as of AnnData 0.6.19, backed mode performs initial load fast, but at the
- # cost of significantly slower access to X data.
- backed = "r" if self.config["backed"] else None
- self.data = anndata.read_h5ad(lh, backed=backed)
-
- except ValueError:
- raise ScanpyFileError(
- "File must be in the .h5ad format. Please read "
- "https://github.com/theislab/scanpy_usage/blob/master/170505_seurat/info_h5ad.md to "
- "learn more about this format. You may be able to convert your file into this format "
- "using `cellxgene prepare`, please run `cellxgene prepare --help` for more "
- "information."
- )
- except MemoryError:
- raise ScanpyFileError("Out of memory - file is too large for available memory.")
- except Exception as e:
- raise ScanpyFileError(
- f"{e} - file not found or is inaccessible. File must be an .h5ad object. "
- f"Please check your input and try again."
- )
-
- @requires_data
- def _validate_and_initialize(self):
- if anndata_version_is_pre_070() and self.config['backed']:
- warnings.warn(f"Use of --backed mode with anndata versions older than 0.7 will have serious "
- "performance issues. Please update to at least anndata 0.7 or later.")
-
- # var and obs column names must be unique
- if not self.data.obs.columns.is_unique or not self.data.var.columns.is_unique:
- raise KeyError(f"All annotation column names must be unique.")
-
- self._alias_annotation_names()
- self._validate_data_types()
- self.cell_count = self.data.shape[0]
- self.gene_count = self.data.shape[1]
- self._default_and_validate_layouts()
- self._create_schema()
-
- # if the user has specified a fixed label file, go ahead and validate it
- # so that we can remove errors early in the process.
- if self.config["annotations_file"]:
- self._validate_label_data(read_labels(self.get_anno_fname()))
-
- # heuristic
- n_values = self.data.shape[0] * self.data.shape[1]
- if (n_values > 1e8 and self.config["backed"] is True) or (n_values > 5e8):
- self.config.update({"diffexp_may_be_slow": True})
-
- @requires_data
- def _default_and_validate_layouts(self):
- """ function:
- a) generate list of default layouts, if not already user specified
- b) validate layouts are legal. remove/warn on any that are not
- c) cap total list of layouts at global const MAX_LAYOUTS
- """
- layouts = self.config["layout"]
- # handle default
- if layouts is None or len(layouts) == 0:
- # load default layouts from the data.
- layouts = [key[2:] for key in self.data.obsm_keys() if type(key) == str and key.startswith("X_")]
- if len(layouts) == 0:
- raise PrepareError(f"Unable to find any precomputed layouts within the dataset.")
-
- # remove invalid layouts
- valid_layouts = []
- obsm_keys = self.data.obsm_keys()
- for layout in layouts:
- layout_name = f"X_{layout}"
- if layout_name not in obsm_keys:
- warnings.warn(f"Ignoring unknown layout name: {layout}.")
- elif not self._is_valid_layout(self.data.obsm[layout_name]):
- warnings.warn(f"Ignoring layout due to malformed shape or data type: {layout}")
- else:
- valid_layouts.append(layout)
-
- if len(valid_layouts) == 0:
- raise PrepareError(f"No valid layout data.")
-
- # cap layouts to MAX_LAYOUTS
- self.config["layout"] = valid_layouts[0:MAX_LAYOUTS]
-
- @requires_data
- def _is_valid_layout(self, arr):
- """ return True if this layout data is a valid array for front-end presentation:
- * ndarray, with shape (n_obs, >= 2), dtype float/int/uint
- * contains only finite values
- """
- is_valid = type(arr) == np.ndarray and arr.dtype.kind in "fiu"
- is_valid = is_valid and arr.shape[0] == self.data.n_obs and arr.shape[1] >= 2
- is_valid = is_valid and np.all(np.isfinite(arr))
- return is_valid
-
- @requires_data
- def _validate_data_types(self):
- # The backed API does not support interrogation of the underlying sparsity or sparse matrix type
- # Fake it by asking for a small subarray and testing it. NOTE: if the user has ignored our
- # anndata <= 0.7 warning, opted for the --backed option, and specified a large, sparse dataset,
- # this "small" indexing request will load the entire X array. This is due to a bug in anndata<=0.7
- # which will load the entire X matrix to fullfill any slicing request if X is sparse. See
- # user warning in _load_data().
- X0 = self.data.X[0, 0:1]
- if sparse.isspmatrix(X0) and not sparse.isspmatrix_csc(X0):
- warnings.warn(
- f"Scanpy data matrix is sparse, but not a CSC (columnar) matrix. "
- f"Performance may be improved by using CSC."
- )
- if self.data.X.dtype != "float32":
- warnings.warn(
- f"Scanpy data matrix is in {self.data.X.dtype} format not float32. " f"Precision may be truncated."
- )
- for ax in Axis:
- curr_axis = getattr(self.data, str(ax))
- for ann in curr_axis:
- datatype = curr_axis[ann].dtype
- downcast_map = {
- "int64": "int32",
- "uint32": "int32",
- "uint64": "int32",
- "float64": "float32",
- }
- if datatype in downcast_map:
- warnings.warn(
- f"Scanpy annotation {ax}:{ann} is in unsupported format: {datatype}. "
- f"Data will be downcast to {downcast_map[datatype]}."
- )
- if isinstance(datatype, CategoricalDtype):
- category_num = len(curr_axis[ann].dtype.categories)
- if category_num > 500 and category_num > self.config["max_category_items"]:
- warnings.warn(
- f"{str(ax).title()} annotation '{ann}' has {category_num} categories, this may be "
- f"cumbersome or slow to display. We recommend setting the "
- f"--max-category-items option to 500, this will hide categorical "
- f"annotations with more than 500 categories in the UI"
- )
-
- @requires_data
- def _validate_label_data(self, labels):
- """
- labels is None if disabled, empty if enabled by no data
- """
- if labels is None or labels.empty:
- return
-
- # all lables must have a name, which must be unique and not used in obs column names
- if not labels.columns.is_unique:
- raise KeyError(f"All column names specified in user annotations must be unique.")
-
- # the label index must be unique, and must have same values the anndata obs index
- if not labels.index.is_unique:
- raise KeyError(f"All row index values specified in user annotations must be unique.")
-
- if not labels.index.equals(self.original_obs_index):
- raise KeyError(
- "Label file row index does not match H5AD file index. "
- "Please ensure that column zero (0) in the label file contain the same "
- "index values as the H5AD file."
- )
-
- duplicate_columns = list(set(labels.columns) & set(self.data.obs.columns))
- if len(duplicate_columns) > 0:
- raise KeyError(
- f"Labels file may not contain column names which overlap " f"with h5ad obs columns {duplicate_columns}"
- )
-
- # labels must have same count as obs annotations
- if labels.shape[0] != self.data.obs.shape[0]:
- raise ValueError("Labels file must have same number of rows as h5ad file.")
-
- @staticmethod
- def _annotation_filter_to_mask(filter, d_axis, count):
- mask = np.ones((count,), dtype=bool)
- for v in filter:
- if d_axis[v["name"]].dtype.name in ["boolean", "category", "object"]:
- key_idx = np.in1d(getattr(d_axis, v["name"]), v["values"])
- mask = np.logical_and(mask, key_idx)
- else:
- min_ = v.get("min", None)
- max_ = v.get("max", None)
- if min_ is not None:
- key_idx = (getattr(d_axis, v["name"]) >= min_).ravel()
- mask = np.logical_and(mask, key_idx)
- if max_ is not None:
- key_idx = (getattr(d_axis, v["name"]) <= max_).ravel()
- mask = np.logical_and(mask, key_idx)
- return mask
-
- @staticmethod
- def _index_filter_to_mask(filter, count):
- mask = np.zeros((count,), dtype=bool)
- for i in filter:
- if type(i) == list:
- mask[i[0] : i[1]] = True
- else:
- mask[i] = True
- return mask
-
- @staticmethod
- def _axis_filter_to_mask(filter, d_axis, count):
- mask = np.ones((count,), dtype=bool)
- if "index" in filter:
- mask = np.logical_and(mask, ScanpyEngine._index_filter_to_mask(filter["index"], count))
- if "annotation_value" in filter:
- mask = np.logical_and(
- mask, ScanpyEngine._annotation_filter_to_mask(filter["annotation_value"], d_axis, count),
- )
- return mask
-
- @requires_data
- def _filter_to_mask(self, filter, use_slices=True):
- if use_slices:
- obs_selector = slice(0, self.data.n_obs)
- var_selector = slice(0, self.data.n_vars)
- else:
- obs_selector = None
- var_selector = None
-
- if filter is not None:
- if Axis.OBS in filter:
- obs_selector = self._axis_filter_to_mask(filter["obs"], self.data.obs, self.data.n_obs)
- if Axis.VAR in filter:
- var_selector = self._axis_filter_to_mask(filter["var"], self.data.var, self.data.n_vars)
- return obs_selector, var_selector
-
- @requires_data
- def annotation_to_fbs_matrix(self, axis, fields=None, uid=None, collection=None):
- if axis == Axis.OBS:
- if self.config["annotations"]:
- try:
- labels = read_labels(self.get_anno_fname(uid, collection))
- except Exception as e:
- raise ScanpyFileError(
- f"Error while loading label file: {e}, File must be in the .csv format, please check "
- f"your input and try again."
- )
- else:
- labels = None
-
- if labels is not None and not labels.empty:
- df = self.data.obs.join(labels, self.config["obs_names"])
- else:
- df = self.data.obs
- else:
- df = self.data.var
- if fields is not None and len(fields) > 0:
- df = df[fields]
- return encode_matrix_fbs(df, col_idx=df.columns)
-
- @requires_data
- def annotation_put_fbs(self, axis, fbs, uid=None, collection=None):
- if not self.config["annotations"]:
- raise DisabledFeatureError("Writable annotations are not enabled")
-
- fname = self.get_anno_fname(uid, collection)
- if not fname:
- raise ScanpyFileError("Writable annotations - unable to determine file name for annotations")
-
- if axis != Axis.OBS:
- raise ValueError("Only OBS dimension access is supported")
-
- new_label_df = decode_matrix_fbs(fbs)
- if not new_label_df.empty:
- new_label_df.index = self.original_obs_index
- self._validate_label_data(new_label_df) # paranoia
-
- # if any of the new column labels overlap with our existing labels, raise error
- duplicate_columns = list(set(new_label_df.columns) & set(self.data.obs.columns))
- if not new_label_df.columns.is_unique or len(duplicate_columns) > 0:
- raise KeyError(
- f"Labels file may not contain column names which overlap " f"with h5ad obs columns {duplicate_columns}"
- )
-
- # update our internal state and save it. Multi-threading often enabled,
- # so treat this as a critical section.
- with self.label_lock:
- lastmod = self.data_locator.lastmodtime()
- lastmodstr = "'unknown'" if lastmod is None else lastmod.isoformat(timespec="seconds")
- header = (
- f"# Annotations generated on {datetime.now().isoformat(timespec='seconds')} "
- f"using cellxgene version {cellxgene_version}\n"
- f"# Input data file was {self.data_locator.uri_or_path}, "
- f"which was last modified on {lastmodstr}\n"
- )
- write_labels(fname, new_label_df, header, backup_dir=self.get_anno_backup_dir(uid, collection))
-
- return jsonify_scanpy({"status": "OK"})
-
- @requires_data
- def data_frame_to_fbs_matrix(self, filter, axis):
- """
- Retrieves data 'X' and returns in a flatbuffer Matrix.
- :param filter: filter: dictionary with filter params
- :param axis: string obs or var
- :return: flatbuffer Matrix
-
- Caveats:
- * currently only supports access on VAR axis
- * currently only supports filtering on VAR axis
- """
- if axis != Axis.VAR:
- raise ValueError("Only VAR dimension access is supported")
- try:
- obs_selector, var_selector = self._filter_to_mask(filter, use_slices=False)
- except (KeyError, IndexError, TypeError) as e:
- raise FilterError(f"Error parsing filter: {e}") from e
- if obs_selector is not None:
- raise FilterError("filtering on obs unsupported")
-
- # Currently only handles VAR dimension
- X = self.data.X[:, slice(None) if var_selector is None else var_selector]
- col_idx = np.nonzero([] if var_selector is None else var_selector)[0]
- return encode_matrix_fbs(X, col_idx=col_idx, row_idx=None)
-
- @requires_data
- def diffexp_topN(self, obsFilterA, obsFilterB, top_n=None, interactive_limit=None):
- if Axis.VAR in obsFilterA or Axis.VAR in obsFilterB:
- raise FilterError("Observation filters may not contain vaiable conditions")
- try:
- obs_mask_A = self._axis_filter_to_mask(obsFilterA["obs"], self.data.obs, self.data.n_obs)
- obs_mask_B = self._axis_filter_to_mask(obsFilterB["obs"], self.data.obs, self.data.n_obs)
- except (KeyError, IndexError) as e:
- raise FilterError(f"Error parsing filter: {e}") from e
- if top_n is None:
- top_n = DEFAULT_TOP_N
- result = diffexp_ttest(self.data, obs_mask_A, obs_mask_B, top_n, self.config["diffexp_lfc_cutoff"])
- try:
- return jsonify_scanpy(result)
- except ValueError:
- raise JSONEncodingValueError("Error encoding differential expression to JSON")
-
- @requires_data
- def layout_to_fbs_matrix(self):
- """
- Return the default 2-D layout for cells as a FBS Matrix.
-
- Caveats:
- * does not support filtering
- * only returns Matrix in columnar layout
-
- All embeddings must be individually centered & scaled (isotropically)
- to a [0, 1] range.
- """
- try:
- layout_data = []
- for layout in self.config["layout"]:
- full_embedding = self.data.obsm[f"X_{layout}"]
- embedding = full_embedding[:, :2]
-
- # scale isotropically
- min = embedding.min(axis=0)
- max = embedding.max(axis=0)
- scale = np.amax(max - min)
- normalized_layout = (embedding - min) / scale
-
- # translate to center on both axis
- translate = 0.5 - ((max - min) / scale / 2)
- normalized_layout = normalized_layout + translate
-
- normalized_layout = normalized_layout.astype(dtype=np.float32)
- layout_data.append(pandas.DataFrame(normalized_layout, columns=[f"{layout}_0", f"{layout}_1"]))
-
- except ValueError as e:
- raise PrepareError(
- f"Layout has not been calculated using {self.config['layout']}, "
- f"please prepare your datafile and relaunch cellxgene"
- ) from e
-
- df = pandas.concat(layout_data, axis=1, copy=False)
- return encode_matrix_fbs(df, col_idx=df.columns, row_idx=None)
diff --git a/server/app/util/errors.py b/server/app/util/errors.py
deleted file mode 100644
index 2fe13cc8..00000000
--- a/server/app/util/errors.py
+++ /dev/null
@@ -1,70 +0,0 @@
-class FilterError(Exception):
- """
- Raised when filter is malformed
- """
-
- def __init__(self, message):
- self.message = message
-
-
-class InteractiveError(Exception):
- """
- Raised when computation would exceed interactive time
- """
-
- def __init__(self, message):
- self.message = message
-
-
-class JSONEncodingValueError(Exception):
- """
- Raised when file loaded into scanpy is misformatted
- """
-
- def __init__(self, message):
- self.message = message
-
-
-class MimeTypeError(Exception):
- """
- Raised when incompatible MIME type selected
- """
-
- def __init__(self, message):
- self.message = message
-
-
-class PrepareError(Exception):
- """
- Raised when data is misprepared
- """
-
- def __init__(self, message):
- self.message = message
-
-
-class ScanpyFileError(Exception):
- """
- Raised when file loaded into scanpy is misformatted
- """
-
- def __init__(self, message):
- self.message = message
-
-
-class DriverError(Exception):
- """
- Raised when file loaded into scanpy is misformatted
- """
-
- def __init__(self, message):
- self.message = message
-
-
-class DisabledFeatureError(Exception):
- """
- Raised when an attempt to use a disabled feature occurs
- """
-
- def __init__(self, message):
- self.message = message
diff --git a/server/app/util/ontology.py b/server/app/util/ontology.py
deleted file mode 100644
index 1c205ca0..00000000
--- a/server/app/util/ontology.py
+++ /dev/null
@@ -1,37 +0,0 @@
-"""
-Load and parse ontologies - currently support OBO files only.
-"""
-import fsspec
-import fastobo
-import traceback # use built-in formatter for SyntaxError
-
-
-""" our default ontology is the PURL for the Cell Ontology. See http://www.obofoundry.org/ontology/cl.html """
-DefaultOnotology = "http://purl.obolibrary.org/obo/cl.obo"
-
-
-class OntologyLoadFailure(Exception):
- pass
-
-
-def load_obo(path):
- """ given a URI or path, return an array of term names """
- if path is None:
- path = DefaultOnotology
-
- try:
- with fsspec.open(path) as f:
- obo = fastobo.iter(f)
- terms = filter(lambda stanza: type(stanza) is fastobo.term.TermFrame, obo)
- names = [tag.name for term in terms for tag in term if type(tag) is fastobo.term.NameClause]
- return names
-
- except FileNotFoundError as e:
- raise OntologyLoadFailure(f"Unable to find OBO ontology path: {path}") from e
-
- except SyntaxError as e:
- msg = ''.join(traceback.format_exception_only(SyntaxError, e))
- raise OntologyLoadFailure(msg) from e
-
- except Exception as e:
- raise OntologyLoadFailure(f"Error loading OBO file {path}") from e
diff --git a/server/app/util/utils.py b/server/app/util/utils.py
deleted file mode 100644
index c931ae1b..00000000
--- a/server/app/util/utils.py
+++ /dev/null
@@ -1,44 +0,0 @@
-from functools import wraps
-
-from flask import json
-from numpy import float32, integer
-
-from server.app.util.errors import DriverError
-
-
-class Float32JSONEncoder(json.JSONEncoder):
- def __init__(self, *args, **kwargs):
- """
- NaN/Infinities are illegal in standard JSON. Python extends JSON with
- non-standard symbols that most JavaScript JSON parsers do not understand.
- The `allow_nan` parameter will force Python simplejson to throw an ValueError
- if it runs into non-finite floating point values which are unsupported by
- standard JSON.
- """
- kwargs["allow_nan"] = False
- super().__init__(*args, **kwargs)
-
- def default(self, obj):
- if isinstance(obj, float32):
- return float(obj)
- elif isinstance(obj, integer):
- return int(obj)
- return json.JSONEncoder.default(self, obj)
-
-
-def custom_format_warning(msg, *args, **kwargs):
- return f"[cellxgene] Warning: {msg} \n"
-
-
-def jsonify_scanpy(data):
- return json.dumps(data, cls=Float32JSONEncoder, allow_nan=False)
-
-
-def requires_data(func):
- @wraps(func)
- def wrapped_function(self, *args, **kwargs):
- if self.data is None:
- raise DriverError(f"error data must be loaded before you call {func.__name__}")
- return func(self, *args, **kwargs)
-
- return wrapped_function
diff --git a/server/app/web/webapp.py b/server/app/web/webapp.py
deleted file mode 100644
index 8af94640..00000000
--- a/server/app/web/webapp.py
+++ /dev/null
@@ -1,17 +0,0 @@
-import os
-from flask import Blueprint, render_template, send_from_directory, current_app
-
-
-bp = Blueprint("webapp", __name__, template_folder="templates")
-
-
-@bp.route("/")
-def index():
- dataset_title = current_app.config["DATASET_TITLE"]
- scripts = current_app.config["SCRIPTS"]
- return render_template("index.html", datasetTitle=dataset_title, SCRIPTS=scripts)
-
-
-@bp.route("/favicon.png")
-def favicon():
- return send_from_directory(os.path.join(bp.root_path, "static/img/"), "favicon.png")
diff --git a/server/cli/launch.py b/server/cli/launch.py
index 759821b5..db50ed4c 100644
--- a/server/cli/launch.py
+++ b/server/cli/launch.py
@@ -10,67 +10,20 @@ from urllib.parse import urlparse
import click
-from server.app.app import Server
-from server.app.util.errors import ScanpyFileError
-from server.app.util.utils import custom_format_warning
-from server.utils.utils import find_available_port, is_port_available, sort_options
-from server.app.util.data_locator import DataLocator
-from server.app.util.ontology import load_obo, OntologyLoadFailure
+from server.common.utils import custom_format_warning
+from server.common.utils import find_available_port, is_port_available, sort_options
+from server.data_common.matrix_loader import MatrixDataLoader, MatrixDataCacheManager
+from server.common.annotations import AnnotationsLocalFile
+from server.common.app_config import AppConfig
+
+from server.common.errors import OntologyLoadFailure
# anything bigger than this will generate a special message
BIG_FILE_SIZE_THRESHOLD = 100 * 2 ** 20 # 100MB
DEFAULT_SERVER_PORT = int(environ.get('CXG_SERVER_PORT', '5005'))
-def common_args(func):
- """
- Decorator to contain CLI args that will be common to both CLI and GUI: title and engine args.
- """
-
- @click.option("--title", "-t", metavar="", help="Title to display. If omitted will use file name.")
- @click.option(
- "--about",
- metavar="",
- help="URL providing more information about the dataset " "(hint: must be a fully specified absolute URL).",
- )
- @click.option(
- "--embedding",
- "-e",
- default=[],
- multiple=True,
- show_default=False,
- metavar="",
- help="Embedding name, eg, 'umap'. Repeat option for multiple embeddings. Defaults to all.",
- )
- @click.option(
- "--obs-names",
- "-obs",
- default=None,
- metavar="",
- help="Name of annotation field to use for observations. If not specified cellxgene will use the the obs index.",
- )
- @click.option(
- "--var-names",
- "-var",
- default=None,
- metavar="",
- help="Name of annotation to use for variables. If not specified cellxgene will use the the var index.",
- )
- @click.option(
- "--max-category-items",
- default=1000,
- metavar="",
- show_default=True,
- help="Will not display categories with more distinct values than specified.",
- )
- @click.option(
- "--diffexp-lfc-cutoff",
- "-de",
- default=0.01,
- show_default=True,
- metavar="",
- help="Minimum log fold change threshold for differential expression.",
- )
+def annotation_args(func):
@click.option(
"--experimental-annotations",
is_flag=True,
@@ -101,27 +54,14 @@ def common_args(func):
is_flag=True,
default=False,
show_default=True,
- help="When creating annotations, optionally autocomplete names from ontology terms.",)
+ help="When creating annotations, optionally autocomplete names from ontology terms."
+ )
@click.option(
"--experimental-annotations-ontology-obo",
default=None,
show_default=True,
metavar="",
- help="Location of OBO file defining cell annotatoin autosuggest terms.",)
- @click.option(
- "--backed",
- "-b",
- is_flag=True,
- default=False,
- show_default=False,
- help="Load data in file-backed mode. This may save memory, but may result in slower overall performance.",
- )
- @click.option(
- "--disable-diffexp",
- is_flag=True,
- default=False,
- show_default=False,
- help="Disable on-demand differential expression.",
+ help="Location of OBO file defining cell annotation autosuggest terms."
)
@functools.wraps(func)
def wrapper(*args, **kwargs):
@@ -130,41 +70,188 @@ def common_args(func):
return wrapper
-def parse_engine_args(
- embedding,
- obs_names,
- var_names,
- max_category_items,
- diffexp_lfc_cutoff,
- experimental_annotations,
- experimental_annotations_file,
- experimental_annotations_output_dir,
- backed,
- disable_diffexp,
- experimental_annotations_ontology,
- experimental_annotations_ontology_obo
-):
- annotations_file = experimental_annotations_file if experimental_annotations else None
- annotations_output_dir = experimental_annotations_output_dir if experimental_annotations else None
- annotations_cell_ontology_enabled = experimental_annotations and (
- experimental_annotations_ontology or bool(experimental_annotations_ontology_obo)
+def config_args(func):
+ @click.option(
+ "--max-category-items",
+ default=1000,
+ metavar="",
+ show_default=True,
+ help="Will not display categories with more distinct values than specified.",
)
- annotations_ontology_obopath = experimental_annotations_ontology_obo if annotations_cell_ontology_enabled else None
- return {
- "layout": embedding,
- "max_category_items": max_category_items,
- "diffexp_lfc_cutoff": diffexp_lfc_cutoff,
- "obs_names": obs_names,
- "var_names": var_names,
- "annotations": experimental_annotations,
- "annotations_file": annotations_file,
- "annotations_output_dir": annotations_output_dir,
- "annotations_cell_ontology_enabled": annotations_cell_ontology_enabled,
- "annotations_cell_ontology_obopath": annotations_ontology_obopath,
- "annotations_cell_ontology_terms": None,
- "backed": backed,
- "disable_diffexp": disable_diffexp,
- }
+ @click.option(
+ "--diffexp-lfc-cutoff",
+ "-de",
+ default=0.01,
+ show_default=True,
+ metavar="",
+ help="Minimum log fold change threshold for differential expression.",
+ )
+ @click.option(
+ "--disable-diffexp",
+ is_flag=True,
+ default=False,
+ show_default=False,
+ help="Disable on-demand differential expression.",
+ )
+ @click.option(
+ "--embedding",
+ "-e",
+ default=[],
+ multiple=True,
+ show_default=False,
+ metavar="",
+ help="Embedding name, eg, 'umap'. Repeat option for multiple embeddings. Defaults to all.",
+ )
+ @functools.wraps(func)
+ def wrapper(*args, **kwargs):
+ return func(*args, **kwargs)
+
+ return wrapper
+
+
+def dataset_args(func):
+
+ @click.option(
+ "--obs-names",
+ "-obs",
+ default=None,
+ metavar="",
+ help="Name of annotation field to use for observations. If not specified cellxgene will use the the obs index.",
+ )
+ @click.option(
+ "--var-names",
+ "-var",
+ default=None,
+ metavar="",
+ help="Name of annotation to use for variables. If not specified cellxgene will use the the var index.",
+ )
+ @click.option(
+ "--backed",
+ "-b",
+ is_flag=True,
+ default=False,
+ show_default=False,
+ help="Load anndata in file-backed mode. "
+ "This may save memory, but may result in slower overall performance.",
+ )
+ @click.option(
+ "--title",
+ "-t",
+ metavar="",
+ help="Title to display. If omitted will use file name."
+ )
+ @click.option(
+ "--about",
+ metavar="",
+ help="URL providing more information about the dataset " "(hint: must be a fully specified absolute URL).",
+ )
+ @functools.wraps(func)
+ def wrapper(*args, **kwargs):
+ return func(*args, **kwargs)
+
+ return wrapper
+
+
+def server_args(func):
+ @click.option(
+ "--debug",
+ "-d",
+ is_flag=True,
+ default=False,
+ show_default=True,
+ help="Run in debug mode. This is helpful for cellxgene developers, "
+ "or when you want more information about an error condition.",
+ )
+ @click.option(
+ "--verbose",
+ "-v",
+ is_flag=True,
+ default=False,
+ show_default=True,
+ help="Provide verbose output, including warnings and all server requests.",
+ )
+ @click.option(
+ "--port",
+ "-p",
+ metavar="",
+ default=DEFAULT_SERVER_PORT,
+ show_default=True,
+ help="Port to run server on. If not specified cellxgene will find an available port.",
+ )
+ @click.option(
+ "--host",
+ metavar="",
+ default="127.0.0.1",
+ show_default=False,
+ help="Host IP address. By default cellxgene will use localhost (e.g. 127.0.0.1).",
+ )
+ @click.option(
+ "--scripts",
+ "-s",
+ default=[],
+ multiple=True,
+ metavar="",
+ help="Additional script files to include in HTML page. If not specified, "
+ "no additional script files will be included.",
+ show_default=False,
+ )
+ @functools.wraps(func)
+ def wrapper(*args, **kwargs):
+ return func(*args, **kwargs)
+
+ return wrapper
+
+
+def launch_args(func):
+ @annotation_args
+ @config_args
+ @dataset_args
+ @server_args
+ @click.option(
+ "--dataroot",
+ default=None,
+ metavar="",
+ help="Enable cellxgene to serve multiple files. Supply path (local directory or URL)"
+ " to folder containing H5AD and/or CXG datasets.",
+ hidden=True) # TODO, unhide when dataroot is supported)
+ @click.argument("datapath", required=False, metavar="")
+ @click.option(
+ "--open",
+ "-o",
+ "open_browser",
+ is_flag=True,
+ default=False,
+ show_default=True,
+ help="Open web browser after launch.",
+ )
+ @click.help_option("--help", "-h", help="Show this message and exit.")
+ @functools.wraps(func)
+ def wrapper(*args, **kwargs):
+ return func(*args, **kwargs)
+
+ return wrapper
+
+
+def handle_scripts(scripts):
+ if scripts:
+ click.echo(
+ r"""
+ / / /\ \ \__ _ _ __ _ __ (_)_ __ __ _
+ \ \/ \/ / _` | '__| '_ \| | '_ \ / _` |
+ \ /\ / (_| | | | | | | | | | | (_| |
+ \/ \/ \__,_|_| |_| |_|_|_| |_|\__, |
+ |___/
+ The --scripts flag is intended for developers to include google analytics etc. You could be opening yourself to a
+ security risk by including the --scripts flag. Make sure you trust the scripts that you are including.
+ """
+ )
+ scripts_pretty = ", ".join(scripts)
+ click.confirm(f"Are you sure you want to inject these scripts: {scripts_pretty}?", abort=True)
+
+
+def handle_verbose(verbose):
+ if not verbose:
+ sys.tracebacklimit = 0
@sort_options
@@ -172,62 +259,10 @@ def parse_engine_args(
short_help="Launch the cellxgene data viewer. " "Run `cellxgene launch --help` for more information.",
options_metavar="",
)
-@click.argument("data", nargs=1, metavar="", required=True)
-@click.option(
- "--verbose",
- "-v",
- is_flag=True,
- default=False,
- show_default=True,
- help="Provide verbose output, including warnings and all server requests.",
-)
-@click.option(
- "--debug",
- "-d",
- is_flag=True,
- default=False,
- show_default=True,
- help="Run in debug mode. This is helpful for cellxgene developers, "
- "or when you want more information about an error condition.",
-)
-@click.option(
- "--open",
- "-o",
- "open_browser",
- is_flag=True,
- default=False,
- show_default=True,
- help="Open web browser after launch.",
-)
-@click.option(
- "--port",
- "-p",
- metavar="",
- default=DEFAULT_SERVER_PORT,
- show_default=True,
- help="Port to run server on. If not specified cellxgene will find an available port.",
-)
-@click.option(
- "--host",
- metavar="",
- default="127.0.0.1",
- show_default=False,
- help="Host IP address. By default cellxgene will use localhost (e.g. 127.0.0.1).",
-)
-@click.option(
- "--scripts",
- "-s",
- default=[],
- multiple=True,
- metavar="",
- help="Additional script files to include in HTML page. If not specified, "
- "no additional script files will be included.",
- show_default=False,
-)
-@click.help_option("--help", "-h", help="Show this message and exit.")
-@common_args
+@launch_args
def launch(
- data,
+ datapath,
+ dataroot,
verbose,
debug,
open_browser,
@@ -257,47 +292,41 @@ def launch(
Examples:
- > cellxgene launch example_dataset/pbmc3k.h5ad --title pbmc3k
+ > cellxgene launch example-dataset/pbmc3k.h5ad --title pbmc3k
> cellxgene launch --title
> cellxgene launch """
- e_args = parse_engine_args(
- embedding,
- obs_names,
- var_names,
- max_category_items,
- diffexp_lfc_cutoff,
- experimental_annotations,
- experimental_annotations_file,
- experimental_annotations_output_dir,
- backed,
- disable_diffexp,
- experimental_annotations_ontology,
- experimental_annotations_ontology_obo,
- )
- try:
- data_locator = DataLocator(data)
- except RuntimeError as re:
- raise click.ClickException(f"Unable to access data at {data}. {str(re)}")
+ # TODO Examples to provide when "--dataroot" is unhidden
+ # > cellxgene launch --dataroot example-dataset/
+ #
+ # > cellxgene launch --dataroot
# Startup message
click.echo("[cellxgene] Starting the CLI...")
- # Argument checking
- if data_locator.islocal():
- # if data locator is local, apply file system conventions and other "cheap"
- # validation checks. If a URI, defer until we actually fetch the data and
- # try to read it. Many of these tests don't make sense for URIs (eg, extension-
- # based typing).
- if not data_locator.exists():
- raise click.FileError(data, hint="file does not exist")
- if not data_locator.isfile():
- raise click.FileError(data, hint="data is not a file")
- name, extension = splitext(data)
- if extension != ".h5ad":
- raise click.FileError(basename(data), hint="file type must be .h5ad")
+ if datapath is None and dataroot is None:
+ # TODO: change the error message once dataroot is fully supported
+ raise click.ClickException("Missing argument \".\"")
+ # raise click.ClickException("must supply either or --dataroot")
+ if datapath is not None and dataroot is not None:
+ raise click.ClickException("must supply only one of or --dataroot")
+
+ if datapath:
+ # preload this data set
+ matrix_data_loader = MatrixDataLoader(datapath)
+
+ try:
+ matrix_data_loader.pre_load_validation()
+ except RuntimeError as e:
+ raise click.ClickException(str(e))
+
+ file_size = matrix_data_loader.file_size()
+ if file_size > BIG_FILE_SIZE_THRESHOLD:
+ click.echo(f"[cellxgene] Loading data from {basename(datapath)}, this may take a while...")
+ else:
+ click.echo(f"[cellxgene] Loading data from {basename(datapath)}.")
if debug:
verbose = True
@@ -305,26 +334,11 @@ def launch(
else:
warnings.formatwarning = custom_format_warning
- if not verbose:
- sys.tracebacklimit = 0
+ handle_verbose(verbose)
+ handle_scripts(scripts)
- if scripts:
- click.echo(
- r"""
- / / /\ \ \__ _ _ __ _ __ (_)_ __ __ _
- \ \/ \/ / _` | '__| '_ \| | '_ \ / _` |
- \ /\ / (_| | | | | | | | | | | (_| |
- \/ \/ \__,_|_| |_| |_|_|_| |_|\__, |
- |___/
- The --scripts flag is intended for developers to include google analytics etc. You could be opening yourself to a
- security risk by including the --scripts flag. Make sure you trust the scripts that you are including.
- """
- )
- scripts_pretty = ", ".join(scripts)
- click.confirm(f"Are you sure you want to inject these scripts: {scripts_pretty}?", abort=True)
-
- if not title:
- file_parts = splitext(basename(data))
+ if not title and datapath is not None:
+ file_parts = splitext(basename(datapath))
title = file_parts[0]
if port:
@@ -365,16 +379,7 @@ def launch(
"Unable to create directory specified by " "--experimental-annotations-output-dir"
)
- if e_args.get('annotations_cell_ontology_enabled', False):
- try:
- e_args['annotations_cell_ontology_terms'] = load_obo(
- e_args.get('annotations_cell_ontology_obopath', None)
- )
- except OntologyLoadFailure as e:
- raise click.ClickException("Unable to load ontology terms\n" + str(e))
-
if about:
-
def url_check(url):
try:
result = urlparse(url)
@@ -391,37 +396,61 @@ def launch(
# Setup app
cellxgene_url = f"http://{host}:{port}"
- # Import Flask app
- server = Server()
+ # app config
+ app_config = AppConfig(
+ datapath=datapath,
+ dataroot=dataroot,
+ title=title,
+ about=about,
+ scripts=scripts,
+ layout=embedding,
+ max_category_items=max_category_items,
+ diffexp_lfc_cutoff=diffexp_lfc_cutoff,
+ obs_names=obs_names,
+ var_names=var_names,
+ anndata_backed=backed,
+ disable_diffexp=disable_diffexp)
- server.create_app()
- server.app.config.update(SCRIPTS=scripts)
+ matrix_data_cache_manager = MatrixDataCacheManager()
+ data_adaptor = None
+ if datapath:
+ try:
+ with matrix_data_cache_manager.data_adaptor(datapath, app_config) as data_adaptor:
+ if not disable_diffexp and data_adaptor.parameters.get("diffexp_may_be_slow", False):
+ click.echo(
+ f"[cellxgene] CAUTION: due to the size of your dataset, "
+ f"running differential expression may take longer or fail."
+ )
+ except Exception as e:
+ raise click.ClickException(str(e))
+
+ # create an annotations object. Only AnnotationsLocalFile is used (for now)
+ annotations = None
+
+ if experimental_annotations:
+ annotations = AnnotationsLocalFile(experimental_annotations_output_dir,
+ experimental_annotations_file)
+
+ # if the user has specified a fixed label file, go ahead and validate it
+ # so that we can remove errors early in the process.
+
+ if experimental_annotations_file and data_adaptor:
+ data_adaptor.check_new_labels(annotations.read_labels(data_adaptor))
+
+ if experimental_annotations_ontology or bool(experimental_annotations_ontology_obo):
+ try:
+ annotations.load_ontology(experimental_annotations_ontology_obo)
+ except OntologyLoadFailure as e:
+ raise click.ClickException("Unable to load ontology terms\n" + str(e))
+
+ # create the server
+ from server.app.app import Server
+ server = Server(matrix_data_cache_manager, annotations, app_config)
if not verbose:
log = logging.getLogger("werkzeug")
log.setLevel(logging.ERROR)
- file_size = data_locator.size() if data_locator.islocal() else 0
-
- # if a big file, let the user know it may take a while to load.
- if file_size > BIG_FILE_SIZE_THRESHOLD:
- click.echo(f"[cellxgene] Loading data from {basename(data)}, this may take a while...")
- else:
- click.echo(f"[cellxgene] Loading data from {basename(data)}.")
-
- from server.app.scanpy_engine.scanpy_engine import ScanpyEngine
-
- try:
- server.attach_data(ScanpyEngine(data_locator, e_args), title=title, about=about)
- except ScanpyFileError as e:
- raise click.ClickException(f"{e}")
-
- if not disable_diffexp and server.app.data.config["diffexp_may_be_slow"]:
- click.echo(
- f"[cellxgene] CAUTION: due to the size of your dataset, "
- f"running differential expression may take longer or fail."
- )
-
if open_browser:
click.echo(f"[cellxgene] Launching! Opening your browser to {cellxgene_url} now.")
webbrowser.open(cellxgene_url)
diff --git a/server/cli/prepare.py b/server/cli/prepare.py
index 8fd9e0bf..1d075521 100644
--- a/server/cli/prepare.py
+++ b/server/cli/prepare.py
@@ -4,7 +4,7 @@ import click
from numpy import ndarray, unique
from scipy.sparse.csc import csc_matrix
-from server.utils.utils import sort_options
+from server.common.utils import sort_options
@sort_options
diff --git a/server/common/__init__.py b/server/common/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/common/annotations.py b/server/common/annotations.py
new file mode 100644
index 00000000..6d169045
--- /dev/null
+++ b/server/common/annotations.py
@@ -0,0 +1,247 @@
+from datetime import datetime
+import re
+from uuid import uuid4
+import os
+import pandas as pd
+from hashlib import blake2b
+import base64
+from server import __version__ as cellxgene_version
+import threading
+from server.common.errors import AnnotationsError, OntologyLoadFailure
+from server.common.utils import series_to_schema
+import fsspec
+import fastobo
+import traceback # use built-in formatter for SyntaxError
+from flask import session
+from abc import ABCMeta, abstractmethod
+
+
+class Annotations(metaclass=ABCMeta):
+ """ baseclass for annotations, including ontologies"""
+
+ """ our default ontology is the PURL for the Cell Ontology.
+ See http://www.obofoundry.org/ontology/cl.html """
+ DefaultOnotology = "http://purl.obolibrary.org/obo/cl.obo"
+
+ def __init__(self):
+ self.ontology_data = None
+
+ def load_ontology(self, path):
+ """Load and parse ontologies - currently support OBO files only."""
+ if path is None:
+ path = self.DefaultOnotology
+
+ try:
+ with fsspec.open(path) as f:
+ obo = fastobo.iter(f)
+ terms = filter(lambda stanza: type(stanza) is fastobo.term.TermFrame, obo)
+ names = [tag.name for term in terms for tag in term if type(tag) is fastobo.term.NameClause]
+ self.ontology_data = names
+
+ except FileNotFoundError as e:
+ raise OntologyLoadFailure(f"Unable to find OBO ontology path: {path}") from e
+
+ except SyntaxError as e:
+ msg = ''.join(traceback.format_exception_only(SyntaxError, e))
+ raise OntologyLoadFailure(msg) from e
+
+ except Exception as e:
+ raise OntologyLoadFailure(f"Error loading OBO file {path}") from e
+
+ def get_schema(self, data_adaptor):
+ labels = self.read_labels(data_adaptor)
+ schema = []
+ if labels is not None and not labels.empty:
+ for col in labels.columns:
+ col_schema = dict(name=col, writable=True)
+ col_schema.update(series_to_schema(labels[col]))
+ schema.append(col_schema)
+
+ return schema
+
+ @abstractmethod
+ def set_collection(self, name):
+ """set or create a new annotation collection"""
+ pass
+
+ @abstractmethod
+ def read_labels(self, data_adaptor):
+ """Return the labels as a pandas.DataFrame"""
+ pass
+
+ @abstractmethod
+ def write_labels(self, df, data_adaptor):
+ """Write the labels (df) to a persistent storage such that it can later be read"""
+ pass
+
+ @abstractmethod
+ def update_parameters(self, parameters, data_adaptor):
+ """Update configuration parameters that describe information about the annotations feature"""
+ pass
+
+
+class AnnotationsLocalFile(Annotations):
+
+ CXGUID = "cxguid"
+ CXG_ANNO_COLLECTION = "cxg_anno_collection"
+
+ def __init__(self, output_dir, output_file):
+ super().__init__()
+ self.output_dir = output_dir
+ self.output_file = output_file
+ # lock used to protect label file write ops
+ self.label_lock = threading.RLock()
+
+ def is_safe_collection_name(self, name):
+ """
+ return true if this is a safe collection name
+
+ this is ultra conservative. If we want to allow full legal file name syntax,
+ we could look at modules like `pathvalidate`
+ """
+ if name is None:
+ return False
+ return re.match(r"^[\w\-]+$", name) is not None
+
+ def set_collection(self, name):
+ session[self.CXG_ANNO_COLLECTION] = name
+ session.permanent = True
+
+ def get_collection(self):
+ if session is None:
+ return None
+ return session.get(self.CXG_ANNO_COLLECTION)
+
+ def read_labels(self, data_adaptor):
+ fname = self._get_filename(data_adaptor)
+ if fname is not None and os.path.exists(fname) and os.path.getsize(fname) > 0:
+ return pd.read_csv(fname, dtype="category", index_col=0, header=0, comment="#", keep_default_na=False)
+ else:
+ return pd.DataFrame()
+
+ def write_labels(self, df, data_adaptor):
+ # update our internal state and save it. Multi-threading often enabled,
+ # so treat this as a critical section.
+ with self.label_lock:
+ lastmod = data_adaptor.get_last_mod_time()
+ lastmodstr = "'unknown'" if lastmod is None else lastmod.isoformat(timespec="seconds")
+ header = (
+ f"# Annotations generated on {datetime.now().isoformat(timespec='seconds')} "
+ f"using cellxgene version {cellxgene_version}\n"
+ f"# Input data file was {data_adaptor.get_location()}, "
+ f"which was last modified on {lastmodstr}\n"
+ )
+
+ fname = self._get_filename(data_adaptor)
+ self._backup(fname)
+ if not df.empty:
+ with open(fname, "w", newline="") as f:
+ if header is not None:
+ f.write(header)
+ df.to_csv(f)
+ else:
+ open(fname, "w").close()
+
+ def _get_userid(self):
+ if self.CXGUID not in session:
+ session[self.CXGUID] = uuid4().hex
+ session.permanent = True
+ return session[self.CXGUID]
+
+ def _get_userdata_idhash(self, data_adaptor):
+ """
+ Return a short hash that weakly identifies the user and dataset.
+ Used to create safe annotations output file names.
+ """
+ uid = self._get_userid()
+ id = (uid + data_adaptor.get_location()).encode()
+ idhash = base64.b32encode(blake2b(id, digest_size=5).digest()).decode("utf-8")
+ return idhash
+
+ def _get_output_dir(self):
+ if self.output_dir:
+ return self.output_dir
+
+ if self.output_file:
+ return os.path.dirname(self.path.abspath(self.output_dir))
+
+ return os.getcwd()
+
+ def _get_filename(self, data_adaptor):
+ """ return the current annotation file name """
+ if self.output_file:
+ return self.output_file
+
+ # we need to generate a file name, which we can only do if we have a UID and collection name
+ if session is None:
+ raise AnnotationsError("unable to determine file name for annotations")
+
+ collection = self.get_collection()
+ if collection is None:
+ return None
+
+ if data_adaptor is None:
+ raise AnnotationsError("unable to determine file name for annotations")
+
+ idhash = self._get_userdata_idhash(data_adaptor)
+ return os.path.join(self._get_output_dir(), f"{collection}-{idhash}.csv")
+
+ def _backup(self, fname, max_backups=9):
+ """
+ save N backups of file to backup_dir.
+ 1. fname -> backup_dir/fname-TIME
+ 2. delete excess files in backup_dir
+ """
+ root, ext = os.path.splitext(fname)
+ backup_dir = f"{root}-backups"
+
+ # Make sure there is work to do
+ if not os.path.exists(fname):
+ return
+
+ # Ensure backup_dir exists
+ if not os.path.exists(backup_dir):
+ os.mkdir(backup_dir)
+
+ # Save current file to backup_dir
+ fname_base = os.path.basename(fname)
+ fname_base_root, fname_base_ext = os.path.splitext(fname_base)
+ # don't use ISO standard time format, as it contains characters illegal on some filesytems.
+ nowish = datetime.now().strftime("%Y-%m-%dT%H-%M-%S")
+ backup_fname = os.path.join(backup_dir, f"{fname_base_root}-{nowish}{fname_base_ext}")
+ if os.path.exists(backup_fname):
+ os.remove(backup_fname)
+ os.rename(fname, backup_fname)
+
+ # prune the backup_dir to max number of backup files, keeping the most recent backups
+ backups = list(filter(lambda s: s.startswith(fname_base_root), os.listdir(backup_dir)))
+ excess_count = len(backups) - max_backups
+ if excess_count > 0:
+ backups.sort()
+ for bu in backups[0:excess_count]:
+ os.remove(os.path.join(backup_dir, bu))
+
+ def update_parameters(self, parameters, data_adaptor):
+ params = {}
+ params["annotations"] = True
+
+ if self.ontology_data:
+ params["annotations_cell_ontology_enabled"] = True
+ params["annotations_cell_ontology_terms"] = self.ontology_data
+ else:
+ params["annotations_cell_ontology_enabled"] = False
+
+ if self.output_file is not None:
+ # user has hard-wired the name of the annotation data collection
+ fname = os.path.basename(self.output_file)
+ collection_fname = os.path.splitext(fname)[0]
+ params["annotations-data-collection-is-read-only"] = True
+ params["annotations-data-collection-name"] = collection_fname
+
+ elif session is not None:
+ collection = self.get_collection()
+ params["annotations-user-data-idhash"] = self._get_userdata_idhash(data_adaptor)
+ params["annotations-data-collection-is-read-only"] = False
+ params["annotations-data-collection-name"] = collection
+
+ parameters.update(params)
diff --git a/server/common/app_config.py b/server/common/app_config.py
new file mode 100644
index 00000000..47d8d5a5
--- /dev/null
+++ b/server/common/app_config.py
@@ -0,0 +1,138 @@
+# -*- coding: utf-8 -*-
+
+from server import __version__ as cellxgene_version
+from os.path import basename, splitext
+
+
+class AppFeature(object):
+ def __init__(self, path, available=False, method="POST", extra={}):
+ self.path = path
+ self.available = available
+ self.method = method
+ self.extra = extra
+ for k, v in extra.items():
+ setattr(self, k, v)
+
+ def todict(self):
+ d = dict(
+ available=self.available,
+ method=self.method,
+ path=self.path)
+ d.update(self.extra)
+ return d
+
+
+class AppConfig(object):
+
+ def __init__(self, **kw):
+ super().__init__()
+
+ # app inputs
+ self.datapath = None
+ self.dataroot = None
+ self.title = ""
+ self.about = None
+ self.scripts = []
+ self.layout = None
+ self.max_category_items = 100
+ self.diffexp_lfc_cutoff = 0.01
+ self.disable_diffexp = False
+ self.anndata_backed = False
+
+ # TODO these options may not apply to all datasets in the multi dataset.
+ # may need to invent a way to associate these config parameters with
+ # specific datasets.
+ self.obs_names = None
+ self.var_names = None
+
+ # parameters
+ self.diffexp_may_be_slow = False
+
+ inputs = ["datapath", "dataroot", "title", "about", "scripts", "layout",
+ "max_category_items", "diffexp_lfc_cutoff",
+ "obs_names", "var_names",
+ "anndata_backed", "disable_diffexp"]
+
+ self.update(inputs, kw)
+
+ def update(self, inputs, kw):
+
+ for k, v in kw.items():
+ if k in inputs:
+ setattr(self, k, v)
+ else:
+ raise RuntimeError(f"unknown config parameter {k}.")
+
+ def get_title(self, data_adaptor):
+ if self.title:
+ return self.title
+
+ # TODO: find a place to stash the dataset title, such as a
+ # json file at the same location as the data matrix.
+ # for example, if the dataset is at abc.cxg then a file with
+ # the title and about info could be at abc.cxg.metadata.
+ # for now just return the basename
+ location = data_adaptor.get_location()
+ if location.endswith("/"):
+ location = location[:-1]
+ return splitext(basename(location))[0]
+
+ def get_about(self, data_adaptor):
+ return self.about
+
+ def get_config(self, data_adaptor, annotation=None):
+
+ # FIXME The current set of config is not consistently presented:
+ # we have camalCase, hyphen-text, and underscore_text
+
+ # features
+ features = [f.todict() for f in data_adaptor.get_features().values()]
+
+ # display_names
+ title = self.get_title(data_adaptor)
+ about = self.get_about(data_adaptor)
+
+ display_names = dict(
+ engine=data_adaptor.get_name(),
+ dataset=title)
+
+ # library_versions
+ library_versions = {}
+ library_versions.update(data_adaptor.get_library_versions())
+ library_versions["cellxgene"] = cellxgene_version
+
+ # links
+ links = {"about-dataset" : about}
+
+ # parameters
+ parameters = {
+ "layout": self.layout,
+ "max-category-items": self.max_category_items,
+ "obs_names": self.obs_names,
+ "var_names": self.var_names,
+ "diffexp_lfc_cutoff": self.diffexp_lfc_cutoff,
+ "backed": self.anndata_backed,
+ "disable-diffexp": self.disable_diffexp,
+ "annotations": False,
+ "annotations_file": None,
+ "annotations_output_dir": None,
+ "annotations_cell_ontology_enabled": False,
+ "annotations_cell_ontology_obopath": None,
+ "annotations_cell_ontology_terms": None,
+ "diffexp-may-be-slow": False,
+ }
+
+ data_adaptor.update_parameters(parameters)
+ if annotation:
+ annotation.update_parameters(parameters, data_adaptor)
+
+ # gather it all together
+ c = {}
+ config = c["config"] = {}
+ config["features"] = features
+ config["displayNames"] = display_names
+ config["library_versions"] = library_versions
+ config["links"] = links
+ config["parameters"] = parameters
+
+ return c
diff --git a/server/app/util/constants.py b/server/common/constants.py
similarity index 91%
rename from server/app/util/constants.py
rename to server/common/constants.py
index 545a8d50..c84ddb7c 100644
--- a/server/app/util/constants.py
+++ b/server/common/constants.py
@@ -31,6 +31,3 @@ JSON_NaN_to_num_warning_msg = "JSON encoding failure - please verify all data ar
REACTIVE_LIMIT = 1_000_000
MAX_LAYOUTS = 30
-
-CXGUID = "cxguid"
-CXG_ANNO_COLLECTION = "cxg_anno_collection"
diff --git a/server/app/util/data_locator.py b/server/common/data_locator.py
similarity index 96%
rename from server/app/util/data_locator.py
rename to server/common/data_locator.py
index c16eb40a..666bbbf9 100644
--- a/server/app/util/data_locator.py
+++ b/server/common/data_locator.py
@@ -92,6 +92,10 @@ class DataLocator:
tmp_path = tmp.name
return LocalFilePath(tmp_path, delete=True)
+ def ls(self):
+ paths = self.fs.ls(self.uri_or_path)
+ return [os.path.basename(p) for p in paths]
+
class LocalFilePath:
def __init__(self, tmp_path, delete=False):
diff --git a/server/common/errors.py b/server/common/errors.py
new file mode 100644
index 00000000..03233195
--- /dev/null
+++ b/server/common/errors.py
@@ -0,0 +1,54 @@
+class FilterError(Exception):
+ """
+ Raised when filter is malformed
+ """
+ pass
+
+
+class JSONEncodingValueError(Exception):
+ """
+ Raised when data cannot be encoded into json
+ """
+ pass
+
+
+class MimeTypeError(Exception):
+ """
+ Raised when incompatible MIME type selected
+ """
+ pass
+
+
+class PrepareError(Exception):
+ """
+ Raised when data is misprepared
+ """
+ pass
+
+
+class DatasetAccessError(Exception):
+ """
+ Raised when file loaded into a DataAdaptor is misformatted
+ """
+ pass
+
+
+class DisabledFeatureError(Exception):
+ """
+ Raised when an attempt to use a disabled feature occurs
+ """
+ pass
+
+
+class AnnotationsError(Exception):
+ """
+ Raised when an attempt to use the annotations feature fails
+ """
+ pass
+
+
+class OntologyLoadFailure(Exception):
+ """
+ Raised when reading the ontology file fails
+ """
+ pass
diff --git a/server/common/rest.py b/server/common/rest.py
new file mode 100644
index 00000000..4516f111
--- /dev/null
+++ b/server/common/rest.py
@@ -0,0 +1,192 @@
+from http import HTTPStatus
+import warnings
+import copy
+from flask import make_response, jsonify
+from server.common.constants import Axis, DiffExpMode, JSON_NaN_to_num_warning_msg
+from server.common.errors import (
+ FilterError,
+ JSONEncodingValueError,
+ PrepareError,
+ DisabledFeatureError,
+)
+
+import json
+from server.data_common.fbs.matrix import decode_matrix_fbs
+
+
+def schema_get_helper(data_adaptor, annotations):
+ """helper function to gather the schema from the data source and annotations"""
+ schema = data_adaptor.get_schema()
+ schema = copy.deepcopy(schema)
+
+ # add label obs annotations as needed
+ if annotations is not None:
+ label_schema = annotations.get_schema(data_adaptor)
+ schema["annotations"]["obs"]["columns"].extend(label_schema)
+
+ return schema
+
+
+def schema_get(data_adaptor, annotations):
+ schema = schema_get_helper(data_adaptor, annotations)
+ return make_response(
+ jsonify({"schema": schema}), HTTPStatus.OK
+ )
+
+
+def config_get(app_config, data_adaptor, annotations):
+ config = app_config.get_config(data_adaptor, annotations)
+ return make_response(make_response(jsonify(config), HTTPStatus.OK))
+
+
+def annotations_obs_get(request, data_adaptor, annotations):
+ fields = request.args.getlist("annotation-name", None)
+ preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
+ if preferred_mimetype != "application/octet-stream":
+ return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
+ try:
+ labels = None
+ if annotations:
+ labels = annotations.read_labels(data_adaptor)
+ fbs = data_adaptor.annotation_to_fbs_matrix(Axis.OBS, fields, labels)
+ return make_response(fbs, HTTPStatus.OK, {"Content-Type": "application/octet-stream"})
+ except KeyError:
+ return make_response(f"Error bad key in {fields}", HTTPStatus.BAD_REQUEST)
+ except ValueError as e:
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
+ except Exception as e:
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
+
+
+def annotations_put_fbs_helper(data_adaptor, annotations, fbs):
+ """helper function to write annotations from fbs"""
+ if annotations is None:
+ raise DisabledFeatureError("Writable annotations are not enabled")
+
+ new_label_df = decode_matrix_fbs(fbs)
+ if not new_label_df.empty:
+ data_adaptor.check_new_labels(new_label_df)
+ annotations.write_labels(new_label_df, data_adaptor)
+
+
+def annotations_obs_put(request, data_adaptor, annotations):
+ anno_collection = request.args.get("annotation-collection-name", default=None)
+ fbs = request.get_data()
+ if annotations is None:
+ return make_response("Error, annotations are not configured", HTTPStatus.BAD_REQUEST)
+
+ if anno_collection is not None:
+ if not annotations.is_safe_collection_name(anno_collection):
+ return make_response(f"Error, bad annotation collection name", HTTPStatus.BAD_REQUEST)
+ annotations.set_collection(anno_collection)
+
+ try:
+ annotations_put_fbs_helper(data_adaptor, annotations, fbs)
+ res = json.dumps({"status": "OK"})
+ return make_response(res, HTTPStatus.OK, {"Content-Type": "application/json"})
+ except (ValueError, DisabledFeatureError, KeyError) as e:
+ return make_response(str(e), HTTPStatus.BAD_REQUEST)
+ except Exception as e:
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
+
+
+def annotations_var_get(request, data_adaptor, annotations):
+ fields = request.args.getlist("annotation-name", None)
+ preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
+ if preferred_mimetype != "application/octet-stream":
+ return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
+ try:
+ labels = None
+ if annotations is not None:
+ labels = annotations.read_labels(data_adaptor)
+ return make_response(
+ data_adaptor.annotation_to_fbs_matrix(Axis.VAR, fields, labels),
+ HTTPStatus.OK,
+ {"Content-Type": "application/octet-stream"},
+ )
+ except KeyError:
+ return make_response(f"Error bad key in {fields}", HTTPStatus.BAD_REQUEST)
+ except ValueError as e:
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
+ except Exception as e:
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
+
+
+def data_var_put(request, data_adaptor):
+ preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
+ if preferred_mimetype != "application/octet-stream":
+ return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
+
+ filter_json = request.get_json()
+ filter = filter_json["filter"] if filter_json else None
+ try:
+ return make_response(
+ data_adaptor.data_frame_to_fbs_matrix(filter, axis=Axis.VAR),
+ HTTPStatus.OK,
+ {"Content-Type": "application/octet-stream"},
+ )
+ except FilterError as e:
+ return make_response(str(e), HTTPStatus.BAD_REQUEST)
+ except ValueError as e:
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
+
+
+def diffexp_obs_post(request, data_adaptor):
+ args = request.get_json()
+ # confirm mode is present and legal
+ try:
+ mode = DiffExpMode(args["mode"])
+ except KeyError:
+ return make_response("Error: mode is required", HTTPStatus.BAD_REQUEST)
+ except ValueError:
+ return make_response(f"Error: invalid mode option {args['mode']}", HTTPStatus.BAD_REQUEST)
+ # Validate filters
+ if mode == DiffExpMode.VAR_FILTER or "varFilter" in args:
+ # not NOT_IMPLEMENTED
+ return make_response("mode=varfilter not implemented", HTTPStatus.NOT_IMPLEMENTED)
+ if mode == DiffExpMode.TOP_N and "count" not in args:
+ return make_response("mode=topN requires a count parameter", HTTPStatus.BAD_REQUEST)
+
+ if "set1" not in args:
+ return make_response("set1 is required.", HTTPStatus.BAD_REQUEST)
+ if Axis.VAR in args["set1"]["filter"]:
+ return make_response("Var filter not allowed for set1", HTTPStatus.BAD_REQUEST)
+ # set2
+ if "set2" not in args:
+ return make_response("Set2 as inverse of set1 is not implemented", HTTPStatus.NOT_IMPLEMENTED)
+ if Axis.VAR in args["set2"]["filter"]:
+ return make_response("Var filter not allowed for set2", HTTPStatus.BAD_REQUEST)
+
+ set1_filter = args["set1"]["filter"]
+ set2_filter = args.get("set2", {"filter": {}})["filter"]
+
+ # TODO: implement varfilter mode
+
+ # mode=topN
+ count = args.get("count", None)
+ try:
+ diffexp = data_adaptor.diffexp_topN(set1_filter, set2_filter, count)
+ return make_response(diffexp, HTTPStatus.OK, {"Content-Type": "application/json"})
+ except (ValueError, FilterError) as e:
+ return make_response(str(e), HTTPStatus.BAD_REQUEST)
+ except JSONEncodingValueError as e:
+ # JSON encoding failure, usually due to bad data
+ warnings.warn(JSON_NaN_to_num_warning_msg)
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
+ except ValueError as e:
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
+
+
+def layout_obs_get(request, data_adaptor):
+ preferred_mimetype = request.accept_mimetypes.best_match(["application/octet-stream"])
+ try:
+ if preferred_mimetype == "application/octet-stream":
+ return make_response(
+ data_adaptor.layout_to_fbs_matrix(), HTTPStatus.OK, {"Content-Type": "application/octet-stream"}
+ )
+ else:
+ return make_response(f"Unsupported MIME type '{request.accept_mimetypes}'", HTTPStatus.NOT_ACCEPTABLE)
+ except PrepareError as e:
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
+ except ValueError as e:
+ return make_response(str(e), HTTPStatus.INTERNAL_SERVER_ERROR)
diff --git a/server/common/utils.py b/server/common/utils.py
new file mode 100644
index 00000000..1a421f01
--- /dev/null
+++ b/server/common/utils.py
@@ -0,0 +1,146 @@
+import contextlib
+import errno
+import socket
+from urllib.parse import urlsplit, urljoin
+import os
+from flask import json
+import numpy as np
+import pandas as pd
+import warnings
+
+
+def find_available_port(host, port=5005):
+ """
+ Helper method to find open port on host. Tries 5000 ports incremented from the specified port
+ """
+ # Takes approx 2 seconds to do a scan of 5000 ports on my laptop
+ num_ports_to_try = 5000
+ for port_to_try in range(port, port + num_ports_to_try):
+ if is_port_available(host, port_to_try):
+ return port_to_try
+ raise socket.error(errno.EADDRINUSE, f"No port in range {port} - {port + num_ports_to_try - 1} available.")
+
+
+def is_port_available(host, port):
+ is_available = False
+ with contextlib.closing(socket.socket(socket.AF_INET, socket.SOCK_STREAM)) as s:
+ try:
+ s.bind((host, port))
+ is_available = True
+ except socket.error:
+ pass
+ return is_available
+
+
+def sort_options(command):
+ """
+ Helper for the click options - will sort options in a command, and can
+ be used as a decorator.
+ """
+ command.params.sort(key=lambda p: p.name)
+ return command
+
+
+def path_join(base, *urls):
+ """
+ this is like urllib.parse.urljoin, except it works around the scheme-specific
+ cleverness in the aforementioned code, ignores anything in the url except the path,
+ and accepts more than one url.
+ """
+ if not base.endswith("/"):
+ base += "/"
+ btpl = urlsplit(base)
+ path = btpl.path
+ for url in urls:
+ utpl = urlsplit(url)
+ if btpl.scheme == "":
+ path = os.path.join(path, utpl.path)
+ path = os.path.normpath(path)
+ else:
+ path = urljoin(path, utpl.path)
+ return btpl._replace(path=path).geturl()
+
+
+class Float32JSONEncoder(json.JSONEncoder):
+ def __init__(self, *args, **kwargs):
+ """
+ NaN/Infinities are illegal in standard JSON. Python extends JSON with
+ non-standard symbols that most JavaScript JSON parsers do not understand.
+ The `allow_nan` parameter will force Python simplejson to throw an ValueError
+ if it runs into non-finite floating point values which are unsupported by
+ standard JSON.
+ """
+ kwargs["allow_nan"] = False
+ super().__init__(*args, **kwargs)
+
+ def default(self, obj):
+ if isinstance(obj, np.float32):
+ return float(obj)
+ elif isinstance(obj, np.integer):
+ return int(obj)
+ return json.JSONEncoder.default(self, obj)
+
+
+def custom_format_warning(msg, *args, **kwargs):
+ return f"[cellxgene] Warning: {msg} \n"
+
+
+def jsonify_numpy(data):
+ return json.dumps(data, cls=Float32JSONEncoder, allow_nan=False)
+
+
+def dtype_to_schema(dtype):
+ schema = {}
+ if dtype == np.float32:
+ schema['type'] = 'float32'
+ elif dtype == np.int32:
+ schema['type'] = 'int32'
+ elif dtype == np.bool_:
+ schema['type'] = 'boolean'
+ elif dtype == np.str:
+ schema['type'] = 'string'
+ elif dtype == "category":
+ schema["type"] = "categorical"
+ schema["categories"] = dtype.categories.tolist()
+ else:
+ raise TypeError(
+ f"Annotations of type {dtype} are unsupported."
+ )
+ return schema
+
+
+def can_cast_to_float32(array):
+ if array.dtype.kind == "f":
+ if not np.can_cast(array.dtype, np.float32):
+ warnings.warn(f"Annotation {array.name} will be converted to 32 bit float and may lose precision.")
+ return True
+ return False
+
+
+def can_cast_to_int32(array):
+ if array.dtype.kind in ["i", "u"]:
+ if np.can_cast(array.dtype, np.int32):
+ return True
+ ii32 = np.iinfo(np.int32)
+ if array.min() >= ii32.min and array.max() <= ii32.max:
+ return True
+ return False
+
+
+def series_to_schema(array):
+ assert type(array) == pd.Series
+ try:
+ return dtype_to_schema(array.dtype)
+ except TypeError:
+ dtype = array.dtype
+ data_kind = dtype.kind
+ schema = {}
+ if can_cast_to_float32(array):
+ schema["type"] = "float32"
+ elif can_cast_to_int32(array):
+ schema["type"] = "int32"
+ elif data_kind == "O" and dtype == "object":
+ schema["type"] = "string"
+ else:
+ raise TypeError(f"Annotations of type {dtype} are unsupported.")
+ return schema
diff --git a/server/common/web/__init__.py b/server/common/web/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/compute/__init__.py b/server/compute/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/app/scanpy_engine/diffexp.py b/server/compute/diffexp.py
similarity index 91%
rename from server/app/scanpy_engine/diffexp.py
rename to server/compute/diffexp.py
index af267636..df476901 100644
--- a/server/app/scanpy_engine/diffexp.py
+++ b/server/compute/diffexp.py
@@ -37,7 +37,7 @@ def _mean_var_n(X):
return mean, v, n
-def diffexp_ttest(adata, maskA, maskB, top_n=8, diffexp_lfc_cutoff=0.01):
+def diffexp_ttest(data, maskA, maskB, top_n=8, diffexp_lfc_cutoff=0.01):
"""
Return differential expression statistics for top N variables.
@@ -55,19 +55,22 @@ def diffexp_ttest(adata, maskA, maskB, top_n=8, diffexp_lfc_cutoff=0.01):
- p-values adjusted with Bonferroni correction.
https://en.wikipedia.org/wiki/Bonferroni_correction
- :param adata: anndata dataframe
+ :param data: DataAdaptor instance
:param maskA: observation selection mask for set 1
:param maskB: observation selection mask for set 2
:param top_n: number of variables to return stats for
:param diffexp_lfc_cutoff: minimum
:return: for top N genes, [ varindex, logfoldchange, pval, pval_adj ]
"""
- if top_n > adata.n_obs:
- top_n = adata.n_obs
+ shape = data.get_shape()
+ n_obs = shape[0]
+ n_var = shape[1]
+ if top_n > n_obs:
+ top_n = n_obs
# mean, variance, N - calculate for both selections
- meanA, vA, nA = _mean_var_n(adata.X[maskA, :])
- meanB, vB, nB = _mean_var_n(adata.X[maskB, :])
+ meanA, vA, nA = _mean_var_n(data.get_X_array(maskA, None))
+ meanB, vB, nB = _mean_var_n(data.get_X_array(maskB, None))
# variance / N
vnA = vA / min(nA, nB) # overestimate variance, would normally be nA
@@ -86,7 +89,7 @@ def diffexp_ttest(adata, maskA, maskB, top_n=8, diffexp_lfc_cutoff=0.01):
# p-value
pvals = stats.t.sf(np.abs(tscores), dof) * 2
- pvals_adj = pvals * adata.X.shape[1]
+ pvals_adj = pvals * n_var
pvals_adj[pvals_adj > 1] = 1 # cap adjusted p-value at 1
# logfoldchanges: log2(meanA / meanB)
diff --git a/server/data_anndata/__init__.py b/server/data_anndata/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/data_anndata/anndata_adaptor.py b/server/data_anndata/anndata_adaptor.py
new file mode 100644
index 00000000..7ee3a463
--- /dev/null
+++ b/server/data_anndata/anndata_adaptor.py
@@ -0,0 +1,322 @@
+import warnings
+
+import numpy as np
+from pandas.core.dtypes.dtypes import CategoricalDtype
+import anndata
+from scipy import sparse
+from packaging import version
+
+from server.data_common.data_adaptor import DataAdaptor
+from server.data_common.fbs.matrix import encode_matrix_fbs
+from server.common.utils import series_to_schema
+from server.common.constants import Axis, MAX_LAYOUTS
+from server.common.errors import PrepareError, DatasetAccessError
+from server.common.data_locator import DataLocator
+
+anndata_version = version.parse(str(anndata.__version__)).release
+
+
+def anndata_version_is_pre_070():
+ major = anndata_version[0]
+ minor = anndata_version[1] if len(anndata_version) > 1 else 0
+ return major == 0 and minor < 7
+
+
+class AnndataAdaptor(DataAdaptor):
+
+ def __init__(self, data_locator, config=None):
+ super().__init__(config)
+ self.data = None
+ self.data_locator = data_locator
+ self._load_data(data_locator)
+ self._validate_and_initialize()
+
+ def cleanup(self):
+ pass
+
+ @staticmethod
+ def pre_load_validation(location):
+ data_locator = DataLocator(location)
+ if data_locator.islocal():
+ # if data locator is local, apply file system conventions and other "cheap"
+ # validation checks. If a URI, defer until we actually fetch the data and
+ # try to read it. Many of these tests don't make sense for URIs (eg, extension-
+ # based typing).
+ if not data_locator.exists():
+ raise DatasetAccessError(f"{location} does not exist")
+ if not data_locator.isfile():
+ raise DatasetAccessError(f"{location} is not a file")
+
+ @staticmethod
+ def file_size(location):
+ data_locator = DataLocator(location)
+ return data_locator.size() if data_locator.islocal() else 0
+
+ @staticmethod
+ def open(location, config):
+ data_locator = DataLocator(location)
+ return AnndataAdaptor(data_locator, config)
+
+ def get_location(self):
+ return self.data_locator.uri_or_path
+
+ def get_name(self):
+ return "cellxgene anndata adaptor version"
+
+ def get_library_versions(self):
+ return dict(anndata=str(anndata.__version__))
+
+ @staticmethod
+ def _create_unique_column_name(df, col_name_prefix):
+ """ given the columns of a dataframe, and a name prefix, return a column name which
+ does not exist in the dataframe, AND which is prefixed by `prefix`
+
+ The approach is to append a numeric suffix, starting at zero and increasing by
+ one, until an unused name is found (eg, prefix_0, prefix_1, ...).
+ """
+ suffix = 0
+ while f"{col_name_prefix}{suffix}" in df:
+ suffix += 1
+ return f"{col_name_prefix}{suffix}"
+
+ def _alias_annotation_names(self):
+ """
+ The front-end relies on the existance of a unique, human-readable
+ index for obs & var (eg, var is typically gene name, obs the cell name).
+ The user can specify these via the --obs-names and --var-names config.
+ If they are not specified, use the existing index to create them, giving
+ the resulting column a unique name (eg, "name").
+
+ In both cases, enforce that the result is unique, and communicate the
+ index column name to the front-end via the obs_names and var_names config
+ (which is incorporated into the schema).
+ """
+ self.original_obs_index = self.data.obs.index
+
+ for (ax_name, config_name) in ((Axis.OBS, "obs_names"), (Axis.VAR, "var_names")):
+ name = getattr(self.config, config_name)
+ df_axis = getattr(self.data, str(ax_name))
+ if name is None:
+ # Default: create unique names from index
+ if not df_axis.index.is_unique:
+ raise KeyError(
+ f"Values in {ax_name}.index must be unique. "
+ "Please prepare data to contain unique index values, or specify an "
+ "alternative with --{ax_name}-name."
+ )
+ name = self._create_unique_column_name(df_axis.columns, "name_")
+ self.parameters[config_name] = name
+ # reset index to simple range; alias name to point at the
+ # previously specified index.
+ df_axis.rename_axis(name, inplace=True)
+ df_axis.reset_index(inplace=True)
+ elif name in df_axis.columns:
+ # User has specified alternative column for unique names, and it exists
+ if not df_axis[name].is_unique:
+ raise KeyError(
+ f"Values in {ax_name}.{name} must be unique. " "Please prepare data to contain unique values."
+ )
+ df_axis.reset_index(drop=True, inplace=True)
+ self.parameters[config_name] = name
+ else:
+ # user specified a non-existent column name
+ raise KeyError(f"Annotation name {name}, specified in --{ax_name}-name does not exist.")
+
+ def _create_schema(self):
+ self.schema = {
+ "dataframe": {"nObs": self.cell_count, "nVar": self.gene_count, "type": str(self.data.X.dtype)},
+ "annotations": {
+ "obs": {"index": self.parameters.get("obs_names"), "columns": []},
+
+ "var": {"index": self.parameters.get("var_names"), "columns": []},
+ },
+ "layout": {"obs": []},
+ }
+ for ax in Axis:
+ curr_axis = getattr(self.data, str(ax))
+ for ann in curr_axis:
+ ann_schema = {"name": ann, "writable": False}
+ ann_schema.update(series_to_schema(curr_axis[ann]))
+ self.schema["annotations"][ax]["columns"].append(ann_schema)
+
+ for layout in self.get_embedding_names():
+ layout_schema = {"name": layout, "type": "float32", "dims": [f"{layout}_0", f"{layout}_1"]}
+ self.schema["layout"]["obs"].append(layout_schema)
+
+ def get_schema(self):
+ return self.schema
+
+ def _load_data(self, data_locator):
+ # as of AnnData 0.6.19, backed mode performs initial load fast, but at the
+ # cost of significantly slower access to X data.
+ try:
+ # there is no guarantee data_locator indicates a local file. The AnnData
+ # API will only consume local file objects. If we get a non-local object,
+ # make a copy in tmp, and delete it after we load into memory.
+ with data_locator.local_handle() as lh:
+ # as of AnnData 0.6.19, backed mode performs initial load fast, but at the
+ # cost of significantly slower access to X data.
+ backed = "r" if self.config.anndata_backed else None
+ self.data = anndata.read_h5ad(lh, backed=backed)
+
+ except ValueError:
+ raise DatasetAccessError(
+ "File must be in the .h5ad format. Please read "
+ "https://github.com/theislab/scanpy_usage/blob/master/170505_seurat/info_h5ad.md to "
+ "learn more about this format. You may be able to convert your file into this format "
+ "using `cellxgene prepare`, please run `cellxgene prepare --help` for more "
+ "information."
+ )
+ except MemoryError:
+ raise DatasetAccessError("Out of memory - file is too large for available memory.")
+ except Exception as e:
+ raise DatasetAccessError(
+ f"{e} - file not found or is inaccessible. File must be an .h5ad object. "
+ f"Please check your input and try again."
+ )
+
+ def _validate_and_initialize(self):
+ if anndata_version_is_pre_070() and self.config.anndata_backed:
+ warnings.warn(f"Use of --backed mode with anndata versions older than 0.7 will have serious "
+ "performance issues. Please update to at least anndata 0.7 or later.")
+
+ # var and obs column names must be unique
+ if not self.data.obs.columns.is_unique or not self.data.var.columns.is_unique:
+ raise KeyError(f"All annotation column names must be unique.")
+
+ self._alias_annotation_names()
+ self._validate_data_types()
+ self.cell_count = self.data.shape[0]
+ self.gene_count = self.data.shape[1]
+ self._create_schema()
+
+ # heuristic
+ n_values = self.data.shape[0] * self.data.shape[1]
+ if (n_values > 1e8 and self.config.anndata_backed is True) or (n_values > 5e8):
+ self.parameters.update({"diffexp_may_be_slow": True})
+
+ def _is_valid_layout(self, arr):
+ """ return True if this layout data is a valid array for front-end presentation:
+ * ndarray, with shape (n_obs, >= 2), dtype float/int/uint
+ * contains only finite values
+ """
+ is_valid = type(arr) == np.ndarray and arr.dtype.kind in "fiu"
+ is_valid = is_valid and arr.shape[0] == self.data.n_obs and arr.shape[1] >= 2
+ is_valid = is_valid and np.all(np.isfinite(arr))
+ return is_valid
+
+ def _validate_data_types(self):
+ # The backed API does not support interrogation of the underlying sparsity or sparse matrix type
+ # Fake it by asking for a small subarray and testing it. NOTE: if the user has ignored our
+ # anndata <= 0.7 warning, opted for the --backed option, and specified a large, sparse dataset,
+ # this "small" indexing request will load the entire X array. This is due to a bug in anndata<=0.7
+ # which will load the entire X matrix to fullfill any slicing request if X is sparse. See
+ # user warning in _load_data().
+ X0 = self.data.X[0, 0:1]
+ if sparse.isspmatrix(X0) and not sparse.isspmatrix_csc(X0):
+ warnings.warn(
+ f"Anndata data matrix is sparse, but not a CSC (columnar) matrix. "
+ f"Performance may be improved by using CSC."
+ )
+ if self.data.X.dtype != "float32":
+ warnings.warn(
+ f"Anndata data matrix is in {self.data.X.dtype} format not float32. " f"Precision may be truncated."
+ )
+ for ax in Axis:
+ curr_axis = getattr(self.data, str(ax))
+ for ann in curr_axis:
+ datatype = curr_axis[ann].dtype
+ downcast_map = {
+ "int64": "int32",
+ "uint32": "int32",
+ "uint64": "int32",
+ "float64": "float32",
+ }
+ if datatype in downcast_map:
+ warnings.warn(
+ f"Anndata annotation {ax}:{ann} is in unsupported format: {datatype}. "
+ f"Data will be downcast to {downcast_map[datatype]}."
+ )
+ if isinstance(datatype, CategoricalDtype):
+ category_num = len(curr_axis[ann].dtype.categories)
+ if category_num > 500 and category_num > self.config.max_category_items:
+ warnings.warn(
+ f"{str(ax).title()} annotation '{ann}' has {category_num} categories, this may be "
+ f"cumbersome or slow to display. We recommend setting the "
+ f"--max-category-items option to 500, this will hide categorical "
+ f"annotations with more than 500 categories in the UI"
+ )
+
+ def annotation_to_fbs_matrix(self, axis, fields=None, labels=None):
+ if axis == Axis.OBS:
+ if labels is not None and not labels.empty:
+ df = self.data.obs.join(labels, self.parameters.get("obs_names"))
+ else:
+ df = self.data.obs
+ else:
+ df = self.data.var
+
+ if fields is not None and len(fields) > 0:
+ df = df[fields]
+ return encode_matrix_fbs(df, col_idx=df.columns)
+
+ def get_embedding_names(self):
+ """ function:
+ a) generate list of default layouts
+ b) validate layouts are legal. remove/warn on any that are not
+ c) cap total list of layouts at global const MAX_LAYOUTS
+ """
+ # load default layouts from the data.
+ layouts = self.config.layout
+
+ if layouts is None or len(layouts) == 0:
+ layouts = [key[2:] for key in self.data.obsm_keys() if type(key) == str and key.startswith("X_")]
+
+ # remove invalid layouts
+ valid_layouts = []
+ obsm_keys = self.data.obsm_keys()
+ for layout in layouts:
+ layout_name = f"X_{layout}"
+ if layout_name not in obsm_keys:
+ warnings.warn(f"Ignoring unknown layout name: {layout}.")
+ elif not self._is_valid_layout(self.data.obsm[layout_name]):
+ warnings.warn(f"Ignoring layout due to malformed shape or data type: {layout}")
+ else:
+ valid_layouts.append(layout)
+
+ if len(valid_layouts) == 0:
+ raise PrepareError(f"No valid layout data.")
+
+ # cap layouts to MAX_LAYOUTS
+ return layouts[0:MAX_LAYOUTS]
+
+ def get_embedding_array(self, ename, dims=2):
+ full_embedding = self.data.obsm[f"X_{ename}"]
+ return full_embedding[:, 0:dims]
+
+ def get_X_array(self, obs_mask=None, var_mask=None):
+ if obs_mask is None:
+ obs_mask = slice(None)
+ if var_mask is None:
+ var_mask = slice(None)
+ X = self.data.X[obs_mask, var_mask]
+ return X
+
+ def get_shape(self):
+ return self.data.shape
+
+ def query_var_array(self, term_name):
+ return getattr(self.data.var, term_name)
+
+ def query_obs_array(self, term_name):
+ return getattr(self.data.obs, term_name)
+
+ def get_obs_index(self):
+ name = getattr(self.config, "obs_names")
+ if name is None:
+ return self.original_obs_index
+ else:
+ return self.data.obs[name]
+
+ def get_obs_columns(self):
+ return self.data.obs.columns
diff --git a/server/data_common/__init__.py b/server/data_common/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/data_common/data_adaptor.py b/server/data_common/data_adaptor.py
new file mode 100644
index 00000000..109b1962
--- /dev/null
+++ b/server/data_common/data_adaptor.py
@@ -0,0 +1,332 @@
+from abc import ABCMeta, abstractmethod
+from server_timing import Timing as ServerTiming
+import numpy as np
+import pandas as pd
+from server.data_common.fbs.matrix import encode_matrix_fbs
+from server.common.constants import Axis, DEFAULT_TOP_N
+from server.common.errors import FilterError, JSONEncodingValueError
+from server.compute.diffexp import diffexp_ttest
+from server.common.utils import jsonify_numpy
+from server.common.app_config import AppFeature, AppConfig
+from server.common.data_locator import DataLocator
+
+
+class DataAdaptor(metaclass=ABCMeta):
+ """Base class for loading and accessing matrix data"""
+
+ def __init__(self, config):
+ # config will normally be a type that inherits from AppConfig.
+ # the following is for backwards compatability with tests
+ if config is None:
+ config = AppConfig()
+ elif type(config) == dict:
+ config = AppConfig(**config)
+
+ # config is the application configuration
+ self.config = config
+
+ # parameters set by this data adaptor based on the data.
+ self.parameters = {}
+
+ @staticmethod
+ @abstractmethod
+ def pre_load_validation(location):
+ pass
+
+ @staticmethod
+ @abstractmethod
+ def open(location, config):
+ pass
+
+ @staticmethod
+ @abstractmethod
+ def file_size(location):
+ pass
+
+ @abstractmethod
+ def get_name(self):
+ """return a string name for this data adaptor"""
+ pass
+
+ @abstractmethod
+ def get_library_versions(self):
+ """return a dictionary of library name to library versions"""
+ pass
+
+ @abstractmethod
+ def get_embedding_names(self):
+ """return a list of embedding names"""
+ pass
+
+ @abstractmethod
+ def get_embedding_array(self, ename, dims=2):
+ """return an numpy array for the given embedding name."""
+ pass
+
+ @abstractmethod
+ def get_X_array(self, obs_mask=None, var_mask=None):
+ """return the X array, possibly filtered by obs_mask or var_mask.
+ the return type is either ndarray or scipy.sparse.spmatrix."""
+ pass
+
+ @abstractmethod
+ def get_shape(self):
+ pass
+
+ @abstractmethod
+ def query_var_array(self, term_var):
+ pass
+
+ @abstractmethod
+ def query_obs_array(self, term_var):
+ pass
+
+ @abstractmethod
+ def get_obs_index(self):
+ pass
+
+ @abstractmethod
+ def get_obs_columns(self):
+ pass
+
+ @abstractmethod
+ def cleanup(self):
+ pass
+
+ @abstractmethod
+ def get_location(self):
+ pass
+
+ @abstractmethod
+ def get_schema(self):
+ """
+ Return current schema
+ """
+ pass
+
+ @abstractmethod
+ def annotation_to_fbs_matrix(self, axis, field=None, uid=None):
+ """
+ Gets annotation value for each observation
+ :param axis: string obs or var
+ :param fields: list of keys for annotation to return, returns all annotation values if not set.
+ :return: flatbuffer: in fbs/matrix.fbs encoding
+ """
+ pass
+
+ def get_features(self):
+ features = {}
+ features["cluster"] = AppFeature("/cluster/")
+
+ if self.get_embedding_names():
+ # TODO handle "var" when gene layout becomes available
+ features["layout_obs"] = AppFeature(
+ "/layout/obs", available=True)
+ else:
+ features["layout_obs"] = AppFeature("/layout/obs")
+
+ if self.config.disable_diffexp:
+ features["diffexp"] = AppFeature("/diffexp/")
+ else:
+ features["diffexp"] = AppFeature(
+ "/diffexp/", available=True)
+
+ return features
+
+ def update_parameters(self, parameters):
+ parameters.update(self.parameters)
+
+ def _index_filter_to_mask(self, filter, count):
+ mask = np.zeros((count,), dtype=np.bool)
+ for i in filter:
+ if type(i) == list:
+ mask[i[0]: i[1]] = True
+ else:
+ mask[i] = True
+ return mask
+
+ def _axis_filter_to_mask(self, axis, filter, count):
+ mask = np.ones((count, ), dtype=np.bool)
+ if 'index' in filter:
+ mask = np.logical_and(mask, self._index_filter_to_mask(filter['index'], count))
+ if 'annotation_value' in filter:
+ mask = np.logical_and(mask, self._annotation_filter_to_mask(axis, filter['annotation_value'], count))
+
+ return mask
+
+ def _annotation_filter_to_mask(self, axis, filter, count):
+ mask = np.ones((count,), dtype=np.bool)
+ for v in filter:
+ name = v["name"]
+ if axis == Axis.VAR:
+ anno_data = self.query_var_array(name)
+ elif axis == Axis.OBS:
+ anno_data = self.query_obs_array(name)
+
+ if anno_data.dtype.name in ["boolean", "category", "object"]:
+ values = v.get('values', [])
+ key_idx = np.in1d(anno_data, values)
+ mask = np.logical_and(mask, key_idx)
+
+ else:
+ min_ = v.get("min", None)
+ max_ = v.get("max", None)
+ if min_ is not None:
+ key_idx = (anno_data >= min_).ravel()
+ mask = np.logical_and(mask, key_idx)
+ if max_ is not None:
+ key_idx = (anno_data <= max_).ravel()
+ mask = np.logical_and(mask, key_idx)
+
+ return mask
+
+ def _filter_to_mask(self, filter):
+ """
+ Return the filter as a row and column selection list.
+ No filter on a dimension means 'all'
+ """
+ shape = self.get_shape()
+ var_selector = None
+ obs_selector = None
+ if filter is not None:
+ if Axis.OBS in filter:
+ obs_selector = self._axis_filter_to_mask(Axis.OBS, filter['obs'], shape[0])
+
+ if Axis.VAR in filter:
+ var_selector = self._axis_filter_to_mask(Axis.VAR, filter['var'], shape[1])
+
+ return (obs_selector, var_selector)
+
+ def check_new_labels(self, labels_df):
+ """Check the new annotations labels, then set the labels_df index"""
+ if labels_df is None or labels_df.empty:
+ return
+
+ labels_df.index = self.get_obs_index()
+
+ # all labels must have a name, which must be unique and not used in obs column names
+ if not labels_df.columns.is_unique:
+ raise KeyError(f"All column names specified in user annotations must be unique.")
+
+ # the label index must be unique, and must have same values the anndata obs index
+ if not labels_df.index.is_unique:
+ raise KeyError(f"All row index values specified in user annotations must be unique.")
+
+ obs_columns = self.get_obs_columns()
+
+ duplicate_columns = list(set(labels_df.columns) & set(obs_columns))
+ if len(duplicate_columns) > 0:
+ raise KeyError(
+ f"Labels file may not contain column names which overlap " f"with h5ad obs columns {duplicate_columns}"
+ )
+
+ # labels must have same count as obs annotations
+ shape = self.get_shape()
+ if labels_df.shape[0] != shape[0]:
+ raise ValueError("Labels file must have same number of rows as data file.")
+
+ def data_frame_to_fbs_matrix(self, filter, axis):
+ """
+ Retrieves data 'X' and returns in a flatbuffer Matrix.
+ :param filter: filter: dictionary with filter params
+ :param axis: string obs or var
+ :return: flatbuffer Matrix
+
+ Caveats:
+ * currently only supports access on VAR axis
+ * currently only supports filtering on VAR axis
+ """
+
+ if axis != Axis.VAR:
+ raise ValueError("Only VAR dimension access is supported")
+
+ try:
+ obs_selector, var_selector = self._filter_to_mask(filter)
+ except (KeyError, IndexError, TypeError, AttributeError) as e:
+ raise FilterError(f"Error parsing filter: {e}") from e
+
+ if obs_selector is not None:
+ raise FilterError("filtering on obs unsupported")
+
+ X = self.get_X_array(obs_selector, var_selector)
+ col_idx = np.nonzero([] if var_selector is None else var_selector)[0]
+ return encode_matrix_fbs(X, col_idx=col_idx, row_idx=None)
+
+ def diffexp_topN(self, obsFilterA, obsFilterB, top_n=None):
+ """
+ Computes the top N differentially expressed variables between two observation sets. If mode
+ is "TOP_N", then stats for the top N
+ dataframes
+ contain a subset of variables, then statistics for all variables will be returned, otherwise
+ only the top N vars will be returned.
+ :param obsFilterA: filter: dictionary with filter params for first set of observations
+ :param obsFilterB: filter: dictionary with filter params for second set of observations
+ :param top_n: Limit results to top N (Top var mode only)
+ :return: top N genes and corresponding stats
+ """
+ if Axis.VAR in obsFilterA or Axis.VAR in obsFilterB:
+ raise FilterError("Observation filters may not contain variable conditions")
+ try:
+ shape = self.get_shape()
+ obs_mask_A = self._axis_filter_to_mask(Axis.OBS, obsFilterA["obs"], shape[0])
+ obs_mask_B = self._axis_filter_to_mask(Axis.OBS, obsFilterB["obs"], shape[0])
+ except (KeyError, IndexError) as e:
+ raise FilterError(f"Error parsing filter: {e}") from e
+ if top_n is None:
+ top_n = DEFAULT_TOP_N
+
+ result = diffexp_ttest(self, obs_mask_A, obs_mask_B, top_n, self.config.diffexp_lfc_cutoff)
+
+ try:
+ return jsonify_numpy(result)
+ except ValueError:
+ raise JSONEncodingValueError("Error encoding differential expression to JSON")
+
+ def layout_to_fbs_matrix(self):
+ """ same as layout, except returns a flatbuffer """
+ """
+ return all embeddings as a flatbuffer, using the cellxgene matrix fbs encoding.
+
+ * returns only first two dimensions, with name {ename}_0 and {ename}_1,
+ where {ename} is the embedding name.
+ * client assumes each will be individually centered & scaled (isotropically)
+ to a [0, 1] range.
+ * does not support filtering
+
+ """
+
+ embeddings = self.get_embedding_names()
+ layout_data = []
+ with ServerTiming.time(f'layout.query'):
+ for ename in embeddings:
+ embedding = self.get_embedding_array(ename, 2)
+
+ # scale isotropically
+ min = embedding.min(axis=0)
+ max = embedding.max(axis=0)
+ scale = np.amax(max - min)
+ normalized_layout = (embedding - min) / scale
+
+ # translate to center on both axis
+ translate = 0.5 - ((max - min) / scale / 2)
+ normalized_layout = normalized_layout + translate
+
+ normalized_layout = normalized_layout.astype(dtype=np.float32)
+ layout_data.append(pd.DataFrame(normalized_layout, columns=[f"{ename}_0", f"{ename}_1"]))
+
+ with ServerTiming.time(f'layout.encode'):
+ if layout_data:
+ df = pd.concat(layout_data, axis=1, copy=False)
+ else:
+ df = pd.DataFrame()
+ fbs = encode_matrix_fbs(df, col_idx=df.columns, row_idx=None)
+
+ return fbs
+
+ def get_last_mod_time(self):
+ try:
+ data_locator = DataLocator(self.get_location())
+ lastmod = data_locator.lastmodtime()
+ except RuntimeError:
+ lastmod = None
+ return lastmod
diff --git a/server/app/util/fbs/NetEncoding/Column.py b/server/data_common/fbs/NetEncoding/Column.py
similarity index 100%
rename from server/app/util/fbs/NetEncoding/Column.py
rename to server/data_common/fbs/NetEncoding/Column.py
diff --git a/server/app/util/fbs/NetEncoding/Float32Array.py b/server/data_common/fbs/NetEncoding/Float32Array.py
similarity index 100%
rename from server/app/util/fbs/NetEncoding/Float32Array.py
rename to server/data_common/fbs/NetEncoding/Float32Array.py
diff --git a/server/app/util/fbs/NetEncoding/Float64Array.py b/server/data_common/fbs/NetEncoding/Float64Array.py
similarity index 100%
rename from server/app/util/fbs/NetEncoding/Float64Array.py
rename to server/data_common/fbs/NetEncoding/Float64Array.py
diff --git a/server/app/util/fbs/NetEncoding/Int32Array.py b/server/data_common/fbs/NetEncoding/Int32Array.py
similarity index 100%
rename from server/app/util/fbs/NetEncoding/Int32Array.py
rename to server/data_common/fbs/NetEncoding/Int32Array.py
diff --git a/server/app/util/fbs/NetEncoding/JSONEncodedArray.py b/server/data_common/fbs/NetEncoding/JSONEncodedArray.py
similarity index 100%
rename from server/app/util/fbs/NetEncoding/JSONEncodedArray.py
rename to server/data_common/fbs/NetEncoding/JSONEncodedArray.py
diff --git a/server/app/util/fbs/NetEncoding/Matrix.py b/server/data_common/fbs/NetEncoding/Matrix.py
similarity index 100%
rename from server/app/util/fbs/NetEncoding/Matrix.py
rename to server/data_common/fbs/NetEncoding/Matrix.py
diff --git a/server/app/util/fbs/NetEncoding/TypedArray.py b/server/data_common/fbs/NetEncoding/TypedArray.py
similarity index 100%
rename from server/app/util/fbs/NetEncoding/TypedArray.py
rename to server/data_common/fbs/NetEncoding/TypedArray.py
diff --git a/server/app/util/fbs/NetEncoding/Uint32Array.py b/server/data_common/fbs/NetEncoding/Uint32Array.py
similarity index 100%
rename from server/app/util/fbs/NetEncoding/Uint32Array.py
rename to server/data_common/fbs/NetEncoding/Uint32Array.py
diff --git a/server/data_common/fbs/NetEncoding/__init__.py b/server/data_common/fbs/NetEncoding/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/data_common/fbs/__init__.py b/server/data_common/fbs/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/app/util/fbs/matrix.py b/server/data_common/fbs/matrix.py
similarity index 94%
rename from server/app/util/fbs/matrix.py
rename to server/data_common/fbs/matrix.py
index 250f31b4..e10bc56a 100644
--- a/server/app/util/fbs/matrix.py
+++ b/server/data_common/fbs/matrix.py
@@ -4,14 +4,14 @@ from scipy import sparse
import pandas as pd
import json
-import server.app.util.fbs.NetEncoding.Column as Column
-import server.app.util.fbs.NetEncoding.TypedArray as TypedArray
-import server.app.util.fbs.NetEncoding.Matrix as Matrix
-import server.app.util.fbs.NetEncoding.Int32Array as Int32Array
-import server.app.util.fbs.NetEncoding.Uint32Array as Uint32Array
-import server.app.util.fbs.NetEncoding.Float32Array as Float32Array
-import server.app.util.fbs.NetEncoding.Float64Array as Float64Array
-import server.app.util.fbs.NetEncoding.JSONEncodedArray as JSONEncodedArray
+import server.data_common.fbs.NetEncoding.Column as Column
+import server.data_common.fbs.NetEncoding.TypedArray as TypedArray
+import server.data_common.fbs.NetEncoding.Matrix as Matrix
+import server.data_common.fbs.NetEncoding.Int32Array as Int32Array
+import server.data_common.fbs.NetEncoding.Uint32Array as Uint32Array
+import server.data_common.fbs.NetEncoding.Float32Array as Float32Array
+import server.data_common.fbs.NetEncoding.Float64Array as Float64Array
+import server.data_common.fbs.NetEncoding.JSONEncodedArray as JSONEncodedArray
# Placeholder until recent enhancements to flatbuffers Python
diff --git a/server/data_common/matrix_loader.py b/server/data_common/matrix_loader.py
new file mode 100644
index 00000000..4f7b8e9f
--- /dev/null
+++ b/server/data_common/matrix_loader.py
@@ -0,0 +1,171 @@
+from enum import Enum
+import threading
+import time
+from server.data_common.rwlock import RWLock
+from server.common.errors import DatasetAccessError
+from contextlib import contextmanager
+
+
+class MatrixDataCacheItem(object):
+ """This class provides access and caching for a dataset. The first time a dataset is accessed, it is
+ opened and cached. Later accesses use the cached version. It may also be deleted by the
+ MatrixDataCacheManager to make room for another dataset. While a dataset is actively being used
+ (during the lifetime of a api request), a reader lock is locked. During that time, the dataset cannot
+ be removed."""
+
+ def __init__(self, loader):
+ self.loader = loader
+ self.data_adaptor = None
+ self.data_lock = RWLock()
+
+ def acquire(self, app_config):
+ """returns the data_adaptor if cached. opens the data_adaptor if not.
+ In either case, the a reader lock is taken. Must call release when
+ the data_adaptor is no longer needed"""
+
+ self.data_lock.r_acquire()
+ if self.data_adaptor:
+ return self.data_adaptor
+
+ self.data_lock.r_release()
+ try:
+ with self.data_lock.w_locked():
+ # the data may have been loaded while waiting on the lock
+ if not self.data_adaptor:
+ self.loader.pre_load_validation()
+ self.data_adaptor = self.loader.open(app_config)
+
+ except Exception:
+ # necessary to acquire after an exception, since the release will occur when
+ # the context exits
+ self.data_lock.r_acquire()
+ raise
+
+ self.data_lock.r_acquire()
+ if self.data_adaptor:
+ return self.data_adaptor
+
+ def release(self):
+ """Release the reader lock"""
+ self.data_lock.r_release()
+
+ def delete(self):
+ """Clear resources used by this dataset"""
+ with self.data_lock.w_locked():
+ if self.data_adaptor:
+ self.data_adaptor.cleanup()
+ self.data_adaptor = None
+
+
+class MatrixDataCacheManager(object):
+ """A class to manage the cached datasets. This is intended to be used as a context manager
+ for handling api requests. When the context is created, the data_adator is either loaded or
+ retrieved from a cache. In either case, the reader lock is taken during this time, and release
+ when the context ends. This class currently implements a simple least recently used cache,
+ which can delete a dataset from the cache to make room for a new oneo
+
+ This is the indended usage pattern:
+
+ m = MatrixDataCacheManager()
+ with m.data_adaptor(location, app_config) as data_adaptor:
+ # use the data_adaptor for some operation
+ """
+
+ # The number of datasets to cache. When MAX_CACHED is reached, the least recently used
+ # cache is replaced with the newly requested one.
+ # TODO: This is very simple. This can be improved by taking into account how much space is actually
+ # taken by each dataset, instead of arbitrarily picking a max datasets to cache.
+ # Also, this should be controlled by a configuration parameter.
+ MAX_CACHED = 3
+
+ # FIXME: If the number of active datasets exceeds the MAX_CACHED, then each request could
+ # lead to a dataset being deleted and a new only being opened: the cache will get thrashed.
+ # In this case, we may need to send back a 503 (Server Unavailable), or some other error message.
+
+ # FIXME: If the actual dataset is changed. E.g. a new set of datafiles replaces an existing set,
+ # then the cache will not react to this. Ideally this would invalidate the cache. One solution is
+ # to keep a small metadata file associated with each dataset, which contains versioning information.
+ # When the dataset is accessed, the current version can be compared with the cached version, and if
+ # there is a mismatch, then the cache can be refreshed.
+
+ def __init__(self):
+ # key is location, value is tuple of (MatrixDataCacheItem, last_accessed)
+ self.datasets = {}
+ self.lock = threading.Lock()
+
+ @contextmanager
+ def data_adaptor(self, location, app_config):
+ # create a loader for to this location if it does not already exist
+ with self.lock:
+ value = self.datasets.get(location)
+ if value is not None:
+ cache_item = value[0]
+ last_accessed = time.time()
+ self.datasets[location] = (cache_item, last_accessed)
+ else:
+ while True:
+ # find the last access times for each loader
+ items = list(self.datasets.items())
+ sorted(items, key=lambda x: x[1][1])
+ if len(items) < self.MAX_CACHED:
+ break
+
+ # close the least recently used loader
+ oldest = items[0]
+ oldest_cache = oldest[1][0]
+ oldest_key = oldest[0]
+ oldest_cache.delete()
+ del self.datasets[oldest_key]
+
+ last_accessed = time.time()
+ loader = MatrixDataLoader(location)
+ cache_item = MatrixDataCacheItem(loader)
+ self.datasets[location] = (cache_item, last_accessed)
+ try:
+ data_adaptor = cache_item.acquire(app_config)
+ yield data_adaptor
+ finally:
+ cache_item.release()
+
+
+class MatrixDataType(Enum):
+ H5AD = "h5ad"
+ CXG = "cxg"
+ UNKNOWN = "unknown"
+
+
+class MatrixDataLoader(object):
+
+ def __init__(self, location, etype=None):
+ self.location = location
+ if etype is None:
+ self.etype = self.matrix_data_type()
+ else:
+ self.etype = etype
+ self.matrix_type = None
+ if self.etype == MatrixDataType.H5AD:
+ from server.data_anndata.anndata_adaptor import AnndataAdaptor
+ self.matrix_type = AnndataAdaptor
+ elif self.etype == MatrixDataType.CXG:
+ from server.data_cxg.cxg_adaptor import CxgAdaptor
+ self.matrix_type = CxgAdaptor
+
+ def matrix_data_type(self):
+ if self.location.endswith(".h5ad"):
+ return MatrixDataType.H5AD
+ elif ".cxg" in self.location:
+ return MatrixDataType.CXG
+ else:
+ return MatrixDataType.UNKNOWN
+
+ def pre_load_validation(self):
+ if self.etype == MatrixDataType.UNKNOWN:
+ raise DatasetAccessError(f"{self.location} does not have a recognized type: .h5ad or .cxg")
+ self.matrix_type.pre_load_validation(self.location)
+
+ def file_size(self):
+ return self.matrix_type.file_size(self.location)
+
+ def open(self, app_config):
+ # create and return a DataAdaptor object
+ return self.matrix_type.open(self.location, app_config)
diff --git a/server/data_common/rwlock.py b/server/data_common/rwlock.py
new file mode 100644
index 00000000..61760bb9
--- /dev/null
+++ b/server/data_common/rwlock.py
@@ -0,0 +1,97 @@
+# -*- coding: utf-8 -*-
+""" rwlock.py
+
+ A class to implement read-write locks on top of the standard threading
+ library.
+
+ This is implemented with two mutexes (threading.Lock instances) as per this
+ wikipedia pseudocode:
+
+ https://en.wikipedia.org/wiki/Readers%E2%80%93writer_lock#Using_two_mutexes
+
+ Code written by Tyler Neylon at Unbox Research.
+
+ This file is public domain.
+"""
+
+
+# _______________________________________________________________________
+# Imports
+
+from contextlib import contextmanager
+from threading import Lock
+
+
+# _______________________________________________________________________
+# Class
+
+class RWLock(object):
+ """ RWLock class; this is meant to allow an object to be read from by
+ multiple threads, but only written to by a single thread at a time. See:
+ https://en.wikipedia.org/wiki/Readers%E2%80%93writer_lock
+
+ Usage:
+
+ from rwlock import RWLock
+
+ my_obj_rwlock = RWLock()
+
+ # When reading from my_obj:
+ with my_obj_rwlock.r_locked():
+ do_read_only_things_with(my_obj)
+
+ # When writing to my_obj:
+ with my_obj_rwlock.w_locked():
+ mutate(my_obj)
+ """
+
+ def __init__(self):
+
+ self.w_lock = Lock()
+ self.num_r_lock = Lock()
+ self.num_r = 0
+
+ # ___________________________________________________________________
+ # Reading methods.
+
+ def r_acquire(self):
+ self.num_r_lock.acquire()
+ self.num_r += 1
+ if self.num_r == 1:
+ self.w_lock.acquire()
+ self.num_r_lock.release()
+
+ def r_release(self):
+ assert self.num_r > 0
+ self.num_r_lock.acquire()
+ self.num_r -= 1
+ if self.num_r == 0:
+ self.w_lock.release()
+ self.num_r_lock.release()
+
+ @contextmanager
+ def r_locked(self):
+ """ This method is designed to be used via the `with` statement. """
+ try:
+ self.r_acquire()
+ yield
+ finally:
+ self.r_release()
+
+ # ___________________________________________________________________
+ # Writing methods.
+
+ def w_acquire(self):
+ self.w_lock.acquire()
+
+ def w_release(self):
+ self.w_lock.release()
+
+ @contextmanager
+ def w_locked(self):
+ """ This method is designed to be used via the `with` statement. """
+ try:
+ self.w_acquire()
+ yield
+ finally:
+ self.w_release()
diff --git a/server/data_cxg/__init__.py b/server/data_cxg/__init__.py
new file mode 100644
index 00000000..e69de29b
diff --git a/server/data_cxg/cxg_adaptor.py b/server/data_cxg/cxg_adaptor.py
new file mode 100644
index 00000000..c51aceec
--- /dev/null
+++ b/server/data_cxg/cxg_adaptor.py
@@ -0,0 +1,321 @@
+import os
+import json
+from server.common.utils import dtype_to_schema
+from server.common.errors import DatasetAccessError
+from server.common.utils import path_join
+from server.common.constants import Axis
+from server.data_common.data_adaptor import DataAdaptor
+from server.data_common.fbs.matrix import encode_matrix_fbs
+import tiledb
+import numpy as np
+import pandas as pd
+from server_timing import Timing as ServerTiming
+import threading
+
+
+class CxgAdaptor(DataAdaptor):
+
+ # TODO: The tiledb context parameters should be a configuration option
+ tiledb_ctx = tiledb.Ctx({
+ 'sm.tile_cache_size': 8 * 1024 * 1024 * 1024,
+ 'sm.num_reader_threads': 32,
+ })
+
+ def __init__(self, location, config=None):
+ super().__init__(config)
+ self.url = location
+ self.arrays = {}
+ self.lock = threading.Lock()
+
+ self.url = location
+ if self.url[-1] != '/':
+ self.url += '/'
+
+ self._validate_and_initialize()
+
+ def cleanup(self):
+ """close all the open tiledb arrays"""
+ for array in self.arrays.values():
+ array.close()
+ self.arrays.clear()
+
+ @staticmethod
+ def pre_load_validation(location):
+ if not CxgAdaptor.isvalid(location):
+ raise DatasetAccessError(f"cxg matrix is not valid: {location}")
+
+ @staticmethod
+ def file_size(location):
+ return 0
+
+ @staticmethod
+ def open(location, args):
+ return CxgAdaptor(location, args)
+
+ def get_location(self):
+ return self.url
+
+ def get_name(self):
+ return "cellxgene cxcxgg adaptor version"
+
+ def get_library_versions(self):
+ return dict(tiledb=tiledb.__version__)
+
+ def get_path(self, *urls):
+ return path_join(self.url, *urls)
+
+ def lsuri(self, uri):
+ """
+ given a URI, do a tiledb.ls but normalizing for all path weirdness:
+ * S3 URIs require trailing slash. file: doesn't care.
+ * results on S3 *have* a trailing slash, Posix does not.
+
+ returns list of (absolute paths, type) *without* trailing slash
+ in the path.
+ """
+ def _cleanpath(p):
+ if p[-1] == '/':
+ return p[:-1]
+ else:
+ return p
+
+ if uri[-1] != '/':
+ uri += '/'
+
+ result = []
+ tiledb.ls(uri,
+ lambda path, type: result.append((_cleanpath(path), type)),
+ ctx=self.tiledb_ctx)
+ return result
+
+ @staticmethod
+ def isvalid(url):
+ """
+ Return True if this looks like a valid CXG, False if not. Just a quick/cheap
+ test, not to be fully trusted.
+ """
+ if not tiledb.object_type(url) == "group":
+ return False
+ if not tiledb.object_type(path_join(url, "obs")) == "array":
+ return False
+ if not tiledb.object_type(path_join(url, "var")) == "array":
+ return False
+ if not tiledb.object_type(path_join(url, "X")) == "array":
+ return False
+ if not tiledb.object_type(path_join(url, "emb")) == "group":
+ return False
+ return True
+
+ def _validate_and_initialize(self):
+ if not self.isvalid(self.url):
+ raise DatasetAccessError(f"invalid cxg dataset {self.url}")
+
+ def open_array(self, name):
+ try:
+ with self.lock:
+ array = self.arrays.get(name)
+ if array:
+ return array
+ p = self.get_path(name)
+ try:
+ array = tiledb.DenseArray(p, mode="r", ctx=self.tiledb_ctx)
+ except tiledb.libtiledb.TileDBError as e:
+ raise AttributeError(str(e))
+ self.arrays[name] = array
+ return array
+ except tiledb.libtiledb.TileDBError as e:
+ raise AttributeError(str(e))
+
+ def get_embedding_array(self, ename, dims=2):
+ array = self.open_array(f"emb/{ename}")
+ return array[:, 0:dims]
+
+ def get_X_array(self, obs_mask=None, var_mask=None):
+ obs_items = self._convert_mask(obs_mask)
+ var_items = self._convert_mask(var_mask)
+ X = self.open_array("X")
+ if obs_items == slice(None) and var_items == slice(None):
+ data = X[:, :]
+ else:
+ data = X.multi_index[obs_items, var_items]['']
+ return data
+
+ def get_shape(self):
+ X = self.open_array("X")
+ return X.shape
+
+ def get_X_array_dtype(self):
+ X = self.open_array("X")
+ return X.dtype
+
+ def query_var_array(self, term_name):
+ var = self.open_array("var")
+ data = var.query(attrs=[term_name])[:][term_name]
+ return data
+
+ def query_obs_array(self, term_name):
+ var = self.open_array("obs")
+ try:
+ data = var.query(attrs=[term_name])[:][term_name]
+ except tiledb.libtiledb.TileDBError as e:
+ raise AttributeError(str(e))
+ return data
+
+ def get_obs_names(self):
+ # get the index from the meta data
+ obs = self.open_array("obs")
+ meta = json.loads(obs.meta["cxg_schema"])
+ index_name = meta["index"]
+ return index_name
+
+ def get_obs_index(self):
+ obs = self.open_array("obs")
+ meta = json.loads(obs.meta["cxg_schema"])
+ index_name = meta["index"]
+ data = obs.query(attrs=[index_name])[:][index_name]
+ return data
+
+ def get_obs_columns(self):
+ obs = self.open_array("obs")
+ schema = obs.schema
+ col_names = [attr.name for attr in schema]
+ return pd.Index(col_names)
+
+ # function to get the embedding
+ # this function to iterate through embeddings.
+ def get_embedding_names(self):
+ with ServerTiming.time(f'layout.lsuri'):
+ pemb = self.get_path("emb")
+ embeddings = [
+ os.path.basename(p) for (p, t) in self.lsuri(pemb) if t == 'array'
+ ]
+ return embeddings
+
+ @staticmethod
+ def _get_col_type(attr, schema_hints={}):
+ type_hint = schema_hints.get(attr.name, {})
+ dtype = attr.dtype
+ schema = {}
+ # type hints take precedence
+ if 'type' in type_hint:
+ schema['type'] = type_hint['type']
+ elif dtype == np.float32:
+ schema['type'] = 'float32'
+ elif dtype == np.int32:
+ schema['type'] = 'int32'
+ elif dtype == np.bool_:
+ schema['type'] = 'boolean'
+ elif dtype == np.str:
+ schema['type'] = 'string'
+ elif dtype == "category":
+ schema["type"] = "categorical"
+ schema["categories"] = dtype.categories.tolist()
+ else:
+ raise TypeError(
+ f"Annotations of type {dtype} are unsupported."
+ )
+
+ if schema['type'] == 'categorical' and 'categories' in schema_hints:
+ schema['categories'] = schema_hints['categories']
+ return schema
+
+ def get_schema(self):
+ shape = self.get_shape()
+ dtype = self.get_X_array_dtype()
+
+ dataframe = {
+ 'nObs': shape[0],
+ 'nVar': shape[1],
+ 'type': dtype.name
+ }
+
+ annotations = {}
+ for ax in ('obs', 'var'):
+ A = self.open_array(ax)
+ schema_hints = json.loads(A.meta['cxg_schema']) if 'cxg_schema' in A.meta else {}
+ if type(schema_hints) is not dict:
+ raise TypeError(f'Array schema was malformed.')
+
+ cols = []
+ for attr in A.schema:
+ schema = dict(name=attr.name, writable=False)
+ type_hint = schema_hints.get(attr.name, {})
+ # type hints take precedence
+ if 'type' in type_hint:
+ schema['type'] = type_hint['type']
+ if schema['type'] == 'categorical' and 'categories' in type_hint:
+ schema['categories'] = type_hint['categories']
+ else:
+ schema.update(dtype_to_schema(attr.dtype))
+ cols.append(schema)
+
+ annotations[ax] = dict(columns=cols)
+
+ if 'index' in schema_hints:
+ annotations[ax].update({'index': schema_hints['index']})
+
+ obs_layout = []
+ embeddings = self.get_embedding_names()
+ for ename in embeddings:
+ A = self.open_array(f"emb/{ename}")
+ obs_layout.append({
+ 'name': ename,
+ 'type': A.dtype.name,
+ 'dims': [f'{ename}_{d}' for d in range(0, A.ndim)]
+ })
+
+ schema = {
+ 'dataframe': dataframe,
+ 'annotations': annotations,
+ 'layout': {'obs': obs_layout}
+ }
+ return schema
+
+ def annotation_to_fbs_matrix(self, axis, fields=None, labels=None):
+ with ServerTiming.time(f'annotations.{axis}.query'):
+ A = self.open_array(str(axis))
+ if fields is not None and len(fields) > 0:
+ try:
+ df = pd.DataFrame(A.query(attrs=fields)[:])
+ except tiledb.libtiledb.TileDBError:
+ raise KeyError("bad field {fields}")
+
+ else:
+ df = pd.DataFrame.from_dict(A[:])
+
+ if axis == Axis.OBS:
+ if labels is not None and not labels.empty:
+ obs_names = self.get_obs_names()
+ df = df.join(labels, obs_names)
+
+ with ServerTiming.time(f'annotations.{axis}.encode'):
+ fbs = encode_matrix_fbs(df, col_idx=df.columns)
+
+ return fbs
+
+ @staticmethod
+ def _convert_mask(boolarray):
+ """Convert an index mask to a list of ranges or indices that can be used in a multi_index."""
+ if boolarray is None:
+ return slice(None)
+ assert type(boolarray) == np.ndarray
+ assert(boolarray.dtype) == bool
+
+ selector = np.nonzero(boolarray)[0]
+
+ if len(selector) == 0:
+ return slice(None)
+
+ result = []
+ current = slice(selector[0], selector[0])
+ for sel in selector[1:]:
+ if sel == current.stop + 1:
+ current = slice(current.start, sel)
+ else:
+ result.append(current if current.start != current.stop else current.start)
+ current = slice(sel, sel)
+
+ if len(result) == 0 or result[-1] != current:
+ result.append(current if current.start != current.stop else current.start)
+
+ return result
diff --git a/server/gui/browser.py b/server/gui/browser.py
deleted file mode 100644
index aa2224e4..00000000
--- a/server/gui/browser.py
+++ /dev/null
@@ -1,92 +0,0 @@
-# flake8: noqa F403, F405
-from cefpython3 import cefpython as cef
-from PySide2.QtCore import *
-from PySide2.QtGui import *
-from PySide2.QtWidgets import *
-
-from server.gui.utils import WINDOWS, LINUX
-
-WindowUtils = cef.WindowUtils()
-
-# OS differences
-# noinspection PyUnresolvedReferences
-CefWidgetParent = QWidget
-
-
-class CefWidget(CefWidgetParent):
- def __init__(self, parent=None):
- super(CefWidget, self).__init__(parent)
- self.parent = parent
- self.browser = None
- # TODO test without this on linux
- self.hidden_window = None # Required for PyQt5 on Linux
- self.show()
-
- def focusInEvent(self, event):
- # This event seems to never get called on Linux, as CEF is
- # stealing all focus due to Issue #284.
- if self.browser:
- if WINDOWS:
- WindowUtils.OnSetFocus(self.getHandle(), 0, 0, 0)
- self.browser.SetFocus(True)
-
- def focusOutEvent(self, event):
- # This event seems to never get called on Linux, as CEF is
- # stealing all focus due to Issue #284.
- if self.browser:
- self.browser.SetFocus(False)
-
- def embedBrowser(self):
- if LINUX:
- self.hidden_window = QWindow()
- window_info = cef.WindowInfo()
- rect = [0, 0, self.width(), self.height()]
- window_info.SetAsChild(self.getHandle(), rect)
- # TODO better splash
- self.browser = cef.CreateBrowserSync(window_info)
-
- def getHandle(self):
- if self.hidden_window:
- # PyQt5 on Linux
- return int(self.hidden_window.winId())
- else:
- return int(self.winId())
-
- def moveEvent(self, _):
- self.x = 0
- self.y = 0
- if self.browser:
- if WINDOWS:
- WindowUtils.OnSize(self.getHandle(), 0, 0, 0)
- elif LINUX:
- self.browser.SetBounds(self.x, self.y, self.width(), self.height())
- self.browser.NotifyMoveOrResizeStarted()
-
- def resizeEvent(self, event):
- size = event.size()
- if self.browser:
- if WINDOWS:
- WindowUtils.OnSize(self.getHandle(), 0, 0, 0)
- elif LINUX:
- self.browser.SetBounds(self.x, self.y, size.width(), size.height())
- self.browser.NotifyMoveOrResizeStarted()
-
-
-class CefApplication(QApplication):
- def __init__(self, args):
- super(CefApplication, self).__init__(args)
- if not cef.GetAppSetting("external_message_pump"):
- self.timer = self.createTimer()
-
- def createTimer(self):
- timer = QTimer()
- timer.timeout.connect(self.onTimer)
- timer.start(10)
- return timer
-
- def onTimer(self):
- cef.MessageLoopWork()
-
- def stopTimer(self):
- # Stop the timer after Qt's message loop has ended
- self.timer.stop()
diff --git a/server/gui/cellxgene.qrc b/server/gui/cellxgene.qrc
deleted file mode 100644
index b7654cd7..00000000
--- a/server/gui/cellxgene.qrc
+++ /dev/null
@@ -1,8 +0,0 @@
-
-
- images/cellxgene_logo.png
- images/properties_contract.svg
- images/properties_expand.svg
- images/properties_expand.svg
-
-
diff --git a/server/gui/cellxgene.spec b/server/gui/cellxgene.spec
deleted file mode 100644
index 0dc65d3b..00000000
--- a/server/gui/cellxgene.spec
+++ /dev/null
@@ -1,29 +0,0 @@
-# -*- mode: python -*-
-
-block_cipher = None
-
-
-a = Analysis(['main.py'],
- pathex=['/Users/charlotteweaver/Documents/Git/cellxgene/server/gui'],
- hookspath=["/Users/charlotteweaver/Documents/Git/cellxgene/server/gui/"],
- win_no_prefer_redirects=False,
- win_private_assemblies=False,
- cipher=block_cipher,
- noarchive=False)
-
-pyz = PYZ(a.pure, a.zipped_data,
- cipher=block_cipher)
-
-exe = EXE(pyz,
- a.scripts,
- a.binaries,
- a.zipfiles,
- a.datas,
- [],
- name='cellxgene',
- debug=False,
- bootloader_ignore_signals=False,
- strip=False,
- upx=True,
- runtime_tmpdir=None,
- console=True )
diff --git a/server/gui/cellxgene_rc.py b/server/gui/cellxgene_rc.py
deleted file mode 100644
index 07fda04b..00000000
--- a/server/gui/cellxgene_rc.py
+++ /dev/null
@@ -1,440 +0,0 @@
-# -*- coding: utf-8 -*-
-
-# Resource object code
-#
-# Created: Wed Jun 19 15:02:04 2019
-# by: The Resource Compiler for PySide2 (Qt v5.12.3)
-#
-# WARNING! All changes made in this file will be lost!
-
-from PySide2 import QtCore
-
-qt_resource_data = b"\
-\x00\x00\x0d\xde\
-\x89\
-PNG\x0d\x0a\x1a\x0a\x00\x00\x00\x0dIHDR\x00\
-\x00\x01;\x00\x00\x00j\x08\x06\x00\x00\x00\xc3y\xf6!\
-\x00\x00\x00\x09pHYs\x00\x00\x0b\x13\x00\x00\x0b\x13\
-\x01\x00\x9a\x9c\x18\x00\x00\x00\x01sRGB\x00\xae\xce\
-\x1c\xe9\x00\x00\x00\x04gAMA\x00\x00\xb1\x8f\x0b\xfc\
-a\x05\x00\x00\x0dsIDATx\x01\xed\xdd\x7fn\
-\x1b\xc7\x15\x07\xf07C\xdbQ\x93\x00f\xfe\x88e7\
-\x05L\x9d J/PJ9@\xe4^@\x12\x90H\
-(\x0aT\xce\x09,\x9d\xc0\xf2\x1fE \xb9\x80\xd9\x0b\
-\xd4\xf2\x01,2'\xb0|\x023@\x0bG\xea\x1fQ\
-\xff\x09\x04\xd1;\xaf\xef-\xc9X\xa48\xb3?\xb8\xfc\
-e~?\x80b\x85\xbb\xab%g\xdf\xbe\x9d\x9d\x99\x1d\
-\x12\x01\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\
-\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\
-\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\
-\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\
-\x00\x00\xc0\x9c24\x22\xa7\xab\x87L3l\xf1xk\
-de\x03\x00\xe3g\x09\x00`\x0e \xd9\x01\xc0\x5c@\
-\xb2\x03\x80\xb9\x80d\x07\x00s\x01\xc9\x0e\x00\xe6\xc2\x0d\
-\x02\x18\xe0m\xf5i\xd5Z\xae\xfb\x96KW{\xed\xee\
-\xf1\xd6f\xef6?T\xac-\xbd!\xbf\xa6\xf4r/\
-\x11\xc0\x04\xccD\xb2\xbb\xf3\xf2;\x1a\x85\xb3\xaf\x9f\x12\
-\x00\xcc\x07\xdc\xc6\x02\xc0\x5c@\xb2\x03\x80\xb9\x80d\x07\
-\x00s\x01\xc9\x0e\x00\xe6\x02\x92\x1d\x00\xcc\x05$;\x00\
-\x98\x0b31\xf4\x04CD\x00`X\xa8\xd9\x01\xc0\x5c\
-\xc0\x13\x14c\xf0\xa6\xfa\xac\xfcI\xa9\xb5\xee\x98\x96\xad\
-\xa1\x0a3W\x88\xcc\xb91t.\xbf\xcb\xbf\xa6\x11E\
-\xd1\x8b{\x8d\xbf4i\x08\xfa\xd4\x831\xbc&?\xf7\
-\xe5\xefW\x88\xb8,\x7f\xbb\xa9\xcb\x9c\xa3\xd7\xcc\xe6\xe8\
-^\xe3\xbb\x06\xc1o\xbaef-}\xa9\xc7E\xcbk\
-\xd019\xad\x1e\xac9k\xca\x83\xfeF\xc9\xb5N\xee\
-4\xfez\x92f\x7f\x1a\x0b\xb7JQ\xb5\xe4\x5c\xb5\xff\
-8\xe9~\x1d\xd1kr\xb6\x91\xf68\xbd]=\xdc\xf0\
-.tQ\xe3jL\xf5\xc6\x07-wc0b\xfeQ\
-\x02\xa46L\xfc%\xc5^dm\xe3\x8b\x97\xdf\x1e\xd1\
-\x04a\xf2N\x8f\x22&\xef\x8c\x1f\x9f2v\x97\x8cY\
-O\xb3\xbe\x04\x9e\x9e`\x9bY\x82NO\x9eO\xed\xbb\
-\x1d&~(\x05^N\xb1I\xd3\x92\xa9}~\xfc\xdd\
-^h\xa5\x0f\xfdq\xb1\xff|}\xb0v\x83\xe9q\xfb\
-\xc4\xf4\x93\xb2\xda\xd5\xb2\x92x~Eq\x82\xb8\xce\x11\
-\xef\xdd;\xde\xde\x0d\xfd\x9d<\xc7\x89$v\x16_~\
-\xfb\xcf\xd0J\xa1\xf3L\xf6\xf5\xfd\xdd\xe3\xed\xfd\xf6\xb1\
-t\xcf\x92?+\xef\x7f~\xbc\xfd=e\xf0\xdf\xd5\xa7\
-\x8f\x8a\x8e\xbdQ\xc1m\xec\x88\x9c}}\xb8S\xb2\xa5\
-Wi\x13\x9db\xa6\xaa&\x0b\x0d\xa04\xebkr\xf9\
-\xd8^\xbe\x92\x93m7e\xb0\xa9\x8a\xae/'\xc9\x1b\
-\xdd\x9e\xa6\x88&\x04\xca)\xcb\xb6g\xabO\x1f\xdf`\
-\xf3<\xe9\xe4WZVg\xab\x87\xcf\x0cS\xe2\xba>\
-\x9al>\xb1\xad7Y\x8f\x13\xb1\xab\xe9\xbe\xf3\x96\x8b\
-\xeeK\xe3\xb0}\xd12\x95\xa4\xf5\x1d\x99\x87\xb2~\x9d\
-R\xd0\xd8\xd1\x0b\xc0,\xc5\xdeDncC\xcf\xba\x0e\
-\xea\x8c\x98\xb5gcO\xf5j\xc7\xbcK9i@H\
-\xc2\xa3\xd0\x15\xf0\xac\xfa\xf7e\xb2\xa5z\x86@\xeb\xa7\
-\xb5\xb0\xba\x04\xdd\xca\xb0\xb7\xcfE\xd0\x93\x92\xb9\xb5{\
-V=\x5c\xb9\xd3\xd8JuK\xd8\xd5\xaeQ^\xd6%\
-14\xee\xf4\xd56\xfb\x9dvj\x22\x94\x81\x94\xf1F\
-\xde{\xa0\xd3\xea?\xd6\xc9\xbaZ\xde\xdb\x1c\xdd\xf7\xc7\
-\xb6\xb5,\x09oe\xa9\xb1y\x9ee[\xa9E\xads\
-\xc6$\xad\x17\xdc\x9fW\x0f\x1ej\x8d\xd0\xb7N\xbb\xbc\
-\xad&\xc5\x0a\xe53\x91\xd8C\xcd\xae`oW\x7f\xd8\
-\x90\x90\xd9\xa5!i\xc2\x93`\xa8\x0e\xdc\x87\x04\x1b\xdb\
-\x1b\xcf\x87Ht]\x1at\xcfi\xc24!\xc8I\xa6\
-'W\x99-\xd5%\xe1-\xa7\xdd\xf6\xfd\x89g*\x9a\
-\x18B\xb5\xe2\xa2\x8eM\xfa\xf7\xf6\xb4\xaa\x89\x8e\x86\xb7\
-\xfc\xa9\xbdLU\xdb\xefS\xa1\x1c\x0c\x19\x7f\x19^)\
-o\x1a\x8e\xc6\xde\xf10\xb5\xf9\xac\x90\xec\x0a\x14\x07\x02\
-\xd9\x92\xda\xfcm)\
-\xbf5\xff\x96\xe6\xcb\xfeW\xda\xb5\xba\x10>Z\xcc\x18\
-{rH\xca\xbf\xb3\x97\x1b\xf2\xab\xb7\x8d\xb0(3[\
-\xb3\xd3\xce\x85,?\xa3\xf6\xef\xb8\x16b*\xbe\xe5\xcc\
-\xee\xc5\xa0\xb1Xz\xa2k\x8d\xc2\xbf\x1d\xf7^a\x13\
-\xae\xac\xce\xb9\x81C\x07\xdaC#\xfc\xb5\x14\x09\xba?\
-\xd1\x04\xe9\xfbKS\xc3\xd3Z\xc0\xe0DG'\xbeD\
-\xf7\x8bl#\x89\xaeJ!lw\x07%\x11}\xcd\x19\
-J\xac\x09]\x95\x14\x0b\x9aX\x17=CU>\xe2\x9b\
-\xbb\x14 5\xae5J\xaf9 \xd1\xc5$N\x9e\x84\
-6\xd4\xf1\xa0\xfd\xafI[\xde7\xa1m\xc2\xb1\xe7\xad\
-\xb5\x1a\x09\xf2\xb1\xc4\x1enc\x0br\xc3r5\xb8\x82\
-1\x0d\xdf\xa2\x88\xf8\x85o\x99\x9c\xd8\xb7\xbb\xbf\xc7\xb7\
-\xb0\x01\x9aPCW}\xc7\xf4\xa3o\x99$\xbb*M\
-X(\xe1i\x8dF\x13\x9d\xf4L\x0eLt\xbf\xf2\xcd\
-\x15\xdf\xdf\xbd\xa0w\xe1\xf6?M>\x8do3%\xb4\
-\x90\xa4X`\xc3\xdeD\xf3Y\xdc\xe3\x1a\xbau\xbe^\
-\xe3\xca\xa9\x99ee\x8d=\xb9`xo7u\x8ch\
-8\xf6\xf8\xb5\x7f[[\xd4g\x0a\xc2mlA\xe4\x16\
-c\xd9\x04n\x93\xd8\xb9\x93\xc0\xc2\x9a\xd4\x1e\x1a\x83\x16\
-9\xb2\xbf\xd5\xfa\x8cu\xcb\xa1[Xc\xccIh\xfc\
-\x925\xe66\xf9Uh\x0a\xf8niu\xa8\xc2\xc7\xd4\
-\xd2\xb2\xe8M\x5c\x9dD\x17\x1a\x96\x91TnL\xce{\
-\x22\xe6!5\xa0`ME\xda\xa9\xce\xf3\x8e3\xb3L\
-\xa9{\xaa\x0be\xdfU\xa41\xc0\xbbX\x9f\x92\x18&\
-\xf6\xf4B\x96uhMVHv\x05)\x91\xb9\xcf\xc1\
-\xe5%\xef\x81\x94+b\x83\xd2`S\x09\x8f\xf72\x8f\
-$)\xe4\xeeq\x1cG\xc0\xa5\xe1Ix\x95k+\xa6\
-Ht\xed\xd5\xa8\x1c.6\x7f\xad;\x0fi\x96(\x87\
-\xda\x07\xe5v\xea\x19\xd9\x12\xe5\x11\xaa]\x8d\x96\xad\x84\
-\x96j;\xa31\xa5\x1d\xcai\x81.\xf4s\x8d4\xf6\
-p\x1b;&\x11EC\x1fH&\x1ei\xa0w\x02n\
-*xni\xdfK\x99\xe8:\xebV\xc2\x8bM\xa1'\
-\xd9\xe4\x12\xd2\xe8\x1863\xff\x99\x90\xec`j]\xb8\
-[\xfb\xbe\xce\x1by}s\x1aj\xa1\x83\xf1\x07\x97\xec\
-\xf8\x03\xf8LHv0\xb5>\xb5\x97\xcf|\xb5$\xe9\
-\xb4x\x9e\xb6\xdd\xcb\x14\x5cs\x83\xd9\x846\xbb\x82H\
-\x8f\xeaO6\xd0\xa0&mvC_\x19\x13OZ\xe6\
-\xa6\xac\xd3\xa4\x9c.haj\x92\x82>\x00\xef\x88B\
-\xc3,R?_)\xbd\x9f\xe7&\xd4\xb1Sp\xadE\
-n\xf9\x9al\x02\xb7\xce:\xf06g\xfb\x94td5\
-i\x12\x92\xf6;T\xec1\x8f#\xf6\x90\xec\x0a\x92\x94\
-\x88$\x11\xde'\x0a\x8f\x90O\xde\x89\x0eI\x08\xf5*\
-\x9a\x17\x8b\xf5\xad\x874\xe34\xd1\xc5\x0f\xdf_\xc5:\
-\xe5\x11\x9f\xf4\x0d\x84M\x99\xf0\x5c3\xd4\x93X\xe0p\
-\x8eX\xd2\x85O\x12\xd6\x93{\xc7\x83\xc7\xbfM\xafp\
-\x19\xb2t\xf2,&L\xc20i\xb8\x8d-\x08\x9b(\
-\x98\xc8\x22\xeb\xbe\xf2-{\xbbrP\xd3y\xc9\x06\xfd\
-\x5c\x9dr'8|\x85\xe2\x1e\xb1\xf5\xb4\x8f\xdeL\xea\
-\xf1\xb0$\xbeD\xe78ZY<\xde~@\xd7\x1f?\
-\xea&\xbc\x8a\xefoF\xae\x14.7\xa9A\x16Y\x1e\
-r\x1c\xc2\x8f}\xc9q\xa2Y\xe3\x06\x0f\x8d\xea\xcaR\
-\x86\x93\x8a=$\xbb\x82,\xb8\x85\xa3\xd0rc\xfcc\
-\xaf\xacI\xf7\xf4\x82\x0eQ\x09=m!\xcaif\xc7\
-h\xcf\x83\xd7\xaa\xa7\x9d7o\x5cB\x89\xae[s\x8b\
-\x9f\x92\xc8\x98\xf0\xfe\xd0\xd8:I*\xb7\xd0\x03\xe9Y\
-{\x22?r\xb7j\xa1\xe5\xfa\xd8W\x9a\xb2o\xcf\x19\
-w\xf0f\x8c\x13\x00x\xb5\xcb?\xf8\x9cp\xf9\x13\xdb\
-zL\x09\xba\x03\xc3'\x11{Hv\x05\xd1\x91\xefr\
-B5|\xcb}\x01~\x1a\xbff*\xbe\xed\xe2)\xb3\
-\xaf\xfe\x7f`\xf4\xbd\x8a'`\x94\xa41\xe8\xc4\xd7@\
-\xd3\x07\xd4;s\x91-\xc7\x13)\xea|kS M\
-\xa2\xeb\xca\x93\xf0\xa4\xfd/i2\x84G:\x8f[\xff\
-\xab:\x83\x8a\xcdX\x13K\x8a\x85\xf6\xfb\x89\xe7,|\
-4\xa8\x96\xa3\xaf\xe9\xb2x\xf2W2\xf14\x5c\xd30\
-\xd1jR\x19\xea\xf1\x93\xe4\xfc<\x14{:\xd9,M\
-(\xf6f\xb6\xcdnT\x13z\x0e#\xe2h\xcf\x9aR\
-\xd5\xb7\xbc3K\xeb\x86\x84\xc5\x89\x8e\xed\x92+\xcdZ\
-b\x97\xbe\xeb\x9d\x0fm\xc1\xdd\xdao\x99\xd6Nh,\
-\x97\x06\x9d\x9c \x1br\x0b\xdcp\xdcnX\xb6:\xd6\
-\xcc\xb4\x96\xb9\xbf\xa3\xc4\xba\xfd\xb3\xea\xe1\xeb\xac\x13f\
-\x16)K\xa2\xeb\xd2\x84w\xbar\xd0\xff\xac\xb0\xb7\x0d\
-\x8f9:J\x1a\xf4*\xed\xae\x8f\xe5\xf8\xecpg\xe2\
-\x05\xe9\xb8\xa8\xea@\xee\xd0`q\x9f\xa4XP\x1a\x0f\
-R\x1b\xd2\xd9\x81O\xba\xc7\xc9\xc4O\x88\xb4$\x19\xf4\
-(O\xc3D\xab\x1a{\x97\xe6r=<)\x82Y\x93\
-\xf7\xba\x962\xf6\x1e\x8f3\xf6P\xb3+P\xe76\xb3\
-\x91\xb0ZE\x03B\x82z#i\xf2M\x9d\x9a\xa7?\
-\xb8\xb5\xd6\xc0\xcc\xa9\x1a\x82\xb56\xa9\xfb\x89\xf7e\xa8\
-\xea\xd9_9\xcb0\x8e\xa2\xc5\xfb\xe5\xbe^\xd7\x84D\
-\xd7\xe5\xab\xe1\x19\xba\xfe\xb0|\xcac\xd3\xd9\xbe]f\
-\xc1\x87\xf9\x13\xe8\xfe\xa4\xfc\x9f$\xad\xa7\xc7\xe4\xeaq\
-\x22\xf2>\x0e6\xf1\x89V5\xf6\x1c\xbb\x22c\xef3\
-\x89\xbd\x7f\x8d+\xf6\x90\xec\x0a&W\xf4\xcd\xbcs\xa0\
-\xf5\x90\x13\x9e]4\xf0\x09\x82\xc5\xc6\xf6Q\xf0\xe9\x82\
-\xec\xfb\x9a\x988\xa1qk\xe5\xca{I\x95\xe8\xba4\
-\xe1\xf5&1\xde\xbb\xdb\x18<\xa5\xb8\x1e\x9b\x84\xb6\xbb\
-B\xdd\xd5\x9e\xf1\x14s\xc7\xa5\xc6T\x9et\xc7\x92&\
-\xf1Y\x8d=$\xbb\x82\xc5S\x02\xb1[\x19*\xe1\xc9\
-\x09o\xb8\xf5 t\xc2'>N\x95a_Y\x92\xcb\
-(\xc4\x13x\xb6'\xaf\xcc\xf5^n\xf2\xcd\x07\xed\x89\
-/yo1\xf0-_\xdd\xc4:\xce\x84\xa7\xc98M\
-\x0d/\x91|>-\x9biyv\xb9\xa0\xd8{#q\
-\xfe\xe7q\xc5\x1e\x92\xdd\x08t\x13^\xca\xdb\xa6\x1e\xf1\
-\xec\xb5\x12\xd4i\xbe\x874\x0e:\x9d\x116gb\xd5\
-\x93\xf0W\xbe\xf9\xd54|\xe1\x8e\xbc\x87\xdab}k\
-)\xcf{\xd1\xdb+\xd9\xf6\xab\xc5\x84\xaf3Tq\xb9\
-jM2m\x99i\x0d{\xc8IM\xb5\x86\xd7N\xe6\
-\xd9\x8fSw\xf6d\xfd|\xd3p\x9c\xba:\xb1\xb7\x94\
-;\xf6\xa4L%\xf6\xfe\x98\xf6\xfbv\x8b0\x91\x0e\x8a\
-\xac\x93i\x8ec\xf2\xcd\xa2u\x02sE\xda#6J\
-&\x9e%\xb7\x1aZ_\x93\x5cd\xf9\xc9\x17\xc7\xdbG\
-\x94Ag\xc6\x94\xa5\xb4\xfb\xd1\x93\xe7\x9d\xf4\xaa\x19\x8e\
-\xf6\xa7\xe9\xe4\x19'=\xc1\xb4\xb1_\xae\xf5\xbbV'\
--\x1d\xd0\xe0\xae\xe5\xa4=\xdf\x17|k\x7f\xc1\x5c>\
-4y\xbf^\xacC\x93\xb9\xfcSK{\x9ct\x98\x87\
-\xf6~\xea\xfe\x97\xea\xd3\xf9\x0cp'~\xf2\xc4\xdeQ\
-\xea\x99~\x0a4\xdc\x11\x0c\xc0\x97d\xf7\xd2\xb6\x96\x05\
-\xbaX\x96\xbat\xe5\xfdt9\xaei\x9d=\xff\x95n\
-5\x8a\xbc=\x89\xc7e\xc9~$\x00oG\xcc\x9f\x95\
-\x8c\xf9E\xda\xab\xfe\xa7\x03C\xe75\xc1\x85t\xcbK\
-\x8f\x8b\x96\x15G\xfc\xd3\xd5c\x22\xb1\xac\x1d\x03\xdeG\
-\xd7$)mf}\x22\x22\x14\x0f\x8e\xdc\xc9\xac\x1e'\
--KK\xb6lJ\xe6\xfe\xb4\xc5\x1e\x92\x9dG\xd1\xc9\
-\x0e\xa6\x8b~\xabY\xda[(\x1d\xd8\x1b\xea\x99\xd5\xa6\
-\x84I\xd4T \x1b\xb4\xd9\xc1\xdc\xd1D\xc7\xf6\xe6\xab\
-4\xa3\xf8\x93\x06}\x8bs$\xba\xd9\x80\x89\x00`\xae\
-\xb4\xbf`\xdc\xc6\xe3\xd5tP\xef\xd9\xea\xe1Z\xcb\xd8\
-\xbd\xcb\xa8\xd4\xd3\x94\xa0\xdf\xb9\xd0\xfe&7\x0e\x7fm\
-%Q\xa66V\x98\x1c$;\x98+\xfd_\xf2,\xc9\
-j\xf9\x06\xbb\xe77\xac\xd3\xdb\xd5\xa6\xbe&\x9d\x11\xe5\
-\xf6\x93-\xc9-\x19\xbe\xb1\x900}\x90\xec`nt\
-\xbe\xb8:PS3\x15\xfdo\xfa\xc6f\xdeC\x87\xcf\
-\xec\x18Y\xb2C\x03?L\x1bv\xb6\x5c\x5c+ux\
-\x003L\x1ftP\xc0\xdc\xf8}c{\x7f\xd8\x91\xff\
-:VLnq\xbfG\xa2\x9b=\xa8}\xc1\xdc\x89\x1f\
-<\x97\xb6;kL\xea)\x860\x18{\xf6!\xd9\xc1\
-\xdc\xea\xcc\xb6Q\xb5\xd6~\xa3S9\xc9\xc9Py?\
-u\x167\xe5\xb5\xf3\x88\xe8Gc\xa2\x93\x0b\xb7p4\
-\xbd\xdff\x06\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\
-\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\
-\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\
-\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\
-\x00\x00\x00\x00\x00\x00\x01\xff\x07\xb4<2M\x93'\x06\
-\xcb\x00\x00\x00\x00IEND\xaeB`\x82\
-\x00\x00\x03X\
-<\
-?xml version=\x221.\
-0\x22 encoding=\x22utf\
--8\x22?>\x0a\x0a\x0a\x0a path{fill\
-:rgb(150, 146, 1\
-44)}\x0a polygon\
-{fill:rgb(150, 1\
-46, 144)}\x0a ci\
-rcle{fill:rgb(15\
-0, 146, 144)}\x0a \
- rect{fill:rgb(\
-150, 146, 144)}\x0a\
- \x0a \x0a\
- \x0a\
-\x00\x00\x03Y\
-<\
-?xml version=\x221.\
-0\x22 encoding=\x22utf\
--8\x22?>\x0a\x0a\x0a\x0a path{fil\
-l:rgb(150, 146, \
-144)}\x0a polygo\
-n{fill:rgb(150, \
-146, 144)}\x0a c\
-ircle{fill:rgb(1\
-50, 146, 144)}\x0a \
- rect{fill:rgb\
-(150, 146, 144)}\
-\x0a \x0a \
-\x0a \x0a\
-\x00\x00\x03X\
-<\
-?xml version=\x221.\
-0\x22 encoding=\x22utf\
--8\x22?>\x0a\x0a\x0a\x0a path{fill\
-:rgb(150, 146, 1\
-44)}\x0a polygon\
-{fill:rgb(150, 1\
-46, 144)}\x0a ci\
-rcle{fill:rgb(15\
-0, 146, 144)}\x0a \
- rect{fill:rgb(\
-150, 146, 144)}\x0a\
- \x0a \x0a\
- \x0a\
-"
-
-qt_resource_name = b"\
-\x00\x08\
-\x05\xe2Y'\
-\x00l\
-\x00o\x00g\x00o\x00.\x00p\x00n\x00g\
-\x00\x08\
-\x0aaZ\xa7\
-\x00i\
-\x00c\x00o\x00n\x00.\x00p\x00n\x00g\
-\x00\x0d\
-\x0dq\x0b\x87\
-\x00c\
-\x00o\x00l\x00l\x00a\x00p\x00s\x00e\x00d\x00.\x00s\x00v\x00g\
-\x00\x0c\
-\x07)\x8aG\
-\x00e\
-\x00x\x00p\x00a\x00n\x00d\x00e\x00d\x00.\x00s\x00v\x00g\
-"
-
-qt_resource_struct = b"\
-\x00\x00\x00\x00\x00\x02\x00\x00\x00\x04\x00\x00\x00\x01\
-\x00\x00\x00\x00\x00\x00\x00\x00\x00\x01\x00\x00\x00\x00\
-\x00\x00\x00L\x00\x00\x00\x00\x00\x01\x00\x00\x14\x9b\
-\x00\x00\x00\x16\x00\x00\x00\x00\x00\x01\x00\x00\x0d\xe2\
-\x00\x00\x00,\x00\x00\x00\x00\x00\x01\x00\x00\x11>\
-"
-
-def qInitResources():
- QtCore.qRegisterResourceData(0x01, qt_resource_struct, qt_resource_name, qt_resource_data)
-
-def qCleanupResources():
- QtCore.qUnregisterResourceData(0x01, qt_resource_struct, qt_resource_name, qt_resource_data)
-
-qInitResources()
diff --git a/server/gui/hook-cefpython3.py b/server/gui/hook-cefpython3.py
deleted file mode 100644
index 6e3b66b1..00000000
--- a/server/gui/hook-cefpython3.py
+++ /dev/null
@@ -1,233 +0,0 @@
-"""
-This is PyInstaller hook file for CEF Python. This file
-helps PyInstaller find CEF Python dependencies that are
-required to run final executable.
-
-See PyInstaller docs for hooks:
-https://pyinstaller.readthedocs.io/en/stable/hooks.html
-"""
-
-import glob
-import os
-import platform
-import re
-import sys
-import PyInstaller
-from PyInstaller.utils.hooks import is_module_satisfies, get_package_paths
-from PyInstaller.compat import is_win, is_darwin, is_linux, is_py2
-from PyInstaller import log as logging
-
-# Constants
-CEFPYTHON_MIN_VERSION = "57.0"
-PYINSTALLER_MIN_VERSION = "3.2.1"
-
-# Makes assumption that using "python.exe" and not "pyinstaller.exe"
-# TODO: use this code to work cross-platform:
-# from PyInstaller.utils.hooks import get_package_paths
-# get_package_paths("cefpython3")
-
-CEFPYTHON3_DIR = get_package_paths("cefpython3")[1]
-
-CYTHON_MODULE_EXT = ".pyd" if is_win else ".so"
-
-# Globals
-logger = logging.getLogger(__name__)
-
-
-# Functions
-def check_platforms():
- if not is_win and not is_darwin and not is_linux:
- raise SystemExit("Error: Currently only Windows, Linux and Darwin " "platforms are supported, see Issue #135.")
-
-
-def check_pyinstaller_version():
- """Using is_module_satisfies() for pyinstaller fails when
- installed using 'pip install develop.zip' command
- (PyInstaller Issue #2802)."""
- # Example version string for dev version of pyinstaller:
- # > 3.3.dev0+g5dc9557c
- version = PyInstaller.__version__
- match = re.search(r"^\d+\.\d+(\.\d+)?", version)
- if not (match.group(0) >= PYINSTALLER_MIN_VERSION):
- raise SystemExit("Error: pyinstaller %s or higher is required" % PYINSTALLER_MIN_VERSION)
-
-
-def check_cefpython3_version():
- if not is_module_satisfies("cefpython3 >= %s" % CEFPYTHON_MIN_VERSION):
- raise SystemExit("Error: cefpython3 %s or higher is required" % CEFPYTHON_MIN_VERSION)
-
-
-def get_cefpython_modules():
- """Get all cefpython Cython modules in the cefpython3 package.
- It returns a list of names without file extension. Eg.
- 'cefpython_py27'. """
- pyds = glob.glob(os.path.join(CEFPYTHON3_DIR, "cefpython_py*" + CYTHON_MODULE_EXT))
- assert len(pyds) > 1, "Missing cefpython3 Cython modules"
- modules = []
- for path in pyds:
- filename = os.path.basename(path)
- mod = filename.replace(CYTHON_MODULE_EXT, "")
- modules.append(mod)
- return modules
-
-
-def get_excluded_cefpython_modules():
- """CEF Python package includes Cython modules for various Python
- versions. When using Python 2.7 pyinstaller should not
- bundle modules for eg. Python 3.6, otherwise it will
- cause to include Python 3 dll dependencies. Returns a list
- of fully qualified names eg. 'cefpython3.cefpython_py27'."""
- pyver = "".join(map(str, sys.version_info[:2]))
- pyver_string = "py%s" % pyver
- modules = get_cefpython_modules()
- excluded = []
- for mod in modules:
- if pyver_string in mod:
- continue
- excluded.append("cefpython3.%s" % mod)
- logger.info("Exclude cefpython3 module: %s" % excluded[-1])
- return excluded
-
-
-def get_cefpython3_datas():
- """Returning almost all of cefpython binaries as DATAS (see exception
- below), because pyinstaller does strange things and fails if these are
- returned as BINARIES. It first updates manifest in .dll files:
- >> Updating manifest in chrome_elf.dll
-
- And then because of that it fails to load the library:
- >> hsrc = win32api.LoadLibraryEx(filename, 0, LOAD_LIBRARY_AS_DATAFILE)
- >> pywintypes.error: (5, 'LoadLibraryEx', 'Access is denied.')
-
- It is not required for pyinstaller to modify in any way
- CEF binaries or to look for its dependencies. CEF binaries
- does not have any external dependencies like MSVCR or similar.
-
- The .pak .dat and .bin files cannot be marked as BINARIES
- as pyinstaller would fail to find binary depdendencies on
- these files.
-
- One exception is subprocess (subprocess.exe on Windows) executable
- file, which is passed to pyinstaller as BINARIES in order to collect
- its dependecies.
-
- DATAS are in format: tuple(full_path, dest_subdir).
- """
- ret = list()
-
- if is_win:
- cefdatadir = "."
- elif is_darwin or is_linux:
- cefdatadir = "."
- else:
- assert False, "Unsupported system {}".format(platform.system())
-
- # Binaries, licenses and readmes in the cefpython3/ directory
- for filename in os.listdir(CEFPYTHON3_DIR):
- # Ignore Cython modules which are already handled by
- # pyinstaller automatically.
- if filename[: -len(CYTHON_MODULE_EXT)] in get_cefpython_modules():
- continue
-
- # CEF binaries and datas
- extension = os.path.splitext(filename)[1]
- if extension in [
- ".exe",
- ".dll",
- ".pak",
- ".dat",
- ".bin",
- ".txt",
- ".so",
- ".plist",
- ] or filename.lower().startswith("license"):
- logger.info("Include cefpython3 data: {}".format(filename))
- ret.append((os.path.join(CEFPYTHON3_DIR, filename), cefdatadir))
-
- if is_darwin:
- # "Chromium Embedded Framework.framework/Resources" with subdirectories
- # is required. Contain .pak files and locales (each locale in separate
- # subdirectory).
- resources_subdir = os.path.join("Chromium Embedded Framework.framework", "Resources")
- base_path = os.path.join(CEFPYTHON3_DIR, resources_subdir)
- assert os.path.exists(base_path), "{} dir not found in cefpython3".format(resources_subdir)
- for path, dirs, files in os.walk(base_path):
- for file in files:
- absolute_file_path = os.path.join(path, file)
- dest_path = os.path.relpath(path, CEFPYTHON3_DIR)
- ret.append((absolute_file_path, dest_path))
- logger.info("Include cefpython3 data: {}/{}".format(dest_path, file))
- elif is_win or is_linux:
- # The .pak files in cefpython3/locales/ directory
- locales_dir = os.path.join(CEFPYTHON3_DIR, "locales")
- assert os.path.exists(locales_dir), "locales/ dir not found in cefpython3"
- for filename in os.listdir(locales_dir):
- logger.info("Include cefpython3 data: {}/{}".format(os.path.basename(locales_dir), filename))
- ret.append((os.path.join(locales_dir, filename), os.path.join(cefdatadir, "locales")))
-
- # Optional .so/.dll files in cefpython3/swiftshader/ directory
- swiftshader_dir = os.path.join(CEFPYTHON3_DIR, "swiftshader")
- if os.path.isdir(swiftshader_dir):
- for filename in os.listdir(swiftshader_dir):
- logger.info("Include cefpython3 data: {}/{}".format(os.path.basename(swiftshader_dir), filename))
- ret.append((os.path.join(swiftshader_dir, filename), os.path.join(cefdatadir, "swiftshader")))
- return ret
-
-
-# ----------------------------------------------------------------------------
-# Main
-# ----------------------------------------------------------------------------
-
-# Checks
-check_platforms()
-check_pyinstaller_version()
-check_cefpython3_version()
-
-# Info
-logger.info("CEF Python package directory: %s" % CEFPYTHON3_DIR)
-
-# Hidden imports.
-# PyInstaller has no way on detecting imports made by Cython
-# modules, so all pure Python imports made in cefpython .pyx
-# files need to be manually entered here.
-# TODO: Write a tool script that would find such imports in
-# .pyx files automatically.
-hiddenimports = [
- "codecs",
- "copy",
- "datetime",
- "inspect",
- "json",
- "os",
- "platform",
- "random",
- "re",
- "sys",
- "time",
- "traceback",
- "types",
- "urllib",
- "weakref",
-]
-if is_py2:
- hiddenimports += [
- "urlparse",
- ]
-
-# Excluded modules
-excludedimports = get_excluded_cefpython_modules()
-
-# Include binaries requiring to collect its dependencies
-if is_darwin or is_linux:
- binaries = [(os.path.join(CEFPYTHON3_DIR, "subprocess"), ".")]
-elif is_win:
- binaries = [(os.path.join(CEFPYTHON3_DIR, "subprocess.exe"), ".")]
-else:
- binaries = []
-
-# Include datas
-datas = get_cefpython3_datas()
-
-# Notify pyinstaller.spec code that this hook was executed
-# and that it succeeded.
-os.environ["PYINSTALLER_CEFPYTHON3_HOOK_SUCCEEDED"] = "1"
diff --git a/server/gui/images/cellxgene_logo.png b/server/gui/images/cellxgene_logo.png
deleted file mode 100644
index 6dc643da..00000000
Binary files a/server/gui/images/cellxgene_logo.png and /dev/null differ
diff --git a/server/gui/images/cxg_icons.icns b/server/gui/images/cxg_icons.icns
deleted file mode 100644
index d57d20cf..00000000
Binary files a/server/gui/images/cxg_icons.icns and /dev/null differ
diff --git a/server/gui/images/favicon.png b/server/gui/images/favicon.png
deleted file mode 100644
index 58f43344..00000000
Binary files a/server/gui/images/favicon.png and /dev/null differ
diff --git a/server/gui/images/icon.ico b/server/gui/images/icon.ico
deleted file mode 100644
index cea1f139..00000000
Binary files a/server/gui/images/icon.ico and /dev/null differ
diff --git a/server/gui/images/properties_contract.svg b/server/gui/images/properties_contract.svg
deleted file mode 100644
index abaaae9c..00000000
--- a/server/gui/images/properties_contract.svg
+++ /dev/null
@@ -1,14 +0,0 @@
-
-
-
-
-
-
diff --git a/server/gui/images/properties_expand.svg b/server/gui/images/properties_expand.svg
deleted file mode 100644
index 966b259e..00000000
--- a/server/gui/images/properties_expand.svg
+++ /dev/null
@@ -1,14 +0,0 @@
-
-
-
-
-
-
diff --git a/server/gui/main.py b/server/gui/main.py
deleted file mode 100644
index 9963c684..00000000
--- a/server/gui/main.py
+++ /dev/null
@@ -1,401 +0,0 @@
-# flake8: noqa F403, F405
-from functools import partialmethod
-from multiprocessing import Pipe, Process, freeze_support
-from os import environ
-from os.path import splitext, basename, dirname, join
-import sys
-import threading
-
-from cefpython3 import cefpython as cef
-import PySide2
-from PySide2.QtGui import *
-from PySide2.QtCore import *
-from PySide2.QtWidgets import *
-
-import server.gui.cellxgene_rc
-from server.gui.browser import CefWidget, CefApplication
-from server.gui.workers import Worker, SiteReadyWorker
-from server.gui.utils import WINDOWS, LINUX, MAC, FileLoadSignals, Emitter, WorkerSignals, FileChanged
-from server.utils.utils import find_available_port
-
-if WINDOWS or LINUX:
- dirname = dirname(PySide2.__file__)
- plugin_path = join(dirname, "plugins", "platforms")
- environ["QT_QPA_PLATFORM_PLUGIN_PATH"] = plugin_path
-
-# Configuration
-# TODO remember this or calculate it?
-WIDTH = 1300
-HEIGHT = 800
-MAX_CONTENT_WIDTH = 700
-GUI_PORT = find_available_port("localhost")
-BROWSER_INDEX = 0
-LOAD_INDEX = 1
-
-
-class MainWindow(QMainWindow):
- def __init__(self):
- super(MainWindow, self).__init__(None)
- self.cef_widget = None
- self.data_widget = None
- self.stacked_layout = None
- self.parent_conn, self.child_conn = None, None
- self.load_emitter = None
- self.emitter_thread = None
- self.worker = None
- self.url = f"http://localhost:{GUI_PORT}/"
- self.setWindowTitle("cellxgene")
-
- # Strong focus - accepts focus by tab & click
- self.setFocusPolicy(Qt.StrongFocus)
- self.setupLayout()
- self.setupMenu()
-
- def showBrowser(self):
- self.stacked_layout.setCurrentIndex(BROWSER_INDEX)
-
- def restartOnError(self):
- self.window().shutdownServer()
- # close emitter on error/finished
- self.parent_conn, self.child_conn = Pipe()
- self.load_emitter = Emitter(self.parent_conn, WorkerSignals)
- self.emitter_thread = threading.Thread(target=self.load_emitter.run, daemon=True)
- self.emitter_thread.start()
- # send to load with error message?
-
- def setupLayout(self):
- self.resize(WIDTH, HEIGHT)
- self.cef_widget = CefWidget(self)
- self.cef_widget.setSizePolicy(QSizePolicy(QSizePolicy.MinimumExpanding, QSizePolicy.MinimumExpanding))
- self.data_widget = LoadWidget(self)
- self.stacked_layout = QStackedLayout()
- self.stacked_layout.addWidget(self.cef_widget)
- self.stacked_layout.addWidget(self.data_widget)
- main_layout = QVBoxLayout()
- main_layout.setContentsMargins(0, 0, 0, 0)
- main_layout.setSpacing(0)
- main_layout.addLayout(self.stacked_layout)
- frame = QFrame()
- frame.setLayout(main_layout)
- self.setCentralWidget(frame)
-
- if WINDOWS:
- # On Windows with PyQt5 main window must be shown first
- # before CEF browser is embedded, otherwise window is
- # not resized and application hangs during resize.
- self.show()
-
- # Browser can be embedded only after layout was set up
- self.cef_widget.embedBrowser()
-
- if LINUX:
- # On Linux with PyQt5 the QX11EmbedContainer widget is
- # no longer available. An equivalent in Qt5 is to create
- # a hidden window, embed CEF browser in it and then
- # create a container for that hidden window and replace
- # cef widget in the layout with the container.
- self.container = QWidget.createWindowContainer(self.cef_widget.hidden_window, parent=self)
- self.stacked_layout.replaceWidget(self.cef_widget, self.container)
- self.stacked_layout.setCurrentIndex(LOAD_INDEX)
-
- def setupServer(self):
- self.shutdownServer()
- # close emitter on error/finished
- self.parent_conn, self.child_conn = Pipe()
- self.load_emitter = Emitter(self.parent_conn, WorkerSignals)
- self.emitter_thread = threading.Thread(target=self.load_emitter.run, daemon=True)
- self.emitter_thread.start()
- # send to load with error message?
-
- def shutdownServer(self):
- if self.worker:
- self.worker.terminate()
- if self.parent_conn:
- self.parent_conn.close()
-
- def setupMenu(self):
- # TODO add communication to subprocess on reload
- main_menu = self.menuBar()
- file_menu = main_menu.addMenu("File")
- load_action = QAction("Load file...", self)
- load_action.setStatusTip("Load file")
- load_action.setShortcut("Ctrl+O")
- load_action.triggered.connect(self.showLoad)
- file_menu.addAction(load_action)
-
- def showLoad(self):
- self.clearMessages()
- self.stacked_layout.setCurrentIndex(LOAD_INDEX)
-
- def clearMessages(self):
- self.data_widget.reset()
-
- def closeEvent(self, event):
- # Close browser (force=True) and free CEF reference
- if self.cef_widget.browser:
- self.cef_widget.browser.CloseBrowser(True)
- self.clearBrowserReferences()
-
- def clearBrowserReferences(self):
- # Clear browser references that you keep anywhere in your
- # code. All references must be cleared for CEF to shutdown cleanly.
- self.cef_widget.browser = None
-
-
-class LoadWidget(QFrame):
- def __init__(self, parent):
- super(LoadWidget, self).__init__(parent=parent)
- # Init layout
- load_ui_layout = QVBoxLayout()
- h_margin = (WIDTH - MAX_CONTENT_WIDTH) // 2
- if h_margin < 10:
- h_margin = 10
- load_ui_layout.setContentsMargins(h_margin, 20, h_margin, 20)
- logo_layout = QHBoxLayout()
- logo_layout.setContentsMargins(0, 0, 0, 20)
-
- file_layout = QVBoxLayout()
-
- message_layout = QHBoxLayout()
- message_layout.setContentsMargins(0, 0, 0, 0)
-
- self.serverError = False
- self.file_name = FilePath()
-
- self.label = QLabel()
- logo = QPixmap(":/logo.png")
- self.label.setPixmap(logo)
- self.label.setContentsMargins(100, 0, 100, 0)
- logo_layout.addWidget(self.label)
-
- # UI section
- # TODO add cancel button to send back to browser (if available)
-
- self.file_area = FileArea(self)
- self.file_name.signals.changed.connect(self.updatePath)
-
- self.launch_widget = QLabel("Select a file to launch cellxgene")
- # self.launch_widget.setEnabled(False)
- # self.launch_widget.clicked.connect(self.onLoad)
-
- self.progress = QProgressBar()
- self.progress.setTextVisible(False)
-
- file_layout.addWidget(self.file_area)
- self.loading_layout = QStackedLayout()
- self.loading_layout.addWidget(self.launch_widget)
- self.loading_layout.addWidget(self.progress)
- file_layout.addLayout(self.loading_layout)
- file_layout.setStretch(0, 10)
-
- # Error section
- self.error_label = QLabel("")
- self.error_label.setWordWrap(True)
- self.error_label.setFixedWidth(MAX_CONTENT_WIDTH)
- message_layout.addWidget(self.error_label)
-
- # Options Form
-
- # Layout
- for l in [logo_layout, file_layout, message_layout]:
- load_ui_layout.addLayout(l)
-
- # TODO remove magic number
- load_ui_layout.setStretch(1, 10)
- self.setLayout(load_ui_layout)
-
- self.timer = QTimer()
- self.timer.setInterval(100)
- self.timer.timeout.connect(self.updateProgress)
-
- self.signals = FileLoadSignals()
- self.signals.selectedFile.connect(self.createScanpyEngine)
- self.signals.error.connect(self.onError)
-
- def updatePath(self):
- file_name = self.file_name.value
- if file_name:
- self.file_area.label.setText("File: " + file_name)
- else:
- self.file_area.label.setText("")
- # self.launch_widget.setEnabled(bool(file_name))
-
- def reset(self):
- self.loading_layout.setCurrentIndex(0)
- self.timer.stop()
- self.error_label.setText("")
- self.file_name.updateValue(None)
-
- def updateProgress(self):
- curr_val = self.progress.value()
- next_val = (curr_val + 1) % 100
- self.progress.setValue(next_val)
-
- def resetProgress(self):
- self.progress.setValue(0)
- self.loading_layout.setCurrentIndex(0)
- self.timer.stop()
-
- def createScanpyEngine(self, file_name):
- title = splitext(basename(file_name))[0]
- self.window().setupServer()
- worker = Worker(
- self.window().parent_conn,
- self.window().child_conn,
- file_name,
- host="127.0.0.1",
- port=GUI_PORT,
- title=title,
- engine_options={},
- )
- self.window().load_emitter.signals.ready.connect(self.onDataReady)
- self.window().load_emitter.signals.engine_error.connect(self.onServerError)
- self.window().load_emitter.signals.server_error.connect(self.onServerError)
- # Error is generic error from emitter
- self.window().load_emitter.signals.error.connect(self.onServerError)
- self.window().worker = Process(target=worker.run, daemon=True)
- self.window().worker.start()
- self.window().child_conn.close()
-
- def onLoad(self):
- if self.file_name.value:
- # Reset error on reload
- self.serverError = False
- self.loading_layout.setCurrentIndex(1)
- self.timer.start()
- self.signals.selectedFile.emit(self.file_name.value)
- else:
- self.signals.error.emit("Please select a file before launching.")
-
- def onDataReady(self):
- self.site_ready_worker = SiteReadyWorker(self.window().url)
- self.site_ready_worker.signals.ready.connect(self.onServerReady)
- self.site_ready_worker.signals.error.connect(self.onServerError)
-
- srw_thread = threading.Thread(target=self.site_ready_worker.run, daemon=True)
- srw_thread.start()
-
- def onServerReady(self):
- if not self.serverError:
- self.resetProgress()
- self.window().cef_widget.browser.Navigate(self.window().url)
- self.window().showBrowser()
-
- def onError(self, err, server_error=False):
- # Restart worker
- if server_error:
- self.serverError = True
- # Report error and switch to load screen
- self.window().shutdownServer()
- self.resetProgress()
- self.window().stacked_layout.setCurrentIndex(LOAD_INDEX)
- self.error_label.setText(f"Error: {err}")
- self.error_label.resize(MAX_CONTENT_WIDTH, self.error_label.height())
- self.window().repaint()
-
- onServerError = partialmethod(onError, server_error=True)
-
-
-class FilePath(QObject):
- def __init__(self):
- super(FilePath, self).__init__()
- self.value = ""
- self.signals = FileChanged()
-
- def updateValue(self, path=None):
- self.value = path
- self.signals.changed.emit(self.value != path)
-
-
-class FileArea(QFrame):
- def __init__(self, parent):
- super(FileArea, self).__init__()
- self.setFrameShape(QFrame.Box)
- self.setMinimumHeight(100)
- self.setFixedWidth(MAX_CONTENT_WIDTH)
- self.setAcceptDrops(True)
- self.instructions = QLabel(self)
- self.instructions.setText("Drag & Drop a h5ad file to load or open")
- self.instructions.setGeometry(10, 10, MAX_CONTENT_WIDTH, self.instructions.height())
- self.loadButton = QPushButton("Open...", parent=self)
- x_pos = (MAX_CONTENT_WIDTH - self.loadButton.width()) / 2
- self.loadButton.setGeometry(x_pos, 50, self.loadButton.width(), self.loadButton.height())
- self.loadButton.clicked.connect(self.fileBrowse)
- self.label = QLabel(self)
- self.label.setGeometry(10, 75, MAX_CONTENT_WIDTH, self.label.height())
-
- def fileBrowse(self):
- options = QFileDialog.Options()
- # options |= QFileDialog.DontUseNativeDialog
- file_name, _ = QFileDialog.getOpenFileName(self, "Open H5AD File", "", "H5AD Files (*.h5ad)", options=options)
- if file_name:
- self.parent().file_name.updateValue(file_name)
- self.parent().onLoad()
-
- def dragEnterEvent(self, e):
- if e.mimeData().hasUrls:
- e.accept()
- else:
- e.ignore()
-
- def dragMoveEvent(self, e):
- if e.mimeData().hasUrls:
- e.accept()
- else:
- e.ignore()
-
- def dropEvent(self, e):
- """
- Drop files directly onto the widget
- File locations are stored in fname
- :param e:
- :return:
- """
- if e.mimeData().hasUrls:
- e.setDropAction(Qt.CopyAction)
- e.accept()
- for url in e.mimeData().urls():
- file_name = str(url.toLocalFile())
- self.parent().file_name.updateValue(file_name)
- self.parent().onLoad()
- else:
- e.ignore()
-
-
-def main():
- freeze_support()
- # This generates an error.log file on error
- sys.excepthook = cef.ExceptHook # To shutdown all CEF processes on error
- settings = {}
- # Instead of timer loop
- if MAC:
- settings["external_message_pump"] = True
-
- # Create and launch cef browser and qt window
- cef.Initialize(settings)
- app = CefApplication(sys.argv)
- main_window = MainWindow()
- main_window.setWindowTitle("cellxgene")
- main_window.setUnifiedTitleAndToolBarOnMac(True)
- main_window.setWindowIcon(QIcon(":icon.png"))
- main_window.show()
- main_window.activateWindow()
- main_window.raise_()
- try:
- app.exec_()
- except Exception as e:
- raise
- finally:
- # Clean up on close
- if not cef.GetAppSetting("external_message_pump"):
- app.stopTimer()
-
- main_window.shutdownServer()
- del main_window # Just to be safe, similarly to "del app"
- del app # Must destroy app object before calling Shutdown
- cef.Shutdown()
- sys.exit(0)
-
-
-if __name__ == "__main__":
- main()
diff --git a/server/gui/utils.py b/server/gui/utils.py
deleted file mode 100644
index 7873f49d..00000000
--- a/server/gui/utils.py
+++ /dev/null
@@ -1,81 +0,0 @@
-import errno
-import platform
-
-from PySide2.QtCore import QObject, Signal
-
-# Detect OS
-WINDOWS = platform.system() == "Windows"
-LINUX = platform.system() == "Linux"
-MAC = platform.system() == "Darwin"
-
-
-class WorkerSignals(QObject):
- """
- Defines the signals available from a running worker thread.
- Supported signals are:
- finished
- ready
- error - `str` error message
- result - `object` data returned from processing, anything
- """
-
- finished = Signal()
- engine_error = Signal(str)
- server_error = Signal(str)
- error = Signal(str)
- result = Signal(object)
- ready = Signal()
-
-
-class SiteReadySignals(QObject):
- """
- Defines the signals available from a running worker thread.
- Supported signals are:
- timeout
- ready
- error - `str` error message
- """
-
- ready = Signal()
- timeout = Signal()
- error = Signal(str)
-
-
-class FileLoadSignals(QObject):
- selectedFile = Signal(str)
- error = Signal(str)
-
-
-class FileChanged(QObject):
- changed = Signal(bool)
-
-
-class Emitter:
- def __init__(self, transport, signals):
- self.transport = transport
- self.signals = signals()
-
- def _emit(self, signature, args=None):
- if args is None:
- getattr(self.signals, signature).emit()
- else:
- getattr(self.signals, signature).emit(args)
-
- def run(self):
- while True:
- try:
- signature = self.transport.recv()
- except EOFError:
- # Server done
- break
- except OSError as e:
- if e.errno == errno.EBADF:
- break
- else:
- self.signals.error.emit(str(e))
- break
- except Exception as e:
- self.signals.error.emit(str(e))
- break
- else:
- self._emit(*signature)
diff --git a/server/gui/workers.py b/server/gui/workers.py
deleted file mode 100644
index e77e057d..00000000
--- a/server/gui/workers.py
+++ /dev/null
@@ -1,90 +0,0 @@
-from multiprocessing import Process
-import time
-
-import requests
-
-from server.gui.utils import SiteReadySignals
-
-
-class EmittingProcess(Process):
- def __init__(self, parent_conn, child_conn, *arg, **kwargs):
- super(EmittingProcess, self).__init__()
- self.parent_conn = parent_conn
- self.child_conn = child_conn
-
- def run(self):
- self.parent_conn.close()
-
- def emit(self, signal_name, *args):
- message = (signal_name, *args)
- self.child_conn.send(message)
-
-
-class Worker(EmittingProcess):
- def __init__(self, parent_conn, child_conn, data_file, host, port, title, engine_options, *args, **kwargs):
- super(Worker, self).__init__(parent_conn, child_conn)
- self.data_file = data_file
- self.host = host
- self.port = port
- self.title = title
- self.engine_options = engine_options
-
- def run(self):
- super(Worker, self).run()
- if not self.data_file:
- self.emit("finished")
- return
- from server.app.app import Server
- from server.app.scanpy_engine.scanpy_engine import ScanpyEngine
-
- # create server
- try:
- server = Server()
- server.create_app()
- except Exception as e:
- self.emit("server_error", str(e))
- self.emit("finished")
- return
- # load data
- try:
- args = {
- "max_category_items": 100,
- "diffexp_lfc_cutoff": 0.01,
- "obs_names": None,
- "var_names": None,
- }
- args.update(self.engine_options)
- data = ScanpyEngine(self.data_file, args)
- server.attach_data(data, self.title)
- self.emit("ready")
- except Exception as e:
- self.emit("engine_error", str(e))
- self.emit("finished")
- return
- # launch server
- try:
- server.app.run(host=self.host, debug=False, port=self.port, threaded=True)
- except Exception as e:
- self.emit("server_error", str(e))
- finally:
- self.emit("finished")
-
-
-class SiteReadyWorker:
- def __init__(self, location):
- super(SiteReadyWorker, self).__init__()
- self.signals = SiteReadySignals()
- self.location = location
-
- def run(self):
- session = requests.Session()
- for i in range(90):
- try:
- session.head(self.location)
- self.signals.ready.emit()
- break
- except requests.exceptions.ConnectionError:
- time.sleep(1)
- except Exception as e:
- self.signals.error.emit(str(e))
- self.signals.timeout.emit()
diff --git a/server/requirements-gui.txt b/server/requirements-gui.txt
deleted file mode 100644
index 352eb4da..00000000
--- a/server/requirements-gui.txt
+++ /dev/null
@@ -1,4 +0,0 @@
-cefpython3>=66
-requests
-PyInstaller>=3.4
-PySide2>=5.12.3
diff --git a/server/requirements.txt b/server/requirements.txt
index 4c64e693..68901144 100644
--- a/server/requirements.txt
+++ b/server/requirements.txt
@@ -6,11 +6,14 @@ Flask-Caching>=1.4.0
Flask-Compress>=1.4.0
Flask-Cors>=3.0.6
Flask-RESTful>=0.3.6
+flask-server-timing>=0.1.2
flatbuffers>=1.10.0
fsspec>=0.4.4
-numpy>=1.15.2
+numpy>=1.16.0
packaging>=20.0
pandas>=0.24.2
scipy>=1.3.0
tables==3.5.1
requests>=2.22.0
+tiledb>=0.5.3
+s3fs>=0.4.0
diff --git a/server/test/decode_fbs.py b/server/test/decode_fbs.py
index ea029a1c..5debe7ae 100644
--- a/server/test/decode_fbs.py
+++ b/server/test/decode_fbs.py
@@ -2,17 +2,17 @@
Code to decode, for testing purposes, the flatbuffer encoded blobs.
This code will need to be updated if fbs/matrix.fbs changes.
-For more information, see fbs/matrix.fbs and server/app/util/fbs/
+For more information, see fbs/matrix.fbs and server/data_common/fbs/
"""
import json
-import server.app.util.fbs.NetEncoding.TypedArray as TypedArray
-import server.app.util.fbs.NetEncoding.Matrix as Matrix
-import server.app.util.fbs.NetEncoding.Int32Array as Int32Array
-import server.app.util.fbs.NetEncoding.Uint32Array as Uint32Array
-import server.app.util.fbs.NetEncoding.Float32Array as Float32Array
-import server.app.util.fbs.NetEncoding.Float64Array as Float64Array
-import server.app.util.fbs.NetEncoding.JSONEncodedArray as JSONEncodedArray
+import server.data_common.fbs.NetEncoding.TypedArray as TypedArray
+import server.data_common.fbs.NetEncoding.Matrix as Matrix
+import server.data_common.fbs.NetEncoding.Int32Array as Int32Array
+import server.data_common.fbs.NetEncoding.Uint32Array as Uint32Array
+import server.data_common.fbs.NetEncoding.Float32Array as Float32Array
+import server.data_common.fbs.NetEncoding.Float64Array as Float64Array
+import server.data_common.fbs.NetEncoding.JSONEncodedArray as JSONEncodedArray
def decode_typed_array(tarr):
diff --git a/server/test/test_scanpy_engine.py b/server/test/test_anndata_adaptor.py
similarity index 90%
rename from server/test/test_scanpy_engine.py
rename to server/test/test_anndata_adaptor.py
index 6fca92cc..24f17ea9 100644
--- a/server/test/test_scanpy_engine.py
+++ b/server/test/test_anndata_adaptor.py
@@ -9,12 +9,12 @@ from parameterized import parameterized_class
import numpy as np
import pandas as pd
-from server.app.scanpy_engine.scanpy_engine import ScanpyEngine
-from server.app.util.errors import FilterError, DisabledFeatureError
-from server.app.util.data_locator import DataLocator
+from server.data_anndata.anndata_adaptor import AnndataAdaptor
+from server.common.errors import FilterError
+from server.common.data_locator import DataLocator
"""
-Test the scanpy engine using the pbmc3k data set.
+Test the anndata adaptor using the pbmc3k data set.
"""
@@ -29,7 +29,7 @@ Test the scanpy engine using the pbmc3k data set.
("test/test_datasets/pbmc3k-CSR-gz.h5ad", True),
],
)
-class EngineTest(unittest.TestCase):
+class AdaptorTest(unittest.TestCase):
def setUp(self):
args = {
"layout": ["umap"],
@@ -37,10 +37,9 @@ class EngineTest(unittest.TestCase):
"obs_names": None,
"var_names": None,
"diffexp_lfc_cutoff": 0.01,
- "layout_file": None,
- "backed": self.backed,
+ "anndata_backed": self.backed,
}
- self.data = ScanpyEngine(DataLocator(self.data_locator), args)
+ self.data = AnndataAdaptor(DataLocator(self.data_locator), args)
def test_init(self):
self.assertEqual(self.data.cell_count, 2638)
@@ -56,7 +55,7 @@ class EngineTest(unittest.TestCase):
self.assertIn(var_index_col_name, self.data.data.var)
self.assertEqual(list(self.data.data.var.index), list(range(1838)))
- @pytest.mark.filterwarnings("ignore:Scanpy data matrix")
+ @pytest.mark.filterwarnings("ignore:Anndata data matrix")
def test_data_type(self):
# don't run the test on the more exotic data types, as they don't
# support the astype() interface (used by this test, but not underlying app)
@@ -98,9 +97,7 @@ class EngineTest(unittest.TestCase):
self.data._create_schema()
def test_config(self):
- self.assertEqual(
- self.data.features["layout"]["obs"], {"available": True, "interactiveLimit": 50000},
- )
+ self.assertEqual(self.data.get_features()["layout_obs"].available, True)
def test_layout(self):
fbs = self.data.layout_to_fbs_matrix()
@@ -142,10 +139,6 @@ class EngineTest(unittest.TestCase):
self.assertEqual(annotations["n_rows"], 1838)
self.assertEqual(annotations["n_cols"], 1)
- def test_annotation_put(self):
- with self.assertRaises(DisabledFeatureError):
- self.data.annotation_put_fbs(None, "obs")
-
def test_diffexp_topN(self):
f1 = {"filter": {"obs": {"index": [[0, 500]]}}}
f2 = {"filter": {"obs": {"index": [[500, 1000]]}}}
diff --git a/server/test/test_scanpy_engine_data_load.py b/server/test/test_anndata_adaptor_data_load.py
similarity index 57%
rename from server/test/test_scanpy_engine_data_load.py
rename to server/test/test_anndata_adaptor_data_load.py
index 16e5693b..2a7b1fb8 100644
--- a/server/test/test_scanpy_engine_data_load.py
+++ b/server/test/test_anndata_adaptor_data_load.py
@@ -1,49 +1,20 @@
import unittest
import json
-from server.app.scanpy_engine.scanpy_engine import ScanpyEngine
-from server.app.util.errors import DriverError
-from server.app.util.data_locator import DataLocator
+from server.data_anndata.anndata_adaptor import AnndataAdaptor
+from server.common.data_locator import DataLocator
-class DataLoadEngineTest(unittest.TestCase):
+class DataLoadAdaptorTest(unittest.TestCase):
"""
Test file loading, including deferred loading/update.
"""
def setUp(self):
self.data_file = DataLocator("../example-dataset/pbmc3k.h5ad")
- self.data = ScanpyEngine()
-
- def test_init(self):
- self.assertIsNone(self.data.data)
-
- def test_delayed_load_args(self):
- args = {
- "layout": ["tsne"],
- "max_category_items": 1000,
- "obs_names": "foo",
- "var_names": "bar",
- "diffexp_lfc_cutoff": 0.1,
- "annotations": False,
- "annotations_file": None,
- "annotations_output_dir": None,
- "backed": False,
- "diffexp_may_be_slow": False,
- "disable_diffexp": False,
- "annotations_cell_ontology_enabled": False,
- "annotations_cell_ontology_obopath": None,
- "annotations_cell_ontology_terms": None,
- }
- self.data.update(args=args)
- self.assertEqual(args, self.data.config)
-
- def test_requires_data(self):
- with self.assertRaises(DriverError):
- self.data._create_schema()
+ self.data = AnndataAdaptor(self.data_file)
def test_delayed_load_data(self):
- self.data.update(data_locator=self.data_file)
self.data._create_schema()
self.assertEqual(self.data.cell_count, 2638)
self.assertEqual(self.data.gene_count, 1838)
@@ -51,7 +22,6 @@ class DataLoadEngineTest(unittest.TestCase):
self.assertTrue(self.data.data.X[0, 0] - -0.171_469_51 < epsilon)
def test_diffexp_topN(self):
- self.data.update(data_locator=self.data_file)
f1 = {"filter": {"obs": {"index": [[0, 500]]}}}
f2 = {"filter": {"obs": {"index": [[500, 1000]]}}}
result = json.loads(self.data.diffexp_topN(f1["filter"], f2["filter"]))
@@ -60,7 +30,7 @@ class DataLoadEngineTest(unittest.TestCase):
self.assertEqual(len(result), 20)
-class DataLocatorEngineTest(unittest.TestCase):
+class DataLocatorAdaptorTest(unittest.TestCase):
"""
Test various types of data locators we expect to consume
"""
@@ -82,17 +52,17 @@ class DataLocatorEngineTest(unittest.TestCase):
def test_posix_file(self):
locator = DataLocator("../example-dataset/pbmc3k.h5ad")
- data = ScanpyEngine(locator, self.args)
+ data = AnndataAdaptor(locator, self.args)
self.stdAsserts(data)
def test_url_https(self):
url = "https://raw.githubusercontent.com/chanzuckerberg/cellxgene/master/example-dataset/pbmc3k.h5ad"
locator = DataLocator(url)
- data = ScanpyEngine(locator, self.args)
+ data = AnndataAdaptor(locator, self.args)
self.stdAsserts(data)
def test_url_http(self):
url = "http://raw.githubusercontent.com/chanzuckerberg/cellxgene/master/example-dataset/pbmc3k.h5ad"
locator = DataLocator(url)
- data = ScanpyEngine(locator, self.args)
+ data = AnndataAdaptor(locator, self.args)
self.stdAsserts(data)
diff --git a/server/test/test_api.py b/server/test/test_api.py
index d1482ba1..96132eac 100644
--- a/server/test/test_api.py
+++ b/server/test/test_api.py
@@ -7,50 +7,18 @@ import requests
import decode_fbs
-LOCAL_URL = "http://127.0.0.1:5005/"
-VERSION = "v0.2"
-URL_BASE = f"{LOCAL_URL}api/{VERSION}/"
BAD_FILTER = {"filter": {"obs": {"annotation_value": [{"name": "xyz"}]}}}
-class EndPoints(unittest.TestCase):
- """Test Case for endpoints"""
-
- @classmethod
- def setUpClass(cls):
- cls.ps = Popen(
- [
- "cellxgene",
- "--no-upgrade-check",
- "launch",
- "../example-dataset/pbmc3k.h5ad",
- "--verbose",
- "--port",
- "5005",
- ]
- )
- session = requests.Session()
- for i in range(90):
- try:
- result = session.get(f"{URL_BASE}schema")
- cls.schema = result.json()
- except requests.exceptions.ConnectionError:
- time.sleep(1)
-
- @classmethod
- def tearDownClass(cls):
- try:
- cls.ps.terminate()
- except ProcessLookupError:
- pass
+class EndPoints(object):
def setUp(self):
self.session = requests.Session()
def test_initialize(self):
endpoint = "schema"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.headers["Content-Type"], "application/json")
@@ -61,18 +29,18 @@ class EndPoints(unittest.TestCase):
def test_config(self):
endpoint = "config"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.headers["Content-Type"], "application/json")
result_data = result.json()
self.assertIn("library_versions", result_data["config"])
self.assertEqual(result_data["config"]["displayNames"]["dataset"], "pbmc3k")
- self.assertEqual(len(result_data["config"]["features"]), 4)
+ self.assertEqual(len(result_data["config"]["features"]), 3)
def test_get_layout_fbs(self):
endpoint = "layout/obs"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
header = {"Accept": "application/octet-stream"}
result = self.session.get(url, headers=header)
self.assertEqual(result.status_code, HTTPStatus.OK)
@@ -81,22 +49,22 @@ class EndPoints(unittest.TestCase):
self.assertEqual(df["n_rows"], 2638)
self.assertEqual(df["n_cols"], 8)
self.assertIsNotNone(df["columns"])
- self.assertListEqual(
- df["col_idx"],
- ["pca_0", "pca_1", "tsne_0", "tsne_1", "umap_0", "umap_1", "draw_graph_fr_0", "draw_graph_fr_1"],
+ self.assertSetEqual(
+ set(df["col_idx"]),
+ set(["pca_0", "pca_1", "tsne_0", "tsne_1", "umap_0", "umap_1", "draw_graph_fr_0", "draw_graph_fr_1"]),
)
self.assertIsNone(df["row_idx"])
self.assertEqual(len(df["columns"]), df["n_cols"])
def test_bad_filter(self):
endpoint = "data/var"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
result = self.session.put(url, json=BAD_FILTER)
self.assertEqual(result.status_code, HTTPStatus.BAD_REQUEST)
def test_get_annotations_obs_fbs(self):
endpoint = "annotations/obs"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
header = {"Accept": "application/octet-stream"}
result = self.session.get(url, headers=header)
self.assertEqual(result.status_code, HTTPStatus.OK)
@@ -114,7 +82,7 @@ class EndPoints(unittest.TestCase):
def test_get_annotations_obs_keys_fbs(self):
endpoint = "annotations/obs"
query = "annotation-name=n_genes&annotation-name=percent_mito"
- url = f"{URL_BASE}{endpoint}?{query}"
+ url = f"{self.URL_BASE}{endpoint}?{query}"
header = {"Accept": "application/octet-stream"}
result = self.session.get(url, headers=header)
self.assertEqual(result.status_code, HTTPStatus.OK)
@@ -131,13 +99,13 @@ class EndPoints(unittest.TestCase):
def test_get_annotations_obs_error(self):
endpoint = "annotations/obs"
query = "annotation-name=notakey"
- url = f"{URL_BASE}{endpoint}?{query}"
+ url = f"{self.URL_BASE}{endpoint}?{query}"
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.BAD_REQUEST)
def test_diff_exp(self):
endpoint = "diffexp/obs"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
params = {
"mode": "topN",
"set1": {"filter": {"obs": {"annotation_value": [{"name": "louvain", "values": ["NK cells"]}]}}},
@@ -152,7 +120,7 @@ class EndPoints(unittest.TestCase):
def test_diff_exp_indices(self):
endpoint = "diffexp/obs"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
params = {
"mode": "topN",
"count": 10,
@@ -167,7 +135,7 @@ class EndPoints(unittest.TestCase):
def test_get_annotations_var_fbs(self):
endpoint = "annotations/var"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
header = {"Accept": "application/octet-stream"}
result = self.session.get(url, headers=header)
self.assertEqual(result.status_code, HTTPStatus.OK)
@@ -185,7 +153,7 @@ class EndPoints(unittest.TestCase):
def test_get_annotations_var_keys_fbs(self):
endpoint = "annotations/var"
query = "annotation-name=n_cells"
- url = f"{URL_BASE}{endpoint}?{query}"
+ url = f"{self.URL_BASE}{endpoint}?{query}"
header = {"Accept": "application/octet-stream"}
result = self.session.get(url, headers=header)
self.assertEqual(result.status_code, HTTPStatus.OK)
@@ -202,27 +170,27 @@ class EndPoints(unittest.TestCase):
def test_get_annotations_var_error(self):
endpoint = "annotations/var"
query = "annotation-name=notakey"
- url = f"{URL_BASE}{endpoint}?{query}"
+ url = f"{self.URL_BASE}{endpoint}?{query}"
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.BAD_REQUEST)
def test_data_mimetype_error(self):
endpoint = f"data/var"
header = {"Accept": "xxx"}
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
result = self.session.put(url, headers=header)
self.assertEqual(result.status_code, HTTPStatus.NOT_ACCEPTABLE)
def test_fbs_default(self):
endpoint = f"data/var"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
result = self.session.put(url)
self.assertEqual(result.status_code, HTTPStatus.OK)
self.assertEqual(result.headers["Content-Type"], "application/octet-stream")
def test_data_put_fbs(self):
endpoint = f"data/var"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
header = {"Accept": "application/octet-stream"}
result = self.session.put(url, headers=header)
self.assertEqual(result.status_code, HTTPStatus.OK)
@@ -237,7 +205,7 @@ class EndPoints(unittest.TestCase):
def test_data_put_filter_fbs(self):
endpoint = f"data/var"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
header = {"Accept": "application/octet-stream"}
filter = {"filter": {"var": {"index": [0, 1, 4]}}}
result = self.session.put(url, headers=header, json=filter)
@@ -254,7 +222,7 @@ class EndPoints(unittest.TestCase):
def test_data_put_single_var(self):
endpoint = f"data/var"
- url = f"{URL_BASE}{endpoint}"
+ url = f"{self.URL_BASE}{endpoint}"
header = {"Accept": "application/octet-stream"}
index_col_name = self.schema["schema"]["annotations"]["var"]["index"]
var_filter = {"filter": {"var": {"annotation_value": [{"name": index_col_name, "values": ["RER1"]}]}}}
@@ -268,6 +236,80 @@ class EndPoints(unittest.TestCase):
def test_static(self):
endpoint = "static"
file = "js/service-worker.js"
- url = f"{LOCAL_URL}{endpoint}/{file}"
+ url = f"{self.LOCAL_URL}{endpoint}/{file}"
result = self.session.get(url)
self.assertEqual(result.status_code, HTTPStatus.OK)
+
+
+class EndPointsAnndata(unittest.TestCase, EndPoints):
+ """Test Case for endpoints"""
+
+ PORT = 5010
+ LOCAL_URL = f"http://127.0.0.1:{PORT}/"
+ VERSION = "v0.2"
+ URL_BASE = f"{LOCAL_URL}api/{VERSION}/"
+
+ @classmethod
+ def setUpClass(cls):
+ cls.ps = Popen(
+ [
+ "cellxgene",
+ "--no-upgrade-check",
+ "launch",
+ "../example-dataset/pbmc3k.h5ad",
+ "--verbose",
+ "--port",
+ str(cls.PORT),
+ ]
+ )
+ cls.session = requests.Session()
+ for i in range(90):
+ try:
+ result = cls.session.get(f"{cls.URL_BASE}schema")
+ cls.schema = result.json()
+ except requests.exceptions.ConnectionError:
+ time.sleep(1)
+
+ @classmethod
+ def tearDownClass(cls):
+ try:
+ cls.ps.terminate()
+ except ProcessLookupError:
+ pass
+
+
+class EndPointsCxg(unittest.TestCase, EndPoints):
+ """Test Case for endpoints"""
+
+ PORT = 5011
+ LOCAL_URL = f"http://127.0.0.1:{PORT}/"
+ VERSION = "v0.2"
+ URL_BASE = f"{LOCAL_URL}api/{VERSION}/"
+
+ @classmethod
+ def setUpClass(cls):
+ cls.ps = Popen(
+ [
+ "cellxgene",
+ "--no-upgrade-check",
+ "launch",
+ "../example-dataset/pbmc3k.cxg",
+ "--verbose",
+ "--port",
+ str(cls.PORT),
+ ]
+ )
+ cls.session = requests.Session()
+ for i in range(90):
+ try:
+ result = cls.session.get(f"{cls.URL_BASE}schema")
+ cls.schema = result.json()
+ except requests.exceptions.ConnectionError:
+ time.sleep(1)
+
+ @classmethod
+ def tearDownClass(cls):
+ try:
+ cls.ps.terminate()
+ except ProcessLookupError:
+ pass
diff --git a/server/test/test_fbs.py b/server/test/test_fbs.py
index 36c3ec59..ef85c949 100644
--- a/server/test/test_fbs.py
+++ b/server/test/test_fbs.py
@@ -4,7 +4,7 @@ import numpy as np
from scipy import sparse
import decode_fbs
-from server.app.util.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
+from server.data_common.fbs.matrix import encode_matrix_fbs, decode_matrix_fbs
class FbsTests(unittest.TestCase):
diff --git a/server/test/test_nan_scanpy_engine.py b/server/test/test_nan_anndata_adaptor.py
similarity index 88%
rename from server/test/test_nan_scanpy_engine.py
rename to server/test/test_nan_anndata_adaptor.py
index 84fa255e..5d463b23 100644
--- a/server/test/test_nan_scanpy_engine.py
+++ b/server/test/test_nan_anndata_adaptor.py
@@ -5,9 +5,9 @@ import math
import decode_fbs
-from server.app.scanpy_engine.scanpy_engine import ScanpyEngine
-from server.app.util.errors import FilterError
-from server.app.util.data_locator import DataLocator
+from server.data_anndata.anndata_adaptor import AnndataAdaptor
+from server.common.errors import FilterError
+from server.common.data_locator import DataLocator
class NaNTest(unittest.TestCase):
@@ -21,12 +21,12 @@ class NaNTest(unittest.TestCase):
}
with warnings.catch_warnings():
warnings.simplefilter("ignore", category=UserWarning)
- self.data = ScanpyEngine(DataLocator("test/test_datasets/nan.h5ad"), self.args)
+ self.data = AnndataAdaptor(DataLocator("test/test_datasets/nan.h5ad"), self.args)
self.data._create_schema()
def test_load(self):
with self.assertWarns(UserWarning):
- ScanpyEngine(DataLocator("test/test_datasets/nan.h5ad"), self.args)
+ AnndataAdaptor(DataLocator("test/test_datasets/nan.h5ad"), self.args)
def test_init(self):
self.assertEqual(self.data.cell_count, 100)
diff --git a/server/test/test_writable_annotation.py b/server/test/test_writable_annotation.py
index 8cf0c616..16d10ba2 100644
--- a/server/test/test_writable_annotation.py
+++ b/server/test/test_writable_annotation.py
@@ -8,9 +8,11 @@ import shutil
import numpy as np
import pandas as pd
-from server.app.scanpy_engine.scanpy_engine import ScanpyEngine
-from server.app.util.fbs.matrix import encode_matrix_fbs
-from server.app.util.data_locator import DataLocator
+from server.data_anndata.anndata_adaptor import AnndataAdaptor
+from server.data_common.fbs.matrix import encode_matrix_fbs
+from server.common.data_locator import DataLocator
+from server.common.annotations import AnnotationsLocalFile
+from server.common.rest import schema_get_helper, annotations_put_fbs_helper
class WritableAnnotationTest(unittest.TestCase):
@@ -23,11 +25,11 @@ class WritableAnnotationTest(unittest.TestCase):
"obs_names": None,
"var_names": None,
"diffexp_lfc_cutoff": 0.01,
- "annotations": True,
- "annotations_file": self.annotations_file,
- "annotations_output_dir": None,
}
- self.data = ScanpyEngine(DataLocator("../example-dataset/pbmc3k.h5ad"), args)
+ fname = "../example-dataset/pbmc3k.h5ad"
+ data_locator = DataLocator(fname)
+ self.data = AnndataAdaptor(data_locator, args)
+ self.annotations = AnnotationsLocalFile(None, self.annotations_file)
def tearDown(self):
shutil.rmtree(self.tmpDir)
@@ -36,18 +38,19 @@ class WritableAnnotationTest(unittest.TestCase):
df = pd.DataFrame(data)
return encode_matrix_fbs(matrix=df, row_idx=None, col_idx=df.columns)
+ def annotation_put_fbs(self, fbs):
+ annotations_put_fbs_helper(self.data, self.annotations, fbs)
+ res = json.dumps({"status": "OK"})
+ return res
+
def test_error_checks(self):
# verify that the expected errors are generated
n_rows = self.data.data.obs.shape[0]
fbs_bad = self.make_fbs({"louvain": pd.Series(["undefined" for l in range(0, n_rows)], dtype="category")})
- # ensure attempt to change VAR annotation
- with self.assertRaises(ValueError):
- self.data.annotation_put_fbs("var", fbs_bad)
-
# ensure we catch attempt to overwrite non-writable data
with self.assertRaises(KeyError):
- self.data.annotation_put_fbs("obs", fbs_bad)
+ self.annotation_put_fbs(fbs_bad)
def test_write_to_file(self):
# verify the file is written as expected
@@ -58,7 +61,7 @@ class WritableAnnotationTest(unittest.TestCase):
"cat_B": pd.Series(["label_B" for l in range(0, n_rows)], dtype="category"),
}
)
- res = self.data.annotation_put_fbs("obs", fbs)
+ res = self.annotation_put_fbs(fbs)
self.assertEqual(res, json.dumps({"status": "OK"}))
self.assertTrue(path.exists(self.annotations_file))
df = pd.read_csv(self.annotations_file, index_col=0, header=0, comment="#")
@@ -75,7 +78,7 @@ class WritableAnnotationTest(unittest.TestCase):
"cat_C": pd.Series(["label_C" for l in range(0, n_rows)], dtype="category"),
}
)
- res = self.data.annotation_put_fbs("obs", fbs)
+ res = self.annotation_put_fbs(fbs)
self.assertEqual(res, json.dumps({"status": "OK"}))
self.assertTrue(path.exists(self.annotations_file))
df = pd.read_csv(self.annotations_file, index_col=0, header=0, comment="#")
@@ -100,7 +103,7 @@ class WritableAnnotationTest(unittest.TestCase):
}
)
for i in range(0, 11):
- res = self.data.annotation_put_fbs("obs", fbs)
+ res = self.annotation_put_fbs(fbs)
self.assertEqual(res, json.dumps({"status": "OK"}))
name, ext = path.splitext(self.annotations_file)
@@ -122,12 +125,13 @@ class WritableAnnotationTest(unittest.TestCase):
)
# put
- res = self.data.annotation_put_fbs("obs", fbs)
+ res = self.annotation_put_fbs(fbs)
self.assertEqual(res, json.dumps({"status": "OK"}))
# get
- fbsAll = self.data.annotation_to_fbs_matrix("obs")
- schema = self.data.get_schema()
+ labels = self.annotations.read_labels(None)
+ fbsAll = self.data.annotation_to_fbs_matrix("obs", None, labels)
+ schema = schema_get_helper(self.data, self.annotations)
annotations = decode_fbs.decode_matrix_FBS(fbsAll)
obs_index_col_name = schema["annotations"]["obs"]["index"]
self.assertEqual(annotations["n_rows"], n_rows)
diff --git a/server/utils/constants.py b/server/utils/constants.py
deleted file mode 100644
index 48f05329..00000000
--- a/server/utils/constants.py
+++ /dev/null
@@ -1 +0,0 @@
-MODES = ["umap", "tsne", "draw_graph_fa", "draw_graph_fr", "diffmap", "phate"]
diff --git a/server/utils/utils.py b/server/utils/utils.py
deleted file mode 100644
index 417aeb7b..00000000
--- a/server/utils/utils.py
+++ /dev/null
@@ -1,35 +0,0 @@
-import contextlib
-import errno
-import socket
-
-
-def find_available_port(host, port=5005):
- """
- Helper method to find open port on host. Tries 5000 ports incremented from the specified port
- """
- # Takes approx 2 seconds to do a scan of 5000 ports on my laptop
- num_ports_to_try = 5000
- for port_to_try in range(port, port + num_ports_to_try):
- if is_port_available(host, port_to_try):
- return port_to_try
- raise socket.error(errno.EADDRINUSE, f"No port in range {port} - {port + num_ports_to_try - 1} available.")
-
-
-def is_port_available(host, port):
- is_available = False
- with contextlib.closing(socket.socket(socket.AF_INET, socket.SOCK_STREAM)) as s:
- try:
- s.bind((host, port))
- is_available = True
- except socket.error:
- pass
- return is_available
-
-
-def sort_options(command):
- """
- Helper for the click options - will sort options in a command, and can
- be used as a decorator.
- """
- command.params.sort(key=lambda p: p.name)
- return command
diff --git a/setup.cfg b/setup.cfg
index 75c9a1bf..de62c223 100644
--- a/setup.cfg
+++ b/setup.cfg
@@ -2,5 +2,4 @@
max-line-length = 120
ignore = E203, W503
exclude =
- server/app/util/fbs/NetEncoding/,
- server/gui/cellxgene_rc.py
+ server/data_common/fbs/NetEncoding/
diff --git a/setup.py b/setup.py
index 35093f85..afa6b7f2 100644
--- a/setup.py
+++ b/setup.py
@@ -39,5 +39,5 @@ setup(
"Topic :: Scientific/Engineering :: Bio-Informatics",
],
entry_points={"console_scripts": ["cellxgene = server.cli.cli:cli"]},
- extras_require=dict(prepare=requirements_prepare, gui=["PySide2>=5.12.3", "cefpython3>=66", "requests"]),
+ extras_require=dict(prepare=requirements_prepare),
)