large file size guardrails (#763)

* large file guardrails

* fix lint

* PR review

* remove unused import

* use standard slice for CSR

* revert change
This commit is contained in:
Bruce Martin
2019-05-13 18:13:16 -07:00
committed by GitHub
parent 7adac5d004
commit b9a1e30652
3 changed files with 27 additions and 7 deletions
+9 -5
View File
@@ -2,7 +2,7 @@ import warnings
import numpy as np import numpy as np
from pandas.core.dtypes.dtypes import CategoricalDtype from pandas.core.dtypes.dtypes import CategoricalDtype
import scanpy as sc import anndata
from scipy import sparse from scipy import sparse
from server.app.driver.driver import CXGDriver from server.app.driver.driver import CXGDriver
@@ -140,11 +140,10 @@ class ScanpyEngine(CXGDriver):
self.schema["annotations"][ax].append(ann_schema) self.schema["annotations"][ax].append(ann_schema)
def _load_data(self, data): def _load_data(self, data):
# Based on benchmarking, cache=True has no impact on perf. # as of AnnData 0.6.19, backed mode performs initial load fast, but at the
# Note: as of current scanpy/anndata release, setting backed='r' will # cost of significantly slower access to X data.
# result in an error. https://github.com/theislab/anndata/issues/79
try: try:
self.data = sc.read(data, cache=True) self.data = anndata.read_h5ad(data)
except ValueError: except ValueError:
raise ScanpyFileError( raise ScanpyFileError(
"File must be in the .h5ad format. Please read " "File must be in the .h5ad format. Please read "
@@ -174,6 +173,11 @@ class ScanpyEngine(CXGDriver):
@requires_data @requires_data
def _validate_data_types(self): def _validate_data_types(self):
if sparse.isspmatrix(self.data.X) and not sparse.isspmatrix_csc(self.data.X):
warnings.warn(
f"Scanpy data matrix is sparse, but not a CSC (columnar) matrix. "
f"Performance may be improved by using CSC."
)
if self.data.X.dtype != "float32": if self.data.X.dtype != "float32":
warnings.warn( warnings.warn(
f"Scanpy data matrix is in {self.data.X.dtype} format not float32. " f"Scanpy data matrix is in {self.data.X.dtype} format not float32. "
+16 -2
View File
@@ -1,11 +1,12 @@
import logging import logging
from os import devnull from os import devnull
from os.path import splitext, basename from os.path import splitext, basename, getsize
import sys import sys
import warnings import warnings
import webbrowser import webbrowser
import click import click
import psutil
from server.app.app import Server from server.app.app import Server
from server.app.util.errors import ScanpyFileError from server.app.util.errors import ScanpyFileError
@@ -13,6 +14,10 @@ from server.app.util.utils import custom_format_warning
from server.utils.constants import MODES from server.utils.constants import MODES
# anything bigger than this will generate a special message
BIG_FILE_SIZE_THRESHOLD = 100 * 2**20 # 100MB
@click.command() @click.command()
@click.argument("data", metavar="<data file>", type=click.Path(exists=True, file_okay=True, dir_okay=False)) @click.argument("data", metavar="<data file>", type=click.Path(exists=True, file_okay=True, dir_okay=False))
@click.option( @click.option(
@@ -148,7 +153,16 @@ security risk by including the --scripts flag. Make sure you trust the scripts t
log = logging.getLogger("werkzeug") log = logging.getLogger("werkzeug")
log.setLevel(logging.ERROR) log.setLevel(logging.ERROR)
click.echo(f"[cellxgene] Loading data from {basename(data)}, this may take awhile...") file_size = getsize(data)
# if a big file, let the user know it may take a while to load.
if file_size > BIG_FILE_SIZE_THRESHOLD:
click.echo(f"[cellxgene] Loading data from {basename(data)}, this may take awhile...")
else:
click.echo(f"[cellxgene] Loading data from {basename(data)}.")
# if file is larger than main memory, let the user know performance may suffer
if file_size > .95 * psutil.virtual_memory().total:
click.echo(f"[cellxgene] Warning: data file is larger than RAM - application may be very slow.")
# Fix for anaconda python. matplotlib typically expects python to be installed as a framework TKAgg is usually # Fix for anaconda python. matplotlib typically expects python to be installed as a framework TKAgg is usually
# available and fixes this issue. See https://matplotlib.org/faq/virtualenv_faq.html # available and fixes this issue. See https://matplotlib.org/faq/virtualenv_faq.html
+2
View File
@@ -1,3 +1,4 @@
anndata>=0.6.15
click>=6.7 click>=6.7
Flask>=1.0.2 Flask>=1.0.2
Flask-Caching>=1.4.0 Flask-Caching>=1.4.0
@@ -8,6 +9,7 @@ flatbuffers>=1.10.0
matplotlib>=2.2 matplotlib>=2.2
numpy>=1.15.2 numpy>=1.15.2
pandas>=0.23.1 pandas>=0.23.1
psutil>=5.6.2
scanpy>=1.3.7 scanpy>=1.3.7
scipy>=1.1.0 scipy>=1.1.0
scikit-learn>=0.19.1,!=0.20.0 scikit-learn>=0.19.1,!=0.20.0