large file size guardrails (#763)

* large file guardrails

* fix lint

* PR review

* remove unused import

* use standard slice for CSR

* revert change
This commit is contained in:
Bruce Martin
2019-05-13 18:13:16 -07:00
committed by GitHub
parent 7adac5d004
commit b9a1e30652
3 changed files with 27 additions and 7 deletions
+9 -5
View File
@@ -2,7 +2,7 @@ import warnings
import numpy as np
from pandas.core.dtypes.dtypes import CategoricalDtype
import scanpy as sc
import anndata
from scipy import sparse
from server.app.driver.driver import CXGDriver
@@ -140,11 +140,10 @@ class ScanpyEngine(CXGDriver):
self.schema["annotations"][ax].append(ann_schema)
def _load_data(self, data):
# Based on benchmarking, cache=True has no impact on perf.
# Note: as of current scanpy/anndata release, setting backed='r' will
# result in an error. https://github.com/theislab/anndata/issues/79
# as of AnnData 0.6.19, backed mode performs initial load fast, but at the
# cost of significantly slower access to X data.
try:
self.data = sc.read(data, cache=True)
self.data = anndata.read_h5ad(data)
except ValueError:
raise ScanpyFileError(
"File must be in the .h5ad format. Please read "
@@ -174,6 +173,11 @@ class ScanpyEngine(CXGDriver):
@requires_data
def _validate_data_types(self):
if sparse.isspmatrix(self.data.X) and not sparse.isspmatrix_csc(self.data.X):
warnings.warn(
f"Scanpy data matrix is sparse, but not a CSC (columnar) matrix. "
f"Performance may be improved by using CSC."
)
if self.data.X.dtype != "float32":
warnings.warn(
f"Scanpy data matrix is in {self.data.X.dtype} format not float32. "
+16 -2
View File
@@ -1,11 +1,12 @@
import logging
from os import devnull
from os.path import splitext, basename
from os.path import splitext, basename, getsize
import sys
import warnings
import webbrowser
import click
import psutil
from server.app.app import Server
from server.app.util.errors import ScanpyFileError
@@ -13,6 +14,10 @@ from server.app.util.utils import custom_format_warning
from server.utils.constants import MODES
# anything bigger than this will generate a special message
BIG_FILE_SIZE_THRESHOLD = 100 * 2**20 # 100MB
@click.command()
@click.argument("data", metavar="<data file>", type=click.Path(exists=True, file_okay=True, dir_okay=False))
@click.option(
@@ -148,7 +153,16 @@ security risk by including the --scripts flag. Make sure you trust the scripts t
log = logging.getLogger("werkzeug")
log.setLevel(logging.ERROR)
click.echo(f"[cellxgene] Loading data from {basename(data)}, this may take awhile...")
file_size = getsize(data)
# if a big file, let the user know it may take a while to load.
if file_size > BIG_FILE_SIZE_THRESHOLD:
click.echo(f"[cellxgene] Loading data from {basename(data)}, this may take awhile...")
else:
click.echo(f"[cellxgene] Loading data from {basename(data)}.")
# if file is larger than main memory, let the user know performance may suffer
if file_size > .95 * psutil.virtual_memory().total:
click.echo(f"[cellxgene] Warning: data file is larger than RAM - application may be very slow.")
# Fix for anaconda python. matplotlib typically expects python to be installed as a framework TKAgg is usually
# available and fixes this issue. See https://matplotlib.org/faq/virtualenv_faq.html
+2
View File
@@ -1,3 +1,4 @@
anndata>=0.6.15
click>=6.7
Flask>=1.0.2
Flask-Caching>=1.4.0
@@ -8,6 +9,7 @@ flatbuffers>=1.10.0
matplotlib>=2.2
numpy>=1.15.2
pandas>=0.23.1
psutil>=5.6.2
scanpy>=1.3.7
scipy>=1.1.0
scikit-learn>=0.19.1,!=0.20.0