mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-30 23:38:11 +08:00
large file size guardrails (#763)
* large file guardrails * fix lint * PR review * remove unused import * use standard slice for CSR * revert change
This commit is contained in:
@@ -2,7 +2,7 @@ import warnings
|
||||
|
||||
import numpy as np
|
||||
from pandas.core.dtypes.dtypes import CategoricalDtype
|
||||
import scanpy as sc
|
||||
import anndata
|
||||
from scipy import sparse
|
||||
|
||||
from server.app.driver.driver import CXGDriver
|
||||
@@ -140,11 +140,10 @@ class ScanpyEngine(CXGDriver):
|
||||
self.schema["annotations"][ax].append(ann_schema)
|
||||
|
||||
def _load_data(self, data):
|
||||
# Based on benchmarking, cache=True has no impact on perf.
|
||||
# Note: as of current scanpy/anndata release, setting backed='r' will
|
||||
# result in an error. https://github.com/theislab/anndata/issues/79
|
||||
# as of AnnData 0.6.19, backed mode performs initial load fast, but at the
|
||||
# cost of significantly slower access to X data.
|
||||
try:
|
||||
self.data = sc.read(data, cache=True)
|
||||
self.data = anndata.read_h5ad(data)
|
||||
except ValueError:
|
||||
raise ScanpyFileError(
|
||||
"File must be in the .h5ad format. Please read "
|
||||
@@ -174,6 +173,11 @@ class ScanpyEngine(CXGDriver):
|
||||
|
||||
@requires_data
|
||||
def _validate_data_types(self):
|
||||
if sparse.isspmatrix(self.data.X) and not sparse.isspmatrix_csc(self.data.X):
|
||||
warnings.warn(
|
||||
f"Scanpy data matrix is sparse, but not a CSC (columnar) matrix. "
|
||||
f"Performance may be improved by using CSC."
|
||||
)
|
||||
if self.data.X.dtype != "float32":
|
||||
warnings.warn(
|
||||
f"Scanpy data matrix is in {self.data.X.dtype} format not float32. "
|
||||
|
||||
+16
-2
@@ -1,11 +1,12 @@
|
||||
import logging
|
||||
from os import devnull
|
||||
from os.path import splitext, basename
|
||||
from os.path import splitext, basename, getsize
|
||||
import sys
|
||||
import warnings
|
||||
import webbrowser
|
||||
|
||||
import click
|
||||
import psutil
|
||||
|
||||
from server.app.app import Server
|
||||
from server.app.util.errors import ScanpyFileError
|
||||
@@ -13,6 +14,10 @@ from server.app.util.utils import custom_format_warning
|
||||
from server.utils.constants import MODES
|
||||
|
||||
|
||||
# anything bigger than this will generate a special message
|
||||
BIG_FILE_SIZE_THRESHOLD = 100 * 2**20 # 100MB
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.argument("data", metavar="<data file>", type=click.Path(exists=True, file_okay=True, dir_okay=False))
|
||||
@click.option(
|
||||
@@ -148,7 +153,16 @@ security risk by including the --scripts flag. Make sure you trust the scripts t
|
||||
log = logging.getLogger("werkzeug")
|
||||
log.setLevel(logging.ERROR)
|
||||
|
||||
click.echo(f"[cellxgene] Loading data from {basename(data)}, this may take awhile...")
|
||||
file_size = getsize(data)
|
||||
|
||||
# if a big file, let the user know it may take a while to load.
|
||||
if file_size > BIG_FILE_SIZE_THRESHOLD:
|
||||
click.echo(f"[cellxgene] Loading data from {basename(data)}, this may take awhile...")
|
||||
else:
|
||||
click.echo(f"[cellxgene] Loading data from {basename(data)}.")
|
||||
# if file is larger than main memory, let the user know performance may suffer
|
||||
if file_size > .95 * psutil.virtual_memory().total:
|
||||
click.echo(f"[cellxgene] Warning: data file is larger than RAM - application may be very slow.")
|
||||
|
||||
# Fix for anaconda python. matplotlib typically expects python to be installed as a framework TKAgg is usually
|
||||
# available and fixes this issue. See https://matplotlib.org/faq/virtualenv_faq.html
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
anndata>=0.6.15
|
||||
click>=6.7
|
||||
Flask>=1.0.2
|
||||
Flask-Caching>=1.4.0
|
||||
@@ -8,6 +9,7 @@ flatbuffers>=1.10.0
|
||||
matplotlib>=2.2
|
||||
numpy>=1.15.2
|
||||
pandas>=0.23.1
|
||||
psutil>=5.6.2
|
||||
scanpy>=1.3.7
|
||||
scipy>=1.1.0
|
||||
scikit-learn>=0.19.1,!=0.20.0
|
||||
|
||||
Reference in New Issue
Block a user