mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-09 21:10:56 +08:00
large file size guardrails (#763)
* large file guardrails * fix lint * PR review * remove unused import * use standard slice for CSR * revert change
This commit is contained in:
@@ -2,7 +2,7 @@ import warnings
|
|||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
from pandas.core.dtypes.dtypes import CategoricalDtype
|
from pandas.core.dtypes.dtypes import CategoricalDtype
|
||||||
import scanpy as sc
|
import anndata
|
||||||
from scipy import sparse
|
from scipy import sparse
|
||||||
|
|
||||||
from server.app.driver.driver import CXGDriver
|
from server.app.driver.driver import CXGDriver
|
||||||
@@ -140,11 +140,10 @@ class ScanpyEngine(CXGDriver):
|
|||||||
self.schema["annotations"][ax].append(ann_schema)
|
self.schema["annotations"][ax].append(ann_schema)
|
||||||
|
|
||||||
def _load_data(self, data):
|
def _load_data(self, data):
|
||||||
# Based on benchmarking, cache=True has no impact on perf.
|
# as of AnnData 0.6.19, backed mode performs initial load fast, but at the
|
||||||
# Note: as of current scanpy/anndata release, setting backed='r' will
|
# cost of significantly slower access to X data.
|
||||||
# result in an error. https://github.com/theislab/anndata/issues/79
|
|
||||||
try:
|
try:
|
||||||
self.data = sc.read(data, cache=True)
|
self.data = anndata.read_h5ad(data)
|
||||||
except ValueError:
|
except ValueError:
|
||||||
raise ScanpyFileError(
|
raise ScanpyFileError(
|
||||||
"File must be in the .h5ad format. Please read "
|
"File must be in the .h5ad format. Please read "
|
||||||
@@ -174,6 +173,11 @@ class ScanpyEngine(CXGDriver):
|
|||||||
|
|
||||||
@requires_data
|
@requires_data
|
||||||
def _validate_data_types(self):
|
def _validate_data_types(self):
|
||||||
|
if sparse.isspmatrix(self.data.X) and not sparse.isspmatrix_csc(self.data.X):
|
||||||
|
warnings.warn(
|
||||||
|
f"Scanpy data matrix is sparse, but not a CSC (columnar) matrix. "
|
||||||
|
f"Performance may be improved by using CSC."
|
||||||
|
)
|
||||||
if self.data.X.dtype != "float32":
|
if self.data.X.dtype != "float32":
|
||||||
warnings.warn(
|
warnings.warn(
|
||||||
f"Scanpy data matrix is in {self.data.X.dtype} format not float32. "
|
f"Scanpy data matrix is in {self.data.X.dtype} format not float32. "
|
||||||
|
|||||||
+16
-2
@@ -1,11 +1,12 @@
|
|||||||
import logging
|
import logging
|
||||||
from os import devnull
|
from os import devnull
|
||||||
from os.path import splitext, basename
|
from os.path import splitext, basename, getsize
|
||||||
import sys
|
import sys
|
||||||
import warnings
|
import warnings
|
||||||
import webbrowser
|
import webbrowser
|
||||||
|
|
||||||
import click
|
import click
|
||||||
|
import psutil
|
||||||
|
|
||||||
from server.app.app import Server
|
from server.app.app import Server
|
||||||
from server.app.util.errors import ScanpyFileError
|
from server.app.util.errors import ScanpyFileError
|
||||||
@@ -13,6 +14,10 @@ from server.app.util.utils import custom_format_warning
|
|||||||
from server.utils.constants import MODES
|
from server.utils.constants import MODES
|
||||||
|
|
||||||
|
|
||||||
|
# anything bigger than this will generate a special message
|
||||||
|
BIG_FILE_SIZE_THRESHOLD = 100 * 2**20 # 100MB
|
||||||
|
|
||||||
|
|
||||||
@click.command()
|
@click.command()
|
||||||
@click.argument("data", metavar="<data file>", type=click.Path(exists=True, file_okay=True, dir_okay=False))
|
@click.argument("data", metavar="<data file>", type=click.Path(exists=True, file_okay=True, dir_okay=False))
|
||||||
@click.option(
|
@click.option(
|
||||||
@@ -148,7 +153,16 @@ security risk by including the --scripts flag. Make sure you trust the scripts t
|
|||||||
log = logging.getLogger("werkzeug")
|
log = logging.getLogger("werkzeug")
|
||||||
log.setLevel(logging.ERROR)
|
log.setLevel(logging.ERROR)
|
||||||
|
|
||||||
click.echo(f"[cellxgene] Loading data from {basename(data)}, this may take awhile...")
|
file_size = getsize(data)
|
||||||
|
|
||||||
|
# if a big file, let the user know it may take a while to load.
|
||||||
|
if file_size > BIG_FILE_SIZE_THRESHOLD:
|
||||||
|
click.echo(f"[cellxgene] Loading data from {basename(data)}, this may take awhile...")
|
||||||
|
else:
|
||||||
|
click.echo(f"[cellxgene] Loading data from {basename(data)}.")
|
||||||
|
# if file is larger than main memory, let the user know performance may suffer
|
||||||
|
if file_size > .95 * psutil.virtual_memory().total:
|
||||||
|
click.echo(f"[cellxgene] Warning: data file is larger than RAM - application may be very slow.")
|
||||||
|
|
||||||
# Fix for anaconda python. matplotlib typically expects python to be installed as a framework TKAgg is usually
|
# Fix for anaconda python. matplotlib typically expects python to be installed as a framework TKAgg is usually
|
||||||
# available and fixes this issue. See https://matplotlib.org/faq/virtualenv_faq.html
|
# available and fixes this issue. See https://matplotlib.org/faq/virtualenv_faq.html
|
||||||
|
|||||||
@@ -1,3 +1,4 @@
|
|||||||
|
anndata>=0.6.15
|
||||||
click>=6.7
|
click>=6.7
|
||||||
Flask>=1.0.2
|
Flask>=1.0.2
|
||||||
Flask-Caching>=1.4.0
|
Flask-Caching>=1.4.0
|
||||||
@@ -8,6 +9,7 @@ flatbuffers>=1.10.0
|
|||||||
matplotlib>=2.2
|
matplotlib>=2.2
|
||||||
numpy>=1.15.2
|
numpy>=1.15.2
|
||||||
pandas>=0.23.1
|
pandas>=0.23.1
|
||||||
|
psutil>=5.6.2
|
||||||
scanpy>=1.3.7
|
scanpy>=1.3.7
|
||||||
scipy>=1.1.0
|
scipy>=1.1.0
|
||||||
scikit-learn>=0.19.1,!=0.20.0
|
scikit-learn>=0.19.1,!=0.20.0
|
||||||
|
|||||||
Reference in New Issue
Block a user