9 Commits

Author SHA1 Message Date
Alok Saldanha
6bcb594712 #59 add temporary workaround for jinja 2022-03-14 22:58:06 -04:00
Alok Saldanha
f9ed4c4047 #59 document S3_ENABLE_LISTINGS_CACHE 2022-03-14 22:36:42 -04:00
Alok Saldanha
a9753c4101 #59 change s3 cache variable from S3_DISABLE_LISTINGS_CACHE to S3_ENABLE_LISTINGS_CACHE 2022-03-14 22:36:42 -04:00
Alok Saldanha
fd48920c5b #59 add refresh query param to force refresh of S3 cache 2022-03-14 21:44:05 -04:00
Alok Saldanha
09db93b2b5 Merge pull request #62 from arogozhnikov/patch-1
Force reload of s3 file structure on every request
2022-03-14 20:45:20 -04:00
Alex Rogozhnikov
3e3bd22512 add environment variable S3_DISABLE_LISTINGS_CACHE per Alok's request 2022-03-14 09:59:55 -07:00
Alex Rogozhnikov
893b2f1af1 remove listing cache at the level of fs 2022-03-11 03:20:06 -08:00
Alex Rogozhnikov
757487b772 Force reload folder on every request 2022-03-11 02:43:29 -08:00
Alok Saldanha
87a8dbfa78 #57 Reverted incorrect change to unit test 2021-12-21 13:59:30 -05:00
6 changed files with 32 additions and 9 deletions

View File

@@ -39,6 +39,7 @@ jobs:
conda env create -f environment.yml
eval "$(conda shell.bash hook)"
conda activate cellxgene-gateway
pip install markupsafe==2.0.1 # temporary workaround for jinja2-2.11.3 calling soft_unicode in markupsafe
python setup.py install
- name: Run tests

View File

@@ -76,6 +76,7 @@ Optional environment variables:
* `GATEWAY_EXTRA_SCRIPTS` - JSON array of script paths, will be embedded into each page and forwarded with `--scripts` to cellxgene server
* `GATEWAY_ENABLE_ANNOTATIONS` - Set to `true` or to `1` to enable cellxgene annotations.
* `GATEWAY_ENABLE_BACKED_MODE` - Set to `true` or to `1` to load AnnData in file-backed mode. This saves memory and speeds up launch time but may reduce overall performance.
* `S3_ENABLE_LISTINGS_CACHE` - Set to `true` or to `1` to cache listings of S3 folders for performance. Can be overridden by setting `filecrawl.html?refresh=true` query parameter.
If any of the following optional variables are set, [ProxyFix](https://werkzeug.palletsprojects.com/en/1.0.x/middleware/proxy_fix/) will be used.
* `PROXY_FIX_FOR` - Number of upstream proxies setting X-Forwarded-For

View File

@@ -22,9 +22,7 @@ from flask import (
send_from_directory,
url_for,
)
from flask_api import status
from werkzeug.middleware.proxy_fix import ProxyFix
from werkzeug.utils import secure_filename
from cellxgene_gateway import env, flask_util
from cellxgene_gateway.backend_cache import BackendCache

View File

@@ -7,9 +7,11 @@
# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
# the specific language governing permissions and limitations under the License.
from os.path import basename, dirname, join
import os
from os.path import basename, dirname
from typing import List
import flask
import s3fs
from cellxgene_gateway import dir_util
@@ -18,6 +20,10 @@ from cellxgene_gateway.items.item_source import ItemSource, LookupResult
from cellxgene_gateway.items.s3.s3item import S3Item
def truthy(val: str):
return val.lower() in ["true", "1"]
class S3ItemSource(ItemSource):
def __init__(
self,
@@ -28,7 +34,10 @@ class S3ItemSource(ItemSource):
annotation_file_suffix=".csv",
):
self._name = name
self.s3 = s3fs.S3FileSystem()
enable_cache = os.environ.get("S3_ENABLE_LISTINGS_CACHE", "false").lower()
assert enable_cache in ["0", "1", "false", "true"]
self.use_listings_cache = truthy(enable_cache)
self.s3 = s3fs.S3FileSystem(use_listings_cache=self.use_listings_cache)
if bucket.startswith("s3://"):
raise Exception(
f"Bucket name should not include s3:// prefix, got {bucket}"
@@ -67,6 +76,13 @@ class S3ItemSource(ItemSource):
item_tree = self.scan_directory("" if filter is None else filter)
return item_tree
@property
def refresh(self):
return (
truthy(flask.request.args.get("refresh", default="false"))
or not self.use_listings_cache
)
def scan_directory(self, directory_key="") -> dict:
url = self.url(directory_key)
@@ -75,7 +91,7 @@ class S3ItemSource(ItemSource):
s3key_map = dict(
(self.remove_bucket(filepath), "s3://" + filepath)
for filepath in sorted(self.s3.ls(url))
for filepath in sorted(self.s3.ls(url, refresh=self.refresh))
)
def is_annotation_dir(dir_s3key):
@@ -166,7 +182,9 @@ class S3ItemSource(ItemSource):
self.make_s3item_from_key(
basename(annotation), self.remove_bucket(annotation), True
)
for annotation in sorted(self.s3.ls(annotations_fullpath))
for annotation in sorted(
self.s3.ls(annotations_fullpath, refresh=self.refresh)
)
if annotation.endswith(self.annotation_file_suffix)
and self.s3.isfile("s3://" + annotation)
]

View File

@@ -24,7 +24,10 @@ class TestScanDirectory(unittest.TestCase):
)
@patch("s3fs.S3FileSystem")
def test__GIVEN_multilevel_bucket_THEN_properly_recurses_suburls(self, s3func):
@patch("flask.request")
def test__GIVEN_multilevel_bucket_THEN_properly_recurses_suburls(
self, requestMock, s3func
):
class S3Mock:
def exists(path):
if path in [
@@ -38,7 +41,8 @@ class TestScanDirectory(unittest.TestCase):
return True
raise Exception("exists called with " + path)
def ls(path):
def ls(path, refresh):
assert refresh == True
if path == "s3://my-bucket/":
return [
"my-bucket/lvl1",
@@ -79,6 +83,7 @@ class TestScanDirectory(unittest.TestCase):
raise Exception("isfile called with " + path)
s3func.return_value = S3Mock
requestMock.args.get.return_value = "true"
source = S3ItemSource("my-bucket")
tree = source.scan_directory()

View File

@@ -33,7 +33,7 @@ class TestSubprocessBackend(unittest.TestCase):
backend.launch(cellxgene_loc, scripts, entry)
popen.assert_called_once_with(
[
"yes | /some/cellxgene launch /tmp/czi/pbmc3k.h5ad --port 8000 --host 127.0.0.1 --disable-annotations --disable-gene-sets-save --scripts http://example.com/script.js --scripts http://example.com/script2.js"
"yes | /some/cellxgene launch /tmp/czi/pbmc3k.h5ad --port 8000 --host 127.0.0.1 --disable-annotations --scripts http://example.com/script.js --scripts http://example.com/script2.js"
],
shell=True,
stderr=-1,