Compare commits

...
13 Commits
12 changed files with 71 additions and 13 deletions
+1
View File
@@ -39,6 +39,7 @@ jobs:
conda env create -f environment.yml
eval "$(conda shell.bash hook)"
conda activate cellxgene-gateway
pip install markupsafe==2.0.1 # temporary workaround for jinja2-2.11.3 calling soft_unicode in markupsafe
python setup.py install
- name: Run tests
+4
View File
@@ -1,3 +1,7 @@
# 0.3.8
* Fixed bug #57 affecting deeply nested subdirectory listing
# 0.3.7
* added back /metadata/ip_address endpoint
+1
View File
@@ -76,6 +76,7 @@ Optional environment variables:
* `GATEWAY_EXTRA_SCRIPTS` - JSON array of script paths, will be embedded into each page and forwarded with `--scripts` to cellxgene server
* `GATEWAY_ENABLE_ANNOTATIONS` - Set to `true` or to `1` to enable cellxgene annotations.
* `GATEWAY_ENABLE_BACKED_MODE` - Set to `true` or to `1` to load AnnData in file-backed mode. This saves memory and speeds up launch time but may reduce overall performance.
* `S3_ENABLE_LISTINGS_CACHE` - Set to `true` or to `1` to cache listings of S3 folders for performance. Can be overridden by setting `filecrawl.html?refresh=true` query parameter.
If any of the following optional variables are set, [ProxyFix](https://werkzeug.palletsprojects.com/en/1.0.x/middleware/proxy_fix/) will be used.
* `PROXY_FIX_FOR` - Number of upstream proxies setting X-Forwarded-For
+1 -1
View File
@@ -7,4 +7,4 @@
# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
# the specific language governing permissions and limitations under the License.
__version__ = "0.3.7"
__version__ = "0.3.8"
+1 -1
View File
@@ -54,7 +54,7 @@ def render_item_tree(item_tree, item_source):
if item_tree.descriptor:
descriptor = item_tree.descriptor.lstrip("/")
url = f"/filecrawl/{descriptor}?source={item_source.name}"
name = descriptor.rsplit("/")[1] if descriptor.find("/") >= 0 else descriptor
name = descriptor.rsplit("/", 1)[-1]
return f"<li><a href='{url}'>{name}</a>{html}</li>"
else:
return html
+4 -3
View File
@@ -22,9 +22,7 @@ from flask import (
send_from_directory,
url_for,
)
from flask_api import status
from werkzeug.middleware.proxy_fix import ProxyFix
from werkzeug.utils import secure_filename
from cellxgene_gateway import env, flask_util
from cellxgene_gateway.backend_cache import BackendCache
@@ -214,7 +212,10 @@ def do_view(path, source_name=None):
match.status == CacheEntryStatus.loaded
or match.status == CacheEntryStatus.loading
):
return match.serve_content(path)
if source.is_authorized(match.key.descriptor):
return match.serve_content(path)
else:
raise CellxgeneException("User not authorized to access this data", 403)
elif match.status == CacheEntryStatus.error:
raise ProcessException.from_cache_entry(match)
@@ -121,6 +121,9 @@ class FileItemSource(ItemSource):
if self.is_h5ad_file(full_path):
return self.shallowitem_from_descriptor(descriptor)
def is_authorized(self, descriptor):
return True
def lookup(self, indescriptor: str) -> LookupResult:
descriptor = indescriptor.strip("/")
if descriptor.endswith(self.annotation_file_suffix):
+4
View File
@@ -40,6 +40,10 @@ class ItemSource(ABC):
def update(self, item: Item) -> None:
raise Exception('"update" unimplemented')
@abstractmethod
def is_authorized(self, descriptor: str) -> bool:
raise Exception('"is_authorized" unimplemented')
@abstractmethod
def lookup(self, descriptor: str) -> LookupResult:
raise Exception('"lookup" unimplemented')
+25 -4
View File
@@ -7,9 +7,11 @@
# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
# the specific language governing permissions and limitations under the License.
from os.path import basename, dirname, join
import os
from os.path import basename, dirname
from typing import List
import flask
import s3fs
from cellxgene_gateway import dir_util
@@ -18,6 +20,10 @@ from cellxgene_gateway.items.item_source import ItemSource, LookupResult
from cellxgene_gateway.items.s3.s3item import S3Item
def truthy(val: str):
return val.lower() in ["true", "1"]
class S3ItemSource(ItemSource):
def __init__(
self,
@@ -28,7 +34,10 @@ class S3ItemSource(ItemSource):
annotation_file_suffix=".csv",
):
self._name = name
self.s3 = s3fs.S3FileSystem()
enable_cache = os.environ.get("S3_ENABLE_LISTINGS_CACHE", "false").lower()
assert enable_cache in ["0", "1", "false", "true"]
self.use_listings_cache = truthy(enable_cache)
self.s3 = s3fs.S3FileSystem(use_listings_cache=self.use_listings_cache)
if bucket.startswith("s3://"):
raise Exception(
f"Bucket name should not include s3:// prefix, got {bucket}"
@@ -67,6 +76,13 @@ class S3ItemSource(ItemSource):
item_tree = self.scan_directory("" if filter is None else filter)
return item_tree
@property
def refresh(self):
return (
truthy(flask.request.args.get("refresh", default="false"))
or not self.use_listings_cache
)
def scan_directory(self, directory_key="") -> dict:
url = self.url(directory_key)
@@ -75,7 +91,7 @@ class S3ItemSource(ItemSource):
s3key_map = dict(
(self.remove_bucket(filepath), "s3://" + filepath)
for filepath in sorted(self.s3.ls(url))
for filepath in sorted(self.s3.ls(url, refresh=self.refresh))
)
def is_annotation_dir(dir_s3key):
@@ -113,6 +129,9 @@ class S3ItemSource(ItemSource):
def update(self, item: S3Item) -> None:
pass
def is_authorized(self, descriptor):
return True
def lookup_item(self, descriptor):
full_path = self.url(descriptor)
if self.is_h5ad_url(full_path):
@@ -163,7 +182,9 @@ class S3ItemSource(ItemSource):
self.make_s3item_from_key(
basename(annotation), self.remove_bucket(annotation), True
)
for annotation in sorted(self.s3.ls(annotations_fullpath))
for annotation in sorted(
self.s3.ls(annotations_fullpath, refresh=self.refresh)
)
if annotation.endswith(self.annotation_file_suffix)
and self.s3.isfile("s3://" + annotation)
]
@@ -75,7 +75,9 @@
const el = $(this);
const ts = el.text();
const dt = new Date(parseInt(ts * 1000));
el.html(`${dt.toISOString()}<br>(${ts})`);
el.prepend(`${dt.toISOString()}<br>(`);
el.append(')');
});
})
</script>
+7 -2
View File
@@ -24,7 +24,10 @@ class TestScanDirectory(unittest.TestCase):
)
@patch("s3fs.S3FileSystem")
def test__GIVEN_multilevel_bucket_THEN_properly_recurses_suburls(self, s3func):
@patch("flask.request")
def test__GIVEN_multilevel_bucket_THEN_properly_recurses_suburls(
self, requestMock, s3func
):
class S3Mock:
def exists(path):
if path in [
@@ -38,7 +41,8 @@ class TestScanDirectory(unittest.TestCase):
return True
raise Exception("exists called with " + path)
def ls(path):
def ls(path, refresh):
assert refresh == True
if path == "s3://my-bucket/":
return [
"my-bucket/lvl1",
@@ -79,6 +83,7 @@ class TestScanDirectory(unittest.TestCase):
raise Exception("isfile called with " + path)
s3func.return_value = S3Mock
requestMock.args.get.return_value = "true"
source = S3ItemSource("my-bucket")
tree = source.scan_directory()
+17 -1
View File
@@ -1,7 +1,11 @@
import unittest
from unittest.mock import MagicMock, patch
from cellxgene_gateway.filecrawl import render_item, render_item_source
from cellxgene_gateway.filecrawl import (
render_item,
render_item_source,
render_item_tree,
)
from cellxgene_gateway.items.file.fileitem import FileItem
from cellxgene_gateway.items.file.fileitem_source import FileItemSource
from cellxgene_gateway.items.item import ItemTree, ItemType
@@ -41,3 +45,15 @@ class TestRenderItemSource(unittest.TestCase):
rendered,
"<h6><a href='/filecrawl.html?source=FakeSource'>FakeSource</a>:some_filter</h6><li><a href='/filecrawl/rootdir?source=FakeSource'>rootdir</a><ul></ul></li>",
)
class TestRenderItemTree(unittest.TestCase):
@patch("cellxgene_gateway.items.file.fileitem_source.FileItemSource")
def test_GIVEN_deep_nested_dirs_THEN_includes_dirs_in_output(self, item_source):
item_source.name = "FakeSource"
item_tree = ItemTree("foo/bar/baz", [], [])
rendered = render_item_tree(item_tree, item_source)
self.assertEqual(
rendered,
"<li><a href='/filecrawl/foo/bar/baz?source=FakeSource'>baz</a><ul></ul></li>",
)