From b9f4d35812a284fb66459845ad6f6793dad89701 Mon Sep 17 00:00:00 2001 From: Andy Date: Wed, 29 Oct 2025 13:50:09 -0400 Subject: [PATCH 1/6] Fix UnicodeDecodeError when viewing compressed datasets When viewing datasets through the gateway, requests would fail with: UnicodeDecodeError: 'utf-8' codec can't decode byte 0xb5 in position 1 The gateway was copying the accept-encoding header from browser requests when proxying to cellxgene backend servers. When accept-encoding is manually set, the Python requests library assumes the caller will handle decompression and leaves response content compressed. The cellxgene server responded with zstd-compressed content (magic bytes 28 b5 2f fd), but the gateway attempted to decode this compressed binary data as UTF-8 text, causing the decode error. Solution: Remove accept-encoding from the copied headers list in cache_entry.py. This allows the requests library to automatically handle compression negotiation and transparently decompress responses (gzip, deflate, brotli, zstd, etc.). This is the standard practice when proxying with requests and maintains all other gateway functionality (URL rewriting, auth, caching, etc.). Tested: - Dataset viewing works with compressed responses - File browser and static assets load correctly - URL rewriting continues to function properly --- cellxgene_gateway/cache_entry.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/cellxgene_gateway/cache_entry.py b/cellxgene_gateway/cache_entry.py index 8c72efc..c8dbeef 100644 --- a/cellxgene_gateway/cache_entry.py +++ b/cellxgene_gateway/cache_entry.py @@ -148,7 +148,7 @@ class CacheEntry: headers = {} copy_headers = [ "accept", - "accept-encoding", + # "accept-encoding" - removed: let requests library handle compression/decompression "accept-language", "cache-control", "connection", From 08c546f40aae12c0fee5333fca163b6052658c49 Mon Sep 17 00:00:00 2001 From: Andy Date: Wed, 29 Oct 2025 13:58:08 -0400 Subject: [PATCH 2/6] Fix WSGI server initialization by extracting data source setup Addresses the issue where Gunicorn/uWSGI servers import the gateway module but never call main(), leaving item_sources empty and causing the file crawler to fail. Changes: - Extract data source initialization into initialize_data_sources() - Call initialization at module import time for WSGI compatibility - Add _initialized flag to prevent double initialization - Simplify main() to delegate to initialize_data_sources() This ensures data sources are populated when running under WSGI servers (Gunicorn, uWSGI) while maintaining backward compatibility with the Flask development server. Related to GitHub issues #33 and #92 --- cellxgene_gateway/gateway.py | 62 ++++++++++++++++++++++++------------ 1 file changed, 41 insertions(+), 21 deletions(-) diff --git a/cellxgene_gateway/gateway.py b/cellxgene_gateway/gateway.py index 1b9c14e..1e9017b 100644 --- a/cellxgene_gateway/gateway.py +++ b/cellxgene_gateway/gateway.py @@ -78,6 +78,42 @@ if ( cache = BackendCache() +# Initialize data sources - this is defined later in the file but called here +# to ensure initialization happens when WSGI servers (Gunicorn) import the module +def initialize_data_sources(): + """Initialize data sources from environment variables. + Called at module import time for WSGI server compatibility (Gunicorn). + Uses a guard flag to prevent double initialization within a process.""" + global default_item_source + + logging.basicConfig( + level=env.log_level, + format="%(asctime)s:%(name)s:%(levelname)s:%(message)s", + ) + logger = logging.getLogger(__name__) + + cellxgene_data = os.environ.get("CELLXGENE_DATA", None) + cellxgene_bucket = os.environ.get("CELLXGENE_BUCKET", None) + + if cellxgene_bucket is not None: + from cellxgene_gateway.items.s3.s3item_source import S3ItemSource + + item_sources.append(S3ItemSource(cellxgene_bucket, name="s3")) + default_item_source = "s3" + logger.info("Initialized S3 data source") + logger.debug(f"S3 bucket: {cellxgene_bucket}") + if cellxgene_data is not None: + from cellxgene_gateway.items.file.fileitem_source import FileItemSource + + item_sources.append(FileItemSource(cellxgene_data, name="local")) + default_item_source = "local" + logger.info("Initialized local file data source") + logger.debug(f"Data directory: {cellxgene_data}") + if len(item_sources) == 0: + raise Exception("Please specify CELLXGENE_DATA or CELLXGENE_BUCKET") + flask_util.include_source_in_url = len(item_sources) > 1 + + @app.errorhandler(CellxgeneException) def handle_invalid_usage(error): message = f"{error.http_status} Error : {error.message}" @@ -294,29 +330,13 @@ def launch(): app.launchtime = current_time_stamp() app.run(host="0.0.0.0", port=env.gateway_port, debug=False) +# When using servers like Gunicorn or uWSGI, this file is imported rather than run directly. +# As a result, the main() function is never called automatically. +# Therefore, we must initialize the data sources at import time to ensure they are available. +initialize_data_sources() def main(): - logging.basicConfig( - level=env.log_level, - format="%(asctime)s:%(name)s:%(levelname)s:%(message)s", - ) - cellxgene_data = os.environ.get("CELLXGENE_DATA", None) - cellxgene_bucket = os.environ.get("CELLXGENE_BUCKET", None) - - if cellxgene_bucket is not None: - from cellxgene_gateway.items.s3.s3item_source import S3ItemSource - - item_sources.append(S3ItemSource(cellxgene_bucket, name="s3")) - default_item_source = "s3" - if cellxgene_data is not None: - from cellxgene_gateway.items.file.fileitem_source import FileItemSource - - item_sources.append(FileItemSource(cellxgene_data, name="local")) - default_item_source = "local" - if len(item_sources) == 0: - raise Exception("Please specify CELLXGENE_DATA or CELLXGENE_BUCKET") - flask_util.include_source_in_url = len(item_sources) > 1 - + """CLI entry point for Flask development server.""" launch() From 903d25763f57705545223bc986b51816889063d7 Mon Sep 17 00:00:00 2001 From: Andy Date: Wed, 29 Oct 2025 14:00:02 -0400 Subject: [PATCH 3/6] Fix AttributeError by storing ItemSource objects in default_item_source Previously, default_item_source was set to a string ("s3" or "local"), but matching_source() tried to access default_item_source.name, causing: AttributeError: 'str' object has no attribute 'name' This bug occurred when source_name=None (single data source configuration) and has existed since the ItemSource interface was introduced in 2021. Changes: - Store the actual ItemSource object reference instead of string name - Assign to intermediate variables (s3_source, file_source) for clarity Fixes the error when viewing datasets with a single data source configured. --- cellxgene_gateway/gateway.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/cellxgene_gateway/gateway.py b/cellxgene_gateway/gateway.py index 1e9017b..6420f95 100644 --- a/cellxgene_gateway/gateway.py +++ b/cellxgene_gateway/gateway.py @@ -98,15 +98,17 @@ def initialize_data_sources(): if cellxgene_bucket is not None: from cellxgene_gateway.items.s3.s3item_source import S3ItemSource - item_sources.append(S3ItemSource(cellxgene_bucket, name="s3")) - default_item_source = "s3" + s3_source = S3ItemSource(cellxgene_bucket, name="s3") + item_sources.append(s3_source) + default_item_source = s3_source logger.info("Initialized S3 data source") logger.debug(f"S3 bucket: {cellxgene_bucket}") if cellxgene_data is not None: from cellxgene_gateway.items.file.fileitem_source import FileItemSource - item_sources.append(FileItemSource(cellxgene_data, name="local")) - default_item_source = "local" + file_source = FileItemSource(cellxgene_data, name="local") + item_sources.append(file_source) + default_item_source = file_source logger.info("Initialized local file data source") logger.debug(f"Data directory: {cellxgene_data}") if len(item_sources) == 0: From 3e5accad650d35c7484c9dabc9fff00d8519e0ba Mon Sep 17 00:00:00 2001 From: Alok Saldanha Date: Tue, 4 Nov 2025 06:27:56 -0500 Subject: [PATCH 4/6] fixed linting and tests --- cellxgene_gateway/cache_entry.py | 3 +- cellxgene_gateway/gateway.py | 2 ++ tests/items/s3/test_s3item_source.py | 24 +++++++++++-- tests/test_cache_entry.py | 21 +++++++++-- tests/test_filecrawl.py | 54 ++++++++++++++++++++++++++++ 5 files changed, 97 insertions(+), 7 deletions(-) diff --git a/cellxgene_gateway/cache_entry.py b/cellxgene_gateway/cache_entry.py index c8dbeef..8ddb2be 100644 --- a/cellxgene_gateway/cache_entry.py +++ b/cellxgene_gateway/cache_entry.py @@ -168,9 +168,8 @@ class CacheEntry: headers[h] = request.headers[h] full_path = self.cellxgene_basepath() + subpath + querystring() - + cellxgene_response = None try: - cellxgene_response = None if request.method in ["GET", "HEAD", "OPTIONS"]: cellxgene_response = get(full_path, headers=headers) elif request.method == "PUT": diff --git a/cellxgene_gateway/gateway.py b/cellxgene_gateway/gateway.py index 6420f95..f99d09e 100644 --- a/cellxgene_gateway/gateway.py +++ b/cellxgene_gateway/gateway.py @@ -332,11 +332,13 @@ def launch(): app.launchtime = current_time_stamp() app.run(host="0.0.0.0", port=env.gateway_port, debug=False) + # When using servers like Gunicorn or uWSGI, this file is imported rather than run directly. # As a result, the main() function is never called automatically. # Therefore, we must initialize the data sources at import time to ensure they are available. initialize_data_sources() + def main(): """CLI entry point for Flask development server.""" launch() diff --git a/tests/items/s3/test_s3item_source.py b/tests/items/s3/test_s3item_source.py index d96855c..1ff675d 100644 --- a/tests/items/s3/test_s3item_source.py +++ b/tests/items/s3/test_s3item_source.py @@ -1,13 +1,32 @@ import unittest +import os +import shutil +import tempfile + from unittest.mock import MagicMock, Mock, patch -from cellxgene_gateway.gateway import app from cellxgene_gateway.items.item import ItemType from cellxgene_gateway.items.s3.s3item import S3Item from cellxgene_gateway.items.s3.s3item_source import S3ItemSource class TestScanDirectory(unittest.TestCase): + def setUp(self): + self._tmpdir = tempfile.mkdtemp() + self._cellxgene_data = os.environ.get("CELLXGENE_DATA", "") + os.environ["CELLXGENE_DATA"] = self._tmpdir + + from cellxgene_gateway.gateway import app + + self.app = app + + def tearDown(self): + if self._cellxgene_data: + os.environ["CELLXGENE_DATA"] = self._cellxgene_data + else: + del os.environ["CELLXGENE_DATA"] + shutil.rmtree(self._tmpdir) + @patch("s3fs.S3FileSystem") def test_GIVEN_invalid_bucket_THEN_throws_error(self, s3func): class S3Mock: @@ -26,6 +45,7 @@ class TestScanDirectory(unittest.TestCase): @patch("s3fs.S3FileSystem") def test__GIVEN_multilevel_bucket_THEN_properly_recurses_suburls(self, s3func): + class S3Mock: def exists(path): if path in [ @@ -82,7 +102,7 @@ class TestScanDirectory(unittest.TestCase): s3func.return_value = S3Mock source = S3ItemSource("my-bucket") - with app.test_request_context(query_string="refresh=true") as test_context: + with self.app.test_request_context(query_string="refresh=true") as test_context: tree = source.scan_directory() def s3item_compare(i1, i2, msg=""): diff --git a/tests/test_cache_entry.py b/tests/test_cache_entry.py index 697e8d1..3c1054f 100644 --- a/tests/test_cache_entry.py +++ b/tests/test_cache_entry.py @@ -1,14 +1,15 @@ import unittest +import tempfile +import os +import shutil from flask import Flask - from cellxgene_gateway import flask_util from cellxgene_gateway.cache_entry import CacheEntry, CacheEntryStatus from cellxgene_gateway.cache_key import CacheKey -from cellxgene_gateway.gateway import app +from cellxgene_gateway.items.item import ItemType from cellxgene_gateway.items.file.fileitem import FileItem from cellxgene_gateway.items.file.fileitem_source import FileItemSource -from cellxgene_gateway.items.item import ItemType key = CacheKey( FileItem("/czi/", name="pbmc3k.h5ad", type=ItemType.h5ad), @@ -18,11 +19,25 @@ key = CacheKey( class TestRenderEntry(unittest.TestCase): def setUp(self): + self._tmpdir = tempfile.mkdtemp() + self._cellxgene_data = os.environ.get("CELLXGENE_DATA", "") + os.environ["CELLXGENE_DATA"] = self._tmpdir + + from cellxgene_gateway.gateway import app + self.app = app self.app_context = self.app.test_request_context() self.app_context.push() self.client = self.app.test_client() + def tearDown(self): + self.app_context.pop() + if self._cellxgene_data: + os.environ["CELLXGENE_DATA"] = self._cellxgene_data + else: + del os.environ["CELLXGENE_DATA"] + shutil.rmtree(self._tmpdir) + def test_GIVEN_key_and_port_THEN_returns_loading_CacheEntry(self): entry = CacheEntry.for_key("some-key", 1) self.assertEqual(entry.status, CacheEntryStatus.loading) diff --git a/tests/test_filecrawl.py b/tests/test_filecrawl.py index ade01e6..d42d783 100644 --- a/tests/test_filecrawl.py +++ b/tests/test_filecrawl.py @@ -1,3 +1,6 @@ +import os +import shutil +import tempfile import unittest from collections import defaultdict from unittest.mock import patch @@ -25,6 +28,25 @@ def make_entry(subpath="somepath", annotations=None): class TestRenderEntry(unittest.TestCase): + def setUp(self): + self._tmpdir = tempfile.mkdtemp() + self._cellxgene_data = os.environ.get("CELLXGENE_DATA", "") + os.environ["CELLXGENE_DATA"] = self._tmpdir + + from cellxgene_gateway.gateway import app + + self.app = app + self.app_context = self.app.test_request_context() + self.app_context.push() + + def tearDown(self): + self.app_context.pop() + if self._cellxgene_data: + os.environ["CELLXGENE_DATA"] = self._cellxgene_data + else: + del os.environ["CELLXGENE_DATA"] + shutil.rmtree(self._tmpdir) + def test_GIVEN_path_both_slash_THEN_view_has_single_slash(self): entry = make_entry(subpath="/somepath/") rendered = render_item(entry, source) @@ -47,6 +69,26 @@ class TestRenderEntry(unittest.TestCase): class TestRenderAnnotation(unittest.TestCase): + + def setUp(self): + self._tmpdir = tempfile.mkdtemp() + self._cellxgene_data = os.environ.get("CELLXGENE_DATA", "") + os.environ["CELLXGENE_DATA"] = self._tmpdir + + from cellxgene_gateway.gateway import app + + self.app = app + self.app_context = self.app.test_request_context() + self.app_context.push() + + def tearDown(self): + self.app_context.pop() + if self._cellxgene_data: + os.environ["CELLXGENE_DATA"] = self._cellxgene_data + else: + del os.environ["CELLXGENE_DATA"] + shutil.rmtree(self._tmpdir) + @patch("cellxgene_gateway.filecrawl.enable_annotations", new=True) def test_GIVEN_no_annotation_THEN_new_alone(self): entry = make_entry(annotations=None) @@ -103,12 +145,24 @@ class TestRenderItemSource(unittest.TestCase): class TestRenderItemTree(unittest.TestCase): def setUp(self): + self._tmpdir = tempfile.mkdtemp() + self._cellxgene_data = os.environ.get("CELLXGENE_DATA", "") + os.environ["CELLXGENE_DATA"] = self._tmpdir + from cellxgene_gateway.gateway import app self.app = app self.app_context = self.app.test_request_context() self.app_context.push() + def tearDown(self): + self.app_context.pop() + if self._cellxgene_data: + os.environ["CELLXGENE_DATA"] = self._cellxgene_data + else: + del os.environ["CELLXGENE_DATA"] + shutil.rmtree(self._tmpdir) + @patch("cellxgene_gateway.items.file.fileitem_source.FileItemSource") def test_GIVEN_deep_nested_dirs_THEN_includes_dirs_in_output(self, item_source): item_source.name = "FakeSource" From d16a97906cadc932f283f6af162bc48229de0b9d Mon Sep 17 00:00:00 2001 From: Alok Saldanha Date: Wed, 5 Nov 2025 06:44:54 -0500 Subject: [PATCH 5/6] Skip codecov for PRs from fork --- .github/workflows/pr-checks.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/.github/workflows/pr-checks.yaml b/.github/workflows/pr-checks.yaml index 147a4ce..6250cea 100644 --- a/.github/workflows/pr-checks.yaml +++ b/.github/workflows/pr-checks.yaml @@ -55,6 +55,7 @@ jobs: coverage xml -i - name: "Upload coverage to Codecov" + if: ${{ secrets.CODECOV_TOKEN != '' }} uses: codecov/codecov-action@v1 with: token: ${{ secrets.CODECOV_TOKEN }} From 34ac73ba01a0715891a4243c6d045891b6dfaba9 Mon Sep 17 00:00:00 2001 From: Alok Saldanha Date: Wed, 5 Nov 2025 06:47:20 -0500 Subject: [PATCH 6/6] second attempt to skip codecov on PRs from forks --- .github/workflows/pr-checks.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/pr-checks.yaml b/.github/workflows/pr-checks.yaml index 6250cea..8a7c92b 100644 --- a/.github/workflows/pr-checks.yaml +++ b/.github/workflows/pr-checks.yaml @@ -55,7 +55,7 @@ jobs: coverage xml -i - name: "Upload coverage to Codecov" - if: ${{ secrets.CODECOV_TOKEN != '' }} + if: ${{ github.event_name == 'push' || (github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name == github.repository) }} uses: codecov/codecov-action@v1 with: token: ${{ secrets.CODECOV_TOKEN }}