diff --git a/README.md b/README.md
index 8974341..58d6981 100644
--- a/README.md
+++ b/README.md
@@ -59,7 +59,11 @@ cellxgene-gateway
Here's what the environment variables mean:
* `CELLXGENE_LOCATION` - the location of the cellxgene executable, e.g. `~/anaconda2/envs/cellxgene/bin/cellxgene`
+
+At least one of the following is required:
* `CELLXGENE_DATA` - a directory that can contain subdirectories with `.h5ad` data files, *without* trailing slash, e.g. `/mnt/cellxgene_data`
+* `CELLXGENE_BUCKET` - an s3 bucket that can contain keys with `.h5ad` data files, e.g. `my-cellxgene-data-bucket`
+Cellxgene Gateway is designed to make it easy to add additional data sources, please see the source code for gateway.py and the ItemSource interface in items/item_source.py
Optional environment variables:
* `CELLXGENE_ARGS` - catch-all variable that can be used to pass additional command line args to cellxgene server
@@ -68,7 +72,6 @@ Optional environment variables:
* `GATEWAY_IP` - ip addess of instance gateway is running on, mostly used to display SSH instructions. Defaults to `socket.gethostbyname(socket.gethostname())`
* `GATEWAY_PORT` - local port that the gateway should bind to, defaults to 5005
* `GATEWAY_EXTRA_SCRIPTS` - JSON array of script paths, will be embedded into each page and forwarded with `--scripts` to cellxgene server
-* `GATEWAY_ENABLE_UPLOAD` - Set to `true` or `1` to enable HTTP uploads. This is not recommended for a public server.
* `GATEWAY_ENABLE_ANNOTATIONS` - Set to `true` or to `1` to enable cellxgene annotations.
* `GATEWAY_ENABLE_BACKED_MODE` - Set to `true` or to `1` to load AnnData in file-backed mode. This saves memory and speeds up launch time but may reduce overall performance.
@@ -142,9 +145,8 @@ pre-commit install
pip install isort flake8 black
```bash
-isort -rc .
-flake8 .
-black -l 79 .
+isort -rc . # rc means recursive, and was deprecated in dev version of isort
+black .
```
# Getting Help
diff --git a/cellxgene_gateway/backend_cache.py b/cellxgene_gateway/backend_cache.py
index df20eb2..b334cf1 100644
--- a/cellxgene_gateway/backend_cache.py
+++ b/cellxgene_gateway/backend_cache.py
@@ -9,11 +9,13 @@
import time
from threading import Thread
+from typing import List
from flask_api import status
from cellxgene_gateway import env
from cellxgene_gateway.cache_entry import CacheEntry, CacheEntryStatus
+from cellxgene_gateway.cache_key import CacheKey
from cellxgene_gateway.cellxgene_exception import CellxgeneException
from cellxgene_gateway.subprocess_backend import SubprocessBackend
@@ -35,13 +37,13 @@ class BackendCache:
contents = self.entry_list
return [c.port for c in contents]
- def check_entry(self, key):
+ def check_path(self, source, path):
contents = self.entry_list
matches = [
c
for c in contents
- if c.key.dataset == key.dataset
- and c.key.annotation_file == key.annotation_file
+ if c.key.source.name == source.name
+ and path.startswith(c.key.descriptor)
and c.status != CacheEntryStatus.terminated
]
@@ -52,10 +54,28 @@ class BackendCache:
else:
raise CellxgeneException(
status.HTTP_500_INTERNAL_SERVER_ERROR,
- "Found " + str(len(matches)) + " for " + dataset,
+ "Found " + str(len(matches)) + " for " + path,
)
- def create_entry(self, key, scripts):
+ def check_entry(self, key):
+ contents = self.entry_list
+ matches = [
+ c
+ for c in contents
+ if c.key.equals(key) and c.status != CacheEntryStatus.terminated
+ ]
+
+ if len(matches) == 0:
+ return None
+ elif len(matches) == 1:
+ return matches[0]
+ else:
+ raise CellxgeneException(
+ status.HTTP_500_INTERNAL_SERVER_ERROR,
+ "Found " + str(len(matches)) + " for " + key.dataset,
+ )
+
+ def create_entry(self, key: CacheKey, scripts: List[str]):
port = 8000
existing_ports = self.get_ports()
diff --git a/cellxgene_gateway/cache_entry.py b/cellxgene_gateway/cache_entry.py
index 2eb6cca..2e8b1db 100644
--- a/cellxgene_gateway/cache_entry.py
+++ b/cellxgene_gateway/cache_entry.py
@@ -9,11 +9,11 @@
import datetime
import logging
import re
+import urllib.parse
from enum import Enum
import psutil
from flask import make_response, render_template, request
-from flask.helpers import url_for
from flask.wrappers import Response
from requests import get, post, put
@@ -71,6 +71,10 @@ class CacheEntry:
None,
)
+ @property
+ def source_name(self):
+ return self.key.source_name
+
def set_loaded(self, pid):
self.pid = pid
self.status = CacheEntryStatus.loaded
@@ -100,9 +104,13 @@ class CacheEntry:
for child in children:
child.terminate()
psutil.wait_procs(children, callback=on_terminate)
- terminated.append(p.pid)
- p.terminate()
- psutil.wait_procs([p], callback=on_terminate)
+ # the parent process may automatically die once its children have --
+ try:
+ p.terminate()
+ psutil.wait_procs([p], callback=on_terminate)
+ except psutil.NoSuchProcess:
+ pass
+
logging.getLogger("cellxgene_gateway").info(f"terminated {terminated}")
self.status = CacheEntryStatus.terminated
@@ -111,24 +119,20 @@ class CacheEntry:
gateway_content = (
re.sub(
'(="|\()/static/',
- f"\\1{self.gateway_basepath()}static/",
+ f"\\1{self.key.gateway_basepath()}static/",
cellxgene_content,
)
.replace("http://fonts.gstatic.com", "https://fonts.gstatic.com")
- .replace(self.cellxgene_basepath(), self.gateway_basepath())
+ .replace(self.cellxgene_basepath(), self.key.gateway_basepath())
)
return gateway_content
- def gateway_basepath(self):
- return url_for("do_view", path=self.key.pathpart) + "/"
-
def cellxgene_basepath(self):
return f"http://127.0.0.1:{self.port}"
def serve_content(self, path):
- gateway_basepath = self.gateway_basepath()
- subpath = path[len(self.key.pathpart) :] # noqa: E203
-
+ gateway_basepath = self.key.gateway_basepath()
+ subpath = path[len(self.key.descriptor) :] # noqa: E203
if len(subpath) == 0:
r = make_response(f"Redirect to {gateway_basepath}\n", 302)
r.headers["location"] = gateway_basepath + querystring()
@@ -199,5 +203,4 @@ class CacheEntry:
cellxgene_response.status_code,
resp_headers,
)
-
return gateway_response
diff --git a/cellxgene_gateway/cache_key.py b/cellxgene_gateway/cache_key.py
index f45bb92..975f3c8 100644
--- a/cellxgene_gateway/cache_key.py
+++ b/cellxgene_gateway/cache_key.py
@@ -9,15 +9,72 @@
# There are three kinds of CacheKey:
# 1) somedir/dataset.h5ad: a dataset
-# in this case, pathpart == dataset == 'somedir/dataset.h5ad'
-# 2) somedir/dataset_annotations/my_annotations.csv : an actual annotaitons file.
-# in this case, pathpart == 'dataset_annotations/my_annotations.csv', dataset == 'somedir/dataset.h5ad'
+# in this case, descriptor == dataset == 'somedir/dataset.h5ad'
+# 2) somedir/dataset_annotations/my_annotations.csv : an actual annotations file.
+# in this case, descriptor == 'somedir/dataset_annotations/my_annotations.csv', dataset == 'somedir/dataset.h5ad'
# 3) somedir/dataset_annotations: an annotation directory. The corresponding h5ad must exist, but the directory may not.
-# in this case, pathpart == 'dataset_annotations', dataset == 'somedir/dataset.h5ad'
+# in this case, descriptor == 'somedir/dataset_annotations', dataset == 'somedir/dataset.h5ad'
+
+from flask.helpers import url_for
+
+from cellxgene_gateway.items.item import Item
+from cellxgene_gateway.items.item_source import ItemSource, LookupResult
class CacheKey:
- def __init__(self, pathpart, dataset, annotation_file):
- self.pathpart = pathpart
- self.dataset = dataset
- self.annotation_file = annotation_file
+ @property
+ def descriptor(self):
+ if self.annotation_item is None:
+ return self.h5ad_item.descriptor
+ else:
+ return self.annotation_item.descriptor
+
+ @property
+ def file_path(self):
+ return self.source.get_local_path(self.h5ad_item)
+
+ @property
+ def annotation_file_path(self):
+ if self.annotation_item is None:
+ return None
+ else:
+ return self.source.get_local_path(self.annotation_item)
+
+ def relaunch_url(self):
+ return url_for("do_relaunch", source=self.source_name, path=self.descriptor)
+
+ def gateway_basepath(self):
+ return (
+ url_for("do_view", source_name=self.source_name, path=self.descriptor) + "/"
+ )
+
+ @property
+ def source_name(self):
+ return self.source.name
+
+ @property
+ def annotation_descriptor(self):
+ if self.annotation_item is None:
+ return None
+ else:
+ return self.annotation_item.descriptor
+
+ def equals(self, other):
+ return (
+ (self.source.name == other.source.name)
+ and (self.h5ad_item.descriptor == other.h5ad_item.descriptor)
+ and (self.annotation_descriptor == other.annotation_descriptor)
+ )
+
+ def __init__(
+ self, h5ad_item: Item, source: ItemSource, annotation_item: Item = None
+ ):
+ assert h5ad_item is not None
+ assert source is not None
+ self.h5ad_item = h5ad_item
+ self.annotation_item = annotation_item
+ self.source = source
+
+ @classmethod
+ def for_lookup(cls, source: ItemSource, lookup: LookupResult):
+ return CacheKey(lookup.h5ad_item, source, lookup.annotation_item)
diff --git a/cellxgene_gateway/dir_util.py b/cellxgene_gateway/dir_util.py
index a7bc262..3ca9957 100644
--- a/cellxgene_gateway/dir_util.py
+++ b/cellxgene_gateway/dir_util.py
@@ -53,10 +53,11 @@ def create_dir(parent_path, dir_name):
annotations_suffix = "_annotations"
+h5ad_suffix = ".h5ad"
def make_h5ad(el):
- return el[: -len(annotations_suffix)] + ".h5ad"
+ return el[: -len(annotations_suffix)] + h5ad_suffix
def make_annotations(el):
diff --git a/cellxgene_gateway/env.py b/cellxgene_gateway/env.py
index b7a98fd..078a295 100644
--- a/cellxgene_gateway/env.py
+++ b/cellxgene_gateway/env.py
@@ -12,7 +12,7 @@ import os
import socket
cellxgene_location = os.environ.get("CELLXGENE_LOCATION")
-cellxgene_data = os.environ.get("CELLXGENE_DATA")
+cellxgene_data = os.environ.get("CELLXGENE_DATA", "")
cellxgene_args = os.environ.get("CELLXGENE_ARGS", None)
gateway_port = int(os.environ.get("GATEWAY_PORT", "5005"))
external_host = os.environ.get(
@@ -22,13 +22,9 @@ external_host = os.environ.get(
external_protocol = os.environ.get(
"EXTERNAL_PROTOCOL", os.environ.get("GATEWAY_PROTOCOL", "http")
)
-ip = os.environ.get("GATEWAY_IP", "127.0.0.1")
+ip = os.environ.get("GATEWAY_IP")
extra_scripts = os.environ.get("GATEWAY_EXTRA_SCRIPTS")
ttl = os.environ.get("GATEWAY_TTL")
-enable_upload = os.environ.get("GATEWAY_ENABLE_UPLOAD", "").lower() in [
- "true",
- "1",
-]
enable_annotations = os.environ.get("GATEWAY_ENABLE_ANNOTATIONS", "").lower() in [
"true",
"1",
@@ -40,7 +36,6 @@ enable_backed_mode = os.environ.get("GATEWAY_ENABLE_BACKED_MODE", "").lower() in
env_vars = {
"CELLXGENE_LOCATION": cellxgene_location,
- "CELLXGENE_DATA": cellxgene_data,
}
proxy_fix_for = int(os.environ.get("PROXY_FIX_FOR", "0"))
@@ -56,10 +51,10 @@ optional_env_vars = {
"GATEWAY_PORT": gateway_port,
"GATEWAY_EXTRA_SCRIPTS": extra_scripts,
"GATEWAY_TTL": ttl,
- "GATEWAY_ENABLE_UPLOAD": enable_upload,
"GATEWAY_ENABLE_ANNOTATIONS": enable_annotations,
"GATEWAY_ENABLE_BACKED_MODE": enable_backed_mode,
"CELLXGENE_ARGS": cellxgene_args,
+ "CELLXGENE_DATA": cellxgene_data,
"PROXY_FIX_FOR": proxy_fix_for,
"PROXY_FIX_PROTO": proxy_fix_proto,
"PROXY_FIX_HOST": proxy_fix_host,
diff --git a/cellxgene_gateway/filecrawl.py b/cellxgene_gateway/filecrawl.py
index 5a96f9d..c2d5a92 100644
--- a/cellxgene_gateway/filecrawl.py
+++ b/cellxgene_gateway/filecrawl.py
@@ -8,6 +8,7 @@
# the specific language governing permissions and limitations under the License.
import os
+import urllib.parse
from flask import url_for
@@ -15,104 +16,55 @@ from cellxgene_gateway import env
from cellxgene_gateway.dir_util import annotations_suffix, make_annotations, make_h5ad
-def recurse_dir(path):
- if not os.path.exists(path):
- raise CellxgeneException(
- "The given path does not exist.", status.HTTP_400_BAD_REQUEST
- )
-
- all_entries = sorted(os.listdir(path))
-
- def is_h5ad(el):
- return el.endswith(".h5ad") and os.path.isfile(os.path.join(path, el))
-
- h5ad_entries = [x for x in all_entries if is_h5ad(x)]
- annotation_dir_entries = [
- x
- for x in all_entries
- if x.endswith(annotations_suffix) and make_h5ad(x) in h5ad_entries
- ]
-
- def list_annotations(el):
- full_path = os.path.join(path, el)
- if not os.path.isdir(full_path):
- entries = []
- else:
- entries = [
- {
- "name": x[:-13]
- if (len(x) > 13 and x[-13] in ["-", "_"])
- else (x[:-4] if x.endswith(".csv") else x),
- "path": os.path.join(full_path, x + "/").replace(
- env.cellxgene_data, ""
- ),
- }
- for x in sorted(os.listdir(full_path))
- if x.endswith(".csv") and os.path.isfile(os.path.join(full_path, x))
- ]
- return [
- {
- "name": "new",
- "class": "new",
- "path": full_path.replace(env.cellxgene_data, "").rstrip("/"),
- }
- ] + entries
-
- def make_entry(el):
- full_path = os.path.join(path, el)
- if el in h5ad_entries:
- return {
- "path": full_path.replace(env.cellxgene_data, ""),
- "name": el,
- "type": "file",
- "annotations": list_annotations(make_annotations(el)),
- }
- elif os.path.isdir(full_path) and el not in annotation_dir_entries:
- return {
- "path": full_path.replace(env.cellxgene_data, ""),
- "name": el,
- "type": "directory",
- "children": recurse_dir(full_path),
- }
- else:
- return {
- "path": full_path,
- "name": el,
- "type": "neither",
- }
-
- return [make_entry(x) for x in all_entries]
-
-
-def render_entries(entries):
- return "
" + "\n".join([render_entry(e) for e in entries]) + "
"
-
-
-def get_url(entry):
- return url_for("do_view", path=entry["path"].lstrip("/"))
-
-
-def get_class(entry):
- return f" class='{entry['class']}'" if "class" in entry else ""
-
-
-def render_annotations(entry):
- if len(entry["annotations"]) > 0:
- return " | annotations: " + ", ".join(
+def render_annotations(item, item_source):
+ url = url_for(
+ "do_view",
+ path=item_source.get_annotations_subpath(item),
+ source_name=item_source.name,
+ )
+ new_annotation = f"new"
+ annotations = (
+ ", ".join(
[
- f"{a['name']}"
- for a in entry["annotations"]
+ f"{a.name}"
+ for a in item.annotations
]
)
- else:
- return ""
+ + ", "
+ if item.annotations
+ else ""
+ )
+ return " | annotations: " + annotations + new_annotation
-def render_entry(entry):
- if entry["type"] == "file":
- return f" {entry['name']} {render_annotations(entry)}"
- elif entry["type"] == "directory":
- url = f"/filecrawl/{entry['path'].lstrip('/')}"
- return f"{entry['name']}{render_entries(entry['children'])}"
+def render_item(item, item_source):
+ url = url_for("do_view", path=item.descriptor, source_name=item_source.name) + "/"
+ item_string = f" {item.name} {render_annotations(item, item_source)}"
+ return item_string
+
+
+def render_item_tree(item_tree, item_source):
+ items = (
+ "\n".join([render_item(i, item_source) for i in item_tree.items])
+ if item_tree.items
+ else ""
+ )
+ branches = (
+ "\n".join([render_item_tree(b, item_source) for b in item_tree.branches])
+ if item_tree.branches
+ else ""
+ )
+ html = ""
+ if item_tree.descriptor:
+ descriptor = item_tree.descriptor.lstrip("/")
+ url = f"/filecrawl/{descriptor}?source={item_source.name}"
+ name = descriptor.rsplit("/")[1] if descriptor.find("/") >= 0 else descriptor
+ return f"{name}{html}"
else:
- return ""
+ return html
+
+
+def render_item_source(item_source, filter=None):
+ item_tree = item_source.list_items(filter)
+ heading = f""
+ return heading + render_item_tree(item_tree, item_source)
diff --git a/cellxgene_gateway/gateway.py b/cellxgene_gateway/gateway.py
index 08a6f10..e8b8b50 100644
--- a/cellxgene_gateway/gateway.py
+++ b/cellxgene_gateway/gateway.py
@@ -6,10 +6,11 @@
# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES
# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
# the specific language governing permissions and limitations under the License.
-
+# import BaseHTTPServer
import json
import logging
import os
+import urllib.parse
from threading import Lock, Thread
from flask import (
@@ -28,17 +29,20 @@ from werkzeug.utils import secure_filename
from cellxgene_gateway import env
from cellxgene_gateway.backend_cache import BackendCache
from cellxgene_gateway.cache_entry import CacheEntryStatus
+from cellxgene_gateway.cache_key import CacheKey
from cellxgene_gateway.cellxgene_exception import CellxgeneException
from cellxgene_gateway.dir_util import create_dir, is_subdir
from cellxgene_gateway.extra_scripts import get_extra_scripts
-from cellxgene_gateway.filecrawl import recurse_dir, render_entries
-from cellxgene_gateway.path_util import get_key
+from cellxgene_gateway.filecrawl import render_item_source
from cellxgene_gateway.process_exception import ProcessException
from cellxgene_gateway.prune_process_cache import PruneProcessCache
from cellxgene_gateway.util import current_time_stamp
app = Flask(__name__)
+item_sources = []
+default_item_source = None
+
def _force_https(app):
def wrapper(environ, start_response):
@@ -101,8 +105,8 @@ def handle_invalid_process(error):
http_status=error.http_status,
stdout=error.stdout,
stderr=error.stderr,
- dataset=error.key.dataset,
- annotation_file=error.key.annotation_file,
+ relaunch_url=error.key.relaunch_url(),
+ annotation_file=error.key.annotation_descriptor,
),
error.http_status,
)
@@ -119,74 +123,40 @@ def favicon():
@app.route("/")
def index():
- users = [
- name
- for name in os.listdir(env.cellxgene_data)
- if os.path.isdir(os.path.join(env.cellxgene_data, name))
- ]
return render_template(
"index.html",
ip=env.ip,
cellxgene_data=env.cellxgene_data,
extra_scripts=get_extra_scripts(),
- users=users,
- enable_upload=env.enable_upload,
)
-def make_user():
- dir_name = request.form["directory"]
+@app.route("/filecrawl.html")
+@app.route("/filecrawl/")
+def filecrawl(path=None):
+ source_name = request.args.get("source")
+ sources = (
+ filter(
+ lambda x: x.name == urllib.parse.unquote_plus(source_name),
+ item_sources,
+ )
+ if source_name
+ else item_sources
+ )
+ # loop all data sources --
+ rendered_sources = [
+ render_item_source(item_source, path) for item_source in sources
+ ] # will we need to make this async in the page???
+ rendered_html = "\n".join(rendered_sources)
- create_dir(env.cellxgene_data, dir_name)
-
- return redirect(url_for("index"), code=302)
-
-
-def make_subdir():
- parent_path = os.path.join(env.cellxgene_data, request.form["usernames"])
- dir_name = request.form["directory"]
-
- create_dir(parent_path, dir_name)
-
- return redirect(url_for("index"), code=302)
-
-
-def upload_file():
- upload_dir = request.form["path"]
-
- full_upload_path = os.path.join(env.cellxgene_data, upload_dir)
- if is_subdir(full_upload_path, env.cellxgene_data) and os.path.isdir(
- full_upload_path
- ):
- if request.method == "POST":
- if "file" in request.files:
- f = request.files["file"]
- if f and f.filename.endswith(".h5ad"):
- f.save(os.path.join(full_upload_path, secure_filename(f.filename)))
- return redirect(url_for("filecrawl"), code=302)
- else:
- raise CellxgeneException(
- "Uploaded file must be in anndata (.h5ad) format.",
- status.HTTP_400_BAD_REQUEST,
- )
- else:
- raise CellxgeneException(
- "A file must be chosen to upload.",
- status.HTTP_400_BAD_REQUEST,
- )
- else:
- raise CellxgeneException("Invalid directory.", status.HTTP_400_BAD_REQUEST)
-
- return redirect(url_for("index"), code=302)
-
-
-if env.enable_upload:
- app.add_url_rule("/make_user", "make_user", make_user, methods=["POST"])
- app.add_url_rule("/make_subdir", "make_subdir", make_subdir, methods=["POST"])
- app.add_url_rule("/upload_file", "upload_file", upload_file, methods=["POST"])
-
-
-def set_no_cache(resp):
+ resp = make_response(
+ render_template(
+ "filecrawl.html",
+ extra_scripts=get_extra_scripts(),
+ rendered_html=rendered_html,
+ path=path,
+ )
+ )
resp.headers["Cache-Control"] = "no-cache, no-store, must-revalidate"
resp.headers["Pragma"] = "no-cache"
resp.headers["Expires"] = "0"
@@ -194,52 +164,44 @@ def set_no_cache(resp):
return resp
-@app.route("/filecrawl.html")
-def filecrawl():
- entries = recurse_dir(env.cellxgene_data)
- rendered_html = render_entries(entries)
- resp = make_response(
- render_template(
- "filecrawl.html",
- extra_scripts=get_extra_scripts(),
- rendered_html=rendered_html,
- )
- )
- return set_no_cache(resp)
-
-
-@app.route("/filecrawl/")
-def do_filecrawl(path):
- filecrawl_path = os.path.join(env.cellxgene_data, path)
- if not os.path.isdir(filecrawl_path):
- raise CellxgeneException(
- "Path is not directory: " + filecrawl_path,
- status.HTTP_400_BAD_REQUEST,
- )
- entries = recurse_dir(filecrawl_path)
- rendered_html = render_entries(entries)
- return render_template(
- "filecrawl.html",
- extra_scripts=get_extra_scripts(),
- rendered_html=rendered_html,
- path=path,
- )
-
-
entry_lock = Lock()
+def matching_source(source_name):
+ if source_name is None:
+ source_name = default_item_source.name
+ matching = [i for i in item_sources if i.name == source_name]
+ if len(matching) != 1:
+ raise Exception(f"Could not find matching item source {source_name}")
+ source = matching[0]
+ return source
+
+
+@app.route(
+ "/source//view/",
+ methods=["GET", "PUT", "POST"],
+)
@app.route("/view/", methods=["GET", "PUT", "POST"])
-def do_view(path):
- key = get_key(path)
- print(
- f"view path={path}, dataset={key.dataset}, annotation_file= {key.annotation_file}, key={key.pathpart}"
- )
- with entry_lock:
- match = cache.check_entry(key)
- if match is None:
- uascripts = get_extra_scripts()
- match = cache.create_entry(key, uascripts)
+def do_view(path, source_name=None):
+ source = matching_source(source_name)
+ match = cache.check_path(source, path)
+
+ if match is None:
+ lookup = source.lookup(path)
+ if lookup is None:
+ raise CellxgeneException(
+ f"Could not find item for path {path} in source {source.name}",
+ 404,
+ )
+ key = CacheKey.for_lookup(source, lookup)
+ print(
+ f"view path={path}, source_name={source_name}, dataset={key.file_path}, annotation_file= {key.annotation_file_path}, key={key.descriptor}, source={key.source_name}"
+ )
+ with entry_lock:
+ match = cache.check_entry(key)
+ if match is None:
+ uascripts = get_extra_scripts()
+ match = cache.create_entry(key, uascripts)
match.timestamp = current_time_stamp()
@@ -278,7 +240,9 @@ def do_GET_status_json():
@app.route("/relaunch/", methods=["GET"])
def do_relaunch(path):
- key = get_key(path)
+ source_name = request.args.get("source") or default_item_source.name
+ source = matching_source(source_name)
+ key = CacheKey.for_lookup(source, source.lookup(path))
match = cache.check_entry(key)
if not match is None:
match.terminate()
@@ -291,25 +255,24 @@ def do_relaunch(path):
@app.route("/terminate/", methods=["GET"])
def do_terminate(path):
- key = get_key(path)
+ source_name = request.args.get("source_name") or default_item_source.name
+ source = matching_source(source_name)
+ key = CacheKey.for_lookup(source, source.lookup(path))
match = cache.check_entry(key)
if not match is None:
match.terminate()
return redirect(url_for("do_GET_status"), code=302)
-@app.route("/metadata/ip_address", methods=["GET"])
-def ip_address():
- resp = make_response(env.ip)
- return set_no_cache(resp)
-
-
-def main():
- logging.basicConfig(
- level=logging.INFO,
- format="%(asctime)s:%(name)s:%(levelname)s:%(message)s",
- )
+def launch():
env.validate()
+ if not item_sources or not len(item_sources):
+ raise Exception("No data sources specified for Cellxgene Gateway")
+
+ global default_item_source
+ if default_item_source is None:
+ default_item_source = item_sources[0]
+
pruner = PruneProcessCache(cache)
background_thread = Thread(target=pruner)
@@ -319,5 +282,29 @@ def main():
app.run(host="0.0.0.0", port=env.gateway_port, debug=False)
+def main():
+ logging.basicConfig(
+ level=logging.INFO,
+ format="%(asctime)s:%(name)s:%(levelname)s:%(message)s",
+ )
+ cellxgene_data = os.environ.get("CELLXGENE_DATA", None)
+ cellxgene_bucket = os.environ.get("CELLXGENE_BUCKET", None)
+
+ if cellxgene_bucket is not None:
+ from cellxgene_gateway.items.s3.s3item_source import S3ItemSource
+
+ item_sources.append(S3ItemSource(cellxgene_bucket, name="s3"))
+ default_item_source = "s3"
+ if cellxgene_data is not None:
+ from cellxgene_gateway.items.file.fileitem_source import FileItemSource
+
+ item_sources.append(FileItemSource(cellxgene_data, name="local"))
+ default_item_source = "local"
+ if len(item_sources) == 0:
+ raise Exception("Please specify CELLXGENE_DATA or CELLXGENE_BUCKET")
+
+ launch()
+
+
if __name__ == "__main__":
main()
diff --git a/cellxgene_gateway/items/file/fileitem.py b/cellxgene_gateway/items/file/fileitem.py
new file mode 100644
index 0000000..5951b2e
--- /dev/null
+++ b/cellxgene_gateway/items/file/fileitem.py
@@ -0,0 +1,27 @@
+# Copyright 2019 Novartis Institutes for BioMedical Research Inc. Licensed
+# under the Apache License, Version 2.0 (the "License"); you may not use
+# this file except in compliance with the License. You may obtain a copy
+# of the License at http://www.apache.org/licenses/LICENSE-2.0. Unless
+# required by applicable law or agreed to in writing, software distributed
+# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES
+# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
+# the specific language governing permissions and limitations under the License.
+
+import os
+
+from cellxgene_gateway.items.item import Item
+
+
+class FileItem(Item):
+ """e.g. FileItem(subpath = subpath, name = filename, type = ItemType.h5ad)
+
+ The Item superclass expects a 'name' and 'type'.
+ """
+
+ def __init__(self, subpath: str, *args, **kwargs):
+ super().__init__(*args, **kwargs)
+ self.subpath = subpath
+
+ @property
+ def descriptor(self) -> str:
+ return os.path.join(self.subpath, self.name).strip("/")
diff --git a/cellxgene_gateway/items/file/fileitem_source.py b/cellxgene_gateway/items/file/fileitem_source.py
new file mode 100644
index 0000000..d7e44f5
--- /dev/null
+++ b/cellxgene_gateway/items/file/fileitem_source.py
@@ -0,0 +1,175 @@
+# Copyright 2019 Novartis Institutes for BioMedical Research Inc. Licensed
+# under the Apache License, Version 2.0 (the "License"); you may not use
+# this file except in compliance with the License. You may obtain a copy
+# of the License at http://www.apache.org/licenses/LICENSE-2.0. Unless
+# required by applicable law or agreed to in writing, software distributed
+# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES
+# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
+# the specific language governing permissions and limitations under the License.
+
+import os
+from typing import List
+
+from cellxgene_gateway import dir_util
+from cellxgene_gateway.items.file.fileitem import FileItem
+from cellxgene_gateway.items.item import ItemTree, ItemType
+from cellxgene_gateway.items.item_source import ItemSource, LookupResult
+
+
+class FileItemSource(ItemSource):
+ def __init__(
+ self,
+ base_path,
+ name=None,
+ h5ad_suffix=dir_util.h5ad_suffix,
+ annotation_dir_suffix=dir_util.annotations_suffix,
+ annotation_file_suffix=".csv",
+ ):
+ self._name = name
+ self.base_path = base_path
+ self.h5ad_suffix = h5ad_suffix
+ self.annotation_dir_suffix = annotation_dir_suffix
+ self.annotation_file_suffix = annotation_file_suffix
+
+ @property
+ def name(self):
+ return self._name or f"Files:{self.base_path}"
+
+ def is_h5ad_file(self, path: str) -> bool:
+ return path.endswith(self.h5ad_suffix) and os.path.isfile(path)
+
+ def convert_annotation_path_to_h5ad(self, path):
+ return path[: -len(self.annotation_dir_suffix)] + self.h5ad_suffix
+
+ def convert_h5ad_path_to_annotation(self, path):
+ return path[: -len(self.h5ad_suffix)] + self.annotation_dir_suffix
+
+ def get_local_path(self, item: FileItem) -> str:
+ return os.path.join(self.base_path, item.descriptor)
+
+ def get_annotations_subpath(self, item) -> str:
+ return self.convert_h5ad_path_to_annotation(item.descriptor)
+
+ def list_items(self, filter: str = None) -> ItemTree:
+ item_tree = self.scan_directory()
+
+ """def get_items(dir):
+ if dir.branches:
+ return [*dir.items, *[item for subdir in dir.branches for item in get_items(subdir)]]
+ else:
+ return dir.items
+
+ return get_items(self.item_tree)"""
+
+ return item_tree
+
+ def scan_directory(self, subpath="") -> dict:
+ base_path = os.path.join(self.base_path, subpath)
+
+ if not os.path.exists(base_path):
+ raise Exception(f"Path for local files '{base_path}' does not exist.")
+
+ filepath_map = dict(
+ (filepath, os.path.join(base_path, filepath))
+ for filepath in sorted(os.listdir(base_path))
+ )
+
+ def is_annotation_dir(dir):
+ return (
+ dir.endswith(self.annotation_dir_suffix)
+ and self.convert_annotation_path_to_h5ad(dir) in h5ad_paths
+ )
+
+ h5ad_paths = [
+ filepath
+ for filepath, full_path in filepath_map.items()
+ if self.is_h5ad_file(full_path)
+ ]
+
+ subdirs = [
+ filepath
+ for filepath, full_path in filepath_map.items()
+ if os.path.isdir(full_path) and not is_annotation_dir(filepath)
+ ]
+
+ items = [
+ self.make_fileitem_from_path(filename, subpath) for filename in h5ad_paths
+ ]
+ branches = None
+ if len(subdirs) > 0:
+ branches = [
+ self.scan_directory(os.path.join(subpath, subdir)) for subdir in subdirs
+ ]
+
+ return ItemTree(subpath, items, branches)
+
+ def create_annotation(self, item: FileItem, name: str) -> FileItem:
+ annotation = self.make_fileitem_from_path(
+ name, self.get_annotations_subpath(item), is_annotation=True
+ )
+ item.annotations = (item.annotations or []).append(annotation)
+ return annotation
+
+ def update(self, item: FileItem) -> None:
+ pass
+
+ def full_path(self, p):
+ return os.path.join(self.base_path, p)
+
+ def lookup_item(self, descriptor):
+ full_path = self.full_path(descriptor)
+ if self.is_h5ad_file(full_path):
+ return self.shallowitem_from_descriptor(descriptor)
+
+ def lookup(self, indescriptor: str) -> LookupResult:
+ descriptor = indescriptor.strip("/")
+ if descriptor.endswith(self.annotation_file_suffix):
+ annotation_item = self.shallowitem_from_descriptor(descriptor, True)
+ h5ad_descriptor = self.convert_annotation_path_to_h5ad(
+ annotation_item.subpath
+ )
+ item = self.lookup_item(h5ad_descriptor)
+ if item is not None:
+ return LookupResult(item, annotation_item)
+ else:
+ item = self.lookup_item(descriptor)
+ if item is not None:
+ return LookupResult(item)
+
+ def shallowitem_from_descriptor(self, descriptor, is_annotation=False):
+ filename = os.path.basename(descriptor)
+ subpath = os.path.dirname(descriptor)
+ return self.make_fileitem_from_path(
+ filename,
+ subpath,
+ is_annotation,
+ True,
+ )
+
+ def make_fileitem_from_path(
+ self, filename, subpath, is_annotation=False, is_shallow=False
+ ) -> FileItem:
+ item = FileItem(
+ subpath=subpath,
+ name=filename,
+ type=ItemType.annotation if is_annotation else ItemType.h5ad,
+ )
+
+ if not is_annotation and not is_shallow:
+ annotations = self.make_annotations_for_fileitem(item)
+ item.annotations = annotations
+
+ return item
+
+ def make_annotations_for_fileitem(self, item: FileItem) -> List[FileItem]:
+ annotations_subpath = self.get_annotations_subpath(item)
+ annotations_fullpath = self.full_path(annotations_subpath)
+ if os.path.isdir(annotations_fullpath):
+ return [
+ self.make_fileitem_from_path(annotation, annotations_subpath, True)
+ for annotation in sorted(os.listdir(annotations_fullpath))
+ if annotation.endswith(self.annotation_file_suffix)
+ and os.path.isfile(os.path.join(annotations_fullpath, annotation))
+ ]
+ else:
+ return None
diff --git a/cellxgene_gateway/items/item.py b/cellxgene_gateway/items/item.py
new file mode 100644
index 0000000..a0e7718
--- /dev/null
+++ b/cellxgene_gateway/items/item.py
@@ -0,0 +1,41 @@
+# Copyright 2019 Novartis Institutes for BioMedical Research Inc. Licensed
+# under the Apache License, Version 2.0 (the "License"); you may not use
+# this file except in compliance with the License. You may obtain a copy
+# of the License at http://www.apache.org/licenses/LICENSE-2.0. Unless
+# required by applicable law or agreed to in writing, software distributed
+# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES
+# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
+# the specific language governing permissions and limitations under the License.
+
+from abc import ABC, abstractmethod
+from enum import Enum
+from typing import List
+
+
+class ItemType(Enum):
+ annotation = "annotation"
+ h5ad = "h5ad"
+
+
+class Item(ABC):
+ def __init__(self, name: str, type: ItemType, annotations: List["Item"] = None):
+ self.name = name
+ self.type = type
+ self.annotations = annotations
+
+ @property
+ @abstractmethod
+ def descriptor(self):
+ raise Exception('"descriptor" not implemented')
+
+
+class ItemTree:
+ def __init__(
+ self,
+ descriptor: str,
+ items: List[Item] = None,
+ branches: List["ItemTree"] = None,
+ ):
+ self.descriptor = descriptor
+ self.items = items
+ self.branches = branches
diff --git a/cellxgene_gateway/items/item_source.py b/cellxgene_gateway/items/item_source.py
new file mode 100644
index 0000000..e49aaa9
--- /dev/null
+++ b/cellxgene_gateway/items/item_source.py
@@ -0,0 +1,50 @@
+# Copyright 2019 Novartis Institutes for BioMedical Research Inc. Licensed
+# under the Apache License, Version 2.0 (the "License"); you may not use
+# this file except in compliance with the License. You may obtain a copy
+# of the License at http://www.apache.org/licenses/LICENSE-2.0. Unless
+# required by applicable law or agreed to in writing, software distributed
+# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES
+# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
+# the specific language governing permissions and limitations under the License.
+
+from abc import ABC, abstractmethod
+from typing import List
+
+from cellxgene_gateway.items.item import Item
+
+
+class LookupResult:
+ def __init__(self, h5ad_item: Item, annotation_item: Item = None):
+ self.h5ad_item = h5ad_item
+ self.annotation_item = annotation_item
+
+
+class ItemSource(ABC):
+ @abstractmethod
+ def list_items(self, filter: str = None) -> List[Item]:
+ raise Exception('"list_items" unimplemented')
+
+ @abstractmethod
+ def get_local_path(self, item: Item) -> str:
+ raise Exception('"local_path" unimplemented')
+
+ @abstractmethod
+ def get_annotations_subpath(self, item) -> str:
+ raise Exception('"annotations_path" unimplemented')
+
+ @abstractmethod
+ def create_annotation(self, item: Item, name: str) -> Item:
+ raise Exception('"annotation" unimplemented')
+
+ @abstractmethod
+ def update(self, item: Item) -> None:
+ raise Exception('"update" unimplemented')
+
+ @abstractmethod
+ def lookup(self, descriptor: str) -> LookupResult:
+ raise Exception('"lookup" unimplemented')
+
+ @property
+ @abstractmethod
+ def name(self):
+ pass
diff --git a/cellxgene_gateway/items/s3/s3item.py b/cellxgene_gateway/items/s3/s3item.py
new file mode 100644
index 0000000..cf2abda
--- /dev/null
+++ b/cellxgene_gateway/items/s3/s3item.py
@@ -0,0 +1,27 @@
+# Copyright 2019 Novartis Institutes for BioMedical Research Inc. Licensed
+# under the Apache License, Version 2.0 (the "License"); you may not use
+# this file except in compliance with the License. You may obtain a copy
+# of the License at http://www.apache.org/licenses/LICENSE-2.0. Unless
+# required by applicable law or agreed to in writing, software distributed
+# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES
+# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
+# the specific language governing permissions and limitations under the License.
+
+import os
+
+from cellxgene_gateway.items.item import Item
+
+
+class S3Item(Item):
+ """e.g. FileItem(subpath = subpath, name = filename, type = ItemType.h5ad)
+
+ The Item superclass expects a 'name' and 'type'.
+ """
+
+ def __init__(self, s3key: str, *args, **kwargs):
+ super().__init__(*args, **kwargs)
+ self.s3key = s3key
+
+ @property
+ def descriptor(self) -> str:
+ return self.s3key
diff --git a/cellxgene_gateway/items/s3/s3item_source.py b/cellxgene_gateway/items/s3/s3item_source.py
new file mode 100644
index 0000000..bcf7d39
--- /dev/null
+++ b/cellxgene_gateway/items/s3/s3item_source.py
@@ -0,0 +1,169 @@
+# Copyright 2019 Novartis Institutes for BioMedical Research Inc. Licensed
+# under the Apache License, Version 2.0 (the "License"); you may not use
+# this file except in compliance with the License. You may obtain a copy
+# of the License at http://www.apache.org/licenses/LICENSE-2.0. Unless
+# required by applicable law or agreed to in writing, software distributed
+# under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES
+# OR CONDITIONS OF ANY KIND, either express or implied. See the License for
+# the specific language governing permissions and limitations under the License.
+
+from os.path import basename, dirname, join
+from typing import List
+
+import s3fs
+
+from cellxgene_gateway import dir_util
+from cellxgene_gateway.items.item import ItemTree, ItemType
+from cellxgene_gateway.items.item_source import ItemSource, LookupResult
+from cellxgene_gateway.items.s3.s3item import S3Item
+
+
+class S3ItemSource(ItemSource):
+ def __init__(
+ self,
+ bucket,
+ name=None,
+ h5ad_suffix=dir_util.h5ad_suffix,
+ annotation_dir_suffix=dir_util.annotations_suffix,
+ annotation_file_suffix=".csv",
+ ):
+ self._name = name
+ self.s3 = s3fs.S3FileSystem()
+ self.bucket = bucket
+ self.h5ad_suffix = h5ad_suffix
+ self.annotation_dir_suffix = annotation_dir_suffix
+ self.annotation_file_suffix = annotation_file_suffix
+
+ def url(self, path):
+ return "s3://" + join(self.bucket, path)
+
+ @property
+ def name(self):
+ return self._name or f"Items:{self.url('')}"
+
+ def is_h5ad_url(self, s3url: str) -> bool:
+ return s3url.endswith(self.h5ad_suffix) and self.s3.exists(s3url)
+
+ def convert_annotation_key_to_h5ad(self, s3key):
+ return s3key[: -len(self.annotation_dir_suffix)] + self.h5ad_suffix
+
+ def convert_h5ad_key_to_annotation(self, s3key):
+ return s3key[: -len(self.h5ad_suffix)] + self.annotation_dir_suffix
+
+ def get_local_path(self, item: S3Item) -> str:
+ return self.url(item.descriptor)
+
+ def get_annotations_subpath(self, item) -> str:
+ return self.convert_h5ad_key_to_annotation(item.descriptor)
+
+ def list_items(self, filter: str = None) -> ItemTree:
+ item_tree = self.scan_directory()
+ return item_tree
+
+ def scan_directory(self, subpath="") -> dict:
+ url = self.url(subpath)
+
+ if not self.s3.exists(url):
+ raise Exception(f"S3 url '{url}' does not exist.")
+
+ s3key_map = dict(
+ (filepath[len(self.bucket) :], "s3://" + filepath)
+ for filepath in sorted(self.s3.ls(url))
+ )
+
+ def is_annotation_dir(dir_s3key):
+ return (
+ dir_s3key.endswith(self.annotation_dir_suffix)
+ and self.convert_annotation_key_to_h5ad(dir_s3key) in h5ad_paths
+ )
+
+ h5ad_paths = [
+ filepath
+ for filepath, item_url in s3key_map.items()
+ if self.is_h5ad_url(item_url)
+ ]
+
+ subdirs = [
+ filepath
+ for filepath, item_url in s3key_map.items()
+ if self.s3.isdir(item_url) and not is_annotation_dir(filepath)
+ ]
+
+ items = [
+ self.make_s3item_from_key(filename, join(subpath, filename))
+ for filename in h5ad_paths
+ ]
+ branches = None
+ if len(subdirs) > 0:
+ branches = [
+ self.scan_directory(join(subpath, subdir)) for subdir in subdirs
+ ]
+
+ return ItemTree(subpath, items, branches)
+
+ def create_annotation(self, item: S3Item, name: str) -> S3Item:
+ annotation = self.make_s3item_from_key(
+ name, self.get_annotations_subpath(item), is_annotation=True
+ )
+ item.annotations = (item.annotations or []).append(annotation)
+ return annotation
+
+ def update(self, item: S3Item) -> None:
+ pass
+
+ def lookup_item(self, descriptor):
+ full_path = self.url(descriptor)
+ if self.is_h5ad_url(full_path):
+ return self.shallowitem_from_descriptor(descriptor)
+
+ def lookup(self, indescriptor: str) -> LookupResult:
+ descriptor = indescriptor.strip("/")
+ if descriptor.endswith(self.annotation_file_suffix):
+ annotation_item = self.shallowitem_from_descriptor(descriptor, True)
+ if not self.s3.exists(self.url(annotation_item.s3key)):
+ with self.s3.open(self.url(annotation_item.s3key), "w") as f:
+ f.write("")
+ h5ad_descriptor = self.convert_annotation_key_to_h5ad(
+ dirname(annotation_item.s3key)
+ )
+ item = self.shallowitem_from_descriptor(h5ad_descriptor)
+ return LookupResult(item, annotation_item)
+ else:
+ item = self.lookup_item(descriptor)
+ if item is not None:
+ return LookupResult(item)
+
+ def shallowitem_from_descriptor(self, descriptor, is_annotation=False):
+ return self.make_s3item_from_key(
+ basename(descriptor), descriptor, is_annotation, True
+ )
+
+ def make_s3item_from_key(
+ self, name, s3key, is_annotation=False, is_shallow=False
+ ) -> S3Item:
+ item = S3Item(
+ s3key=s3key,
+ name=name,
+ type=ItemType.annotation if is_annotation else ItemType.h5ad,
+ )
+
+ if not is_annotation and not is_shallow:
+ annotations = self.make_annotations_for_fileitem(item)
+ item.annotations = annotations
+
+ return item
+
+ def make_annotations_for_fileitem(self, item: S3Item) -> List[S3Item]:
+ annotations_subpath = self.get_annotations_subpath(item)
+ annotations_fullpath = self.url(annotations_subpath)
+ if self.s3.isdir(annotations_fullpath):
+ return [
+ self.make_s3item_from_key(
+ annotation, join(annotations_subpath, annotation), True
+ )
+ for annotation in sorted(self.s3.ls(annotations_fullpath))
+ if annotation.endswith(self.annotation_file_suffix)
+ and self.s3.isfile(join(annotations_fullpath, annotation))
+ ]
+ else:
+ return None
diff --git a/cellxgene_gateway/path_util.py b/cellxgene_gateway/path_util.py
index 1b23bc8..42ded92 100644
--- a/cellxgene_gateway/path_util.py
+++ b/cellxgene_gateway/path_util.py
@@ -11,43 +11,11 @@ import os
from flask_api import status
-from cellxgene_gateway import env
from cellxgene_gateway.cache_key import CacheKey
from cellxgene_gateway.cellxgene_exception import CellxgeneException
from cellxgene_gateway.dir_util import make_h5ad
-def get_key(path):
- if path == "/" or path == "":
- raise CellxgeneException(
- "No matching dataset found.", status.HTTP_404_NOT_FOUND
- )
-
- trimmed = path[:-1] if path[-1] == "/" else path
- try:
- # valid paths come in three forms:
- if trimmed.endswith(".h5ad") and data_file_exists(trimmed):
- # 1) somedir/dataset.h5ad: a dataset
- return CacheKey(trimmed, trimmed, None)
- elif trimmed.endswith(".csv"):
-
- # 2) somedir/dataset_annotations/my_annotations.csv : an actual annotations file.
- annotations_dir = os.path.split(trimmed)[0]
- dataset = make_h5ad(annotations_dir)
- if data_file_exists(dataset):
- data_dir_ensure(annotations_dir)
- return CacheKey(trimmed, dataset, trimmed)
- elif trimmed.endswith("_annotations") and data_dir_exists(trimmed):
- # 3) somedir/dataset_annotations: an annotation directory. The corresponding h5ad must exist, but the directory may not.
- dataset = make_h5ad(trimmed)
- if data_file_exists(dataset):
- return CacheKey(trimmed, dataset, "")
- except CellxgeneException:
- pass
- split = os.path.split(trimmed)
- return get_key(split[0])
-
-
def validate_exists(file_path):
if not os.path.exists(file_path):
raise CellxgeneException(
@@ -71,37 +39,3 @@ def validate_is_dir(file_path):
"Path is not dir: " + file_path, status.HTTP_400_BAD_REQUEST
)
return
-
-
-def data_file_exists(dataset):
- file_path = os.path.join(env.cellxgene_data, dataset)
- validate_is_file(file_path)
- return True
-
-
-def data_dir_exists(dataset):
- file_path = os.path.join(env.cellxgene_data, dataset)
- validate_is_dir(file_path)
- return True
-
-
-def data_dir_ensure(dataset):
- file_path = os.path.join(env.cellxgene_data, dataset)
- if not os.path.exists(file_path):
- os.makedirs(file_path)
-
-
-def get_file_path(key):
- dataset = key.dataset
- file_path = os.path.join(env.cellxgene_data, dataset)
- validate_is_file(file_path)
- return file_path
-
-
-def get_annotation_file_path(key):
- if key.annotation_file is None:
- return None
- if key.annotation_file == "":
- return ""
- file_path = os.path.join(env.cellxgene_data, key.annotation_file)
- return file_path
diff --git a/cellxgene_gateway/static/js/annotation.js b/cellxgene_gateway/static/js/annotation.js
index 73c1739..7b1032d 100644
--- a/cellxgene_gateway/static/js/annotation.js
+++ b/cellxgene_gateway/static/js/annotation.js
@@ -1,4 +1,5 @@
// neandertal javascript
+// TODO: rewrite this --
const new_annotation_callback = (() =>{
const suffix = `.csv`;
return (e) => {
diff --git a/cellxgene_gateway/subprocess_backend.py b/cellxgene_gateway/subprocess_backend.py
index 7e67959..c9303c9 100644
--- a/cellxgene_gateway/subprocess_backend.py
+++ b/cellxgene_gateway/subprocess_backend.py
@@ -15,7 +15,6 @@ from flask_api import status
from cellxgene_gateway.cache_entry import CacheEntryStatus
from cellxgene_gateway.dir_util import make_annotations
from cellxgene_gateway.env import cellxgene_args, enable_annotations, enable_backed_mode
-from cellxgene_gateway.path_util import get_annotation_file_path, get_file_path
from cellxgene_gateway.process_exception import ProcessException
@@ -38,8 +37,7 @@ class SubprocessBackend:
cmd = (
f"yes | {cellxgene_loc} launch {file_path}"
- + " --port "
- + str(port)
+ + f" --port {port}"
+ " --host 127.0.0.1"
+ extra_args
)
@@ -50,13 +48,12 @@ class SubprocessBackend:
return cmd
def launch(self, cellxgene_loc, scripts, cache_entry):
-
cmd = self.create_cmd(
cellxgene_loc,
- get_file_path(cache_entry.key),
+ cache_entry.key.file_path,
cache_entry.port,
scripts,
- get_annotation_file_path(cache_entry.key),
+ cache_entry.key.annotation_file_path,
)
logging.getLogger("cellxgene_gateway").info(f"launching {cmd}")
process = subprocess.Popen(
diff --git a/cellxgene_gateway/templates/cache_status.html b/cellxgene_gateway/templates/cache_status.html
index 956fe56..5e2ff41 100644
--- a/cellxgene_gateway/templates/cache_status.html
+++ b/cellxgene_gateway/templates/cache_status.html
@@ -10,59 +10,68 @@
-->
+
Cellxgene Gateway - FILE CRAWLER
- {% for script in extra_scripts %}
-
- {% endfor %}
-
+ {% for script in extra_scripts %}
+
+ {% endfor %}
+
+
- Cellxgene Gateway - Cache Status
+ Cellxgene Gateway - Cache Status
-
- | PID |
- dataset |
- annotation_file |
- port |
- launchtime |
- last access |
- status |
- message |
- http_status |
- actions |
-
-
-
- {% for entry in entry_list %}
-
- | {{ entry.pid }} |
- {{ entry.key.dataset }} |
- {{ entry.key.annotation_file }} |
- {{ entry.port }} |
- {{ entry.launchtime }} |
- {{ entry.timestamp }} |
- {{ entry.status.name }} |
- {{ entry.message }} |
- {{ entry.http_status }} |
-
- {% if entry.status.name == 'loaded' %}
- terminate
- {% endif %}
- |
-
- {% endfor %}
-
+
+ | PID |
+ dataset |
+ annotation_file |
+ source |
+ port |
+ launchtime |
+ last access |
+ status |
+ message |
+ http_status |
+ actions |
+
+
+
+ {% for entry in entry_list %}
+
+ | {{ entry.pid }} |
+ {{ entry.key.h5ad_item.descriptor }}
+ |
+ {{ entry.key.annotation_descriptor }} |
+ {{ entry.source_name }} |
+ {{ entry.port }} |
+ {{ entry.launchtime }} |
+ {{ entry.timestamp }} |
+ {{ entry.status.name }} |
+ {{ entry.message }} |
+ {{ entry.http_status }} |
+
+ {% if entry.status.name == 'loaded' %}
+
+ terminate
+ {% endif %}
+ |
+
+ {% endfor %}
+
-
+
+