Refactor czi_hosted and server into backend directory, pull common code into backend/common, refactor tests (#2102)

* move local_server -> backend/server server-> backend/czi_hosted, pull common code into backend/common update imports, tests and make commands
This commit is contained in:
Madison Dunitz
2021-03-26 00:27:07 -05:00
committed by GitHub
parent e6e358ddc8
commit 78c9d24ed4
425 changed files with 734 additions and 5317 deletions
+11
View File
@@ -0,0 +1,11 @@
import random
import string
from os import popen
PROJECT_ROOT = popen("git rev-parse --show-toplevel").read().strip()
FIXTURES_ROOT = PROJECT_ROOT + "/backend/test/fixtures"
H5AD_FIXTURE = FIXTURES_ROOT + "/pbmc3k-CSC-gz.h5ad"
def random_string(n):
return "".join(random.choice(string.ascii_letters) for _ in range(n))
+31
View File
@@ -0,0 +1,31 @@
"""
Code to decode, for testing purposes, the flatbuffer encoded blobs.
This code will need to be updated if fbs/matrix.fbs changes. For more information, see fbs/matrix.fbs and
server/data_common/fbs/
"""
import backend.common.fbs.NetEncoding.Matrix as Matrix
from backend.common.fbs.matrix import deserialize_typed_array
def decode_matrix_FBS(buf):
"""
Given a FBS Matrix, return an decoded Python dict containing same info in native format.
NOTE / TODO: row_idx not currently implemented
"""
df = Matrix.Matrix.GetRootAsMatrix(buf, 0)
n_rows = df.NRows()
n_cols = df.NCols()
columns_length = df.ColumnsLength()
decoded_columns = []
for col_idx in range(0, columns_length):
col = df.Columns(col_idx)
tarr = (col.UType(), col.U())
decoded_columns.append(deserialize_typed_array(tarr))
cidx = deserialize_typed_array((df.ColIndexType(), df.ColIndex()))
return {"n_rows": n_rows, "n_cols": n_cols, "columns": decoded_columns, "col_idx": cidx, "row_idx": None}
Binary file not shown.
View File
Binary file not shown.
@@ -0,0 +1,37 @@
f"""
dataset:
app:
scripts: {scripts} #list of strs (filenames) or dicts containing keys
inline_scripts: {inline_scripts} #list of strs (filenames)
about_legal_tos: {about_legal_tos}
about_legal_privacy: {about_legal_privacy}
authentication_enable: {authentication_enable}
presentation:
max_categories: {max_categories}
custom_colors: {custom_colors}
user_annotations:
enable: {enable_users_annotations}
type: {annotation_type}
hosted_tiledb_array:
db_uri: {db_uri}
hosted_file_directory: {hosted_file_directory}
local_file_csv:
directory: {local_file_csv_directory}
file: {local_file_csv_file}
ontology:
enable: {ontology_enabled}
obo_location: {obo_location}
embeddings:
names: {embedding_names}
enable_reembedding: {enable_reembedding}
diffexp:
enable: {enable_difexp}
lfc_cutoff: {lfc_cutoff}
top_n: {top_n}
"""
@@ -0,0 +1,63 @@
f"""server:
app:
verbose: {verbose}
debug: {debug}
host: {host}
port: {port}
open_browser: {open_browser}
force_https: {force_https}
flask_secret_key: {flask_secret_key}
generate_cache_control_headers: {generate_cache_control_headers}
server_timing_headers: {server_timing_headers}
csp_directives: {csp_directives}
api_base_url: {api_base_url}
web_base_url: {web_base_url}
authentication:
type: {auth_type}
insecure_test_environment: {insecure_test_environment}
params_oauth:
oauth_api_base_url: {oauth_api_base_url}
client_id: {client_id}
client_secret: {client_secret}
jwt_decode_options: {jwt_decode_options}
session_cookie: {session_cookie}
cookie: {cookie}
multi_dataset:
dataroot: {dataroot}
index: {index}
allowed_matrix_types: {allowed_matrix_types}
matrix_cache:
max_datasets: {max_cached_datasets}
timelimit_s: {timelimit_s}
single_dataset:
datapath: {dataset_datapath}
obs_names: {obs_names}
var_names: {var_names}
about: {about}
title: {title}
diffexp:
alg_cxg: # number of threads to use is computed from: min(max_workers, cpu_multipler * cpu_count)
max_workers: {diffexp_max_workers}
cpu_multiplier: {cpu_multiplier}
target_workunit: {target_workunit} # The target number of matrix elements that are evaluated in one thread.
data_locator:
s3:
region_name: {data_locater_region_name}
adaptor:
cxg_adaptor:
tiledb_ctx:
sm.tile_cache_size: {cxg_tile_cache_size}
sm.num_reader_threads: {cxg_num_reader_threads}
anndata_adaptor:
backed: {anndata_backed}
limits:
column_request_max: {column_request_max}
diffexp_cellcount_max: {diffexp_cellcount_max}
"""
+79
View File
@@ -0,0 +1,79 @@
import string
import random
from sqlalchemy import func
from backend.czi_hosted.db.cellxgene_orm import CellxGeneUser, CellxGeneDataset, Annotation, Base
from backend.czi_hosted.db.create_db import create_db
from backend.czi_hosted.db.db_utils import DbUtils
class TestDatabase:
def __init__(self):
local_db_uri = "postgresql://postgres:test_pw@localhost:5432"
create_db(local_db_uri)
self.db = DbUtils(local_db_uri)
self._populate_test_data()
self._populate_test_data_many()
def _populate_test_data(self):
self._create_test_user()
self._create_test_dataset()
self._create_test_annotation()
def _populate_test_data_many(self):
self._create_test_users()
self._create_test_datasets()
self._create_test_annotations()
def _create_test_user(self):
user = CellxGeneUser(id="test_user_id")
user2 = CellxGeneUser(id="1234")
self.db.session.add(user)
self.db.session.add(user2)
self.db.session.commit()
def _create_test_dataset(self):
dataset = CellxGeneDataset(name="test_dataset",)
self.db.session.add(dataset)
self.db.session.commit()
def _create_test_annotation(self):
dataset = self.db.query([CellxGeneDataset], [CellxGeneDataset.name == "test_dataset"],)[0]
annotation = Annotation(tiledb_uri="tiledb_uri", user_id="test_user_id", dataset_id=str(dataset.id))
self.db.session.add(annotation)
self.db.session.commit()
@staticmethod
def get_random_string():
letters = string.ascii_lowercase
return "".join(random.choice(letters) for i in range(12))
def _create_test_users(self, user_count: int = 10):
users = []
for i in range(user_count):
users.append(CellxGeneUser(id=self.get_random_string()))
self.db.session.add_all(users)
self.db.session.commit()
def _create_test_datasets(self, dataset_count: int = 10):
datasets = []
for i in range(dataset_count):
datasets.append(CellxGeneDataset(name=self.get_random_string()))
self.db.session.add_all(datasets)
self.db.session.commit()
def order_by_random(self, table: Base):
return self.db.session.query(table).order_by(func.random()).first()
def _create_test_annotations(self, annotation_count: int = 10):
annotations = []
for i in range(annotation_count):
dataset = self.order_by_random(CellxGeneDataset)
user = self.order_by_random(CellxGeneUser)
annotations.append(
Annotation(tiledb_uri=self.get_random_string(), user_id=user.id, dataset_id=str(dataset.id))
)
self.db.session.add_all(annotations)
self.db.session.commit()
+34
View File
@@ -0,0 +1,34 @@
f"""
dataset:
app:
scripts: {scripts} #list of strs (filenames) or dicts containing keys
inline_scripts: {inline_scripts} #list of strs (filenames)
authentication_enable: {authentication_enable}
presentation:
max_categories: {max_categories}
custom_colors: {custom_colors}
user_annotations:
enable: {enable_users_annotations}
type: {annotation_type}
local_file_csv:
directory: {local_file_csv_directory}
file: {local_file_csv_file}
gene_sets_file: {local_file_csv_gene_sets_file}
ontology:
enable: {ontology_enabled}
obo_location: {obo_location}
gene_sets:
readonly: {gene_sets_readonly}
embeddings:
names: {embedding_names}
enable_reembedding: {enable_reembedding}
diffexp:
enable: {enable_difexp}
lfc_cutoff: {lfc_cutoff}
top_n: {top_n}
"""
+12
View File
@@ -0,0 +1,12 @@
pbmc3k_colors = {
"louvain": {
"B cells": "#2ca02c",
"CD14+ Monocytes": "#ff7f0e",
"CD4 T cells": "#1f77b4",
"CD8 T cells": "#d62728",
"Dendritic cells": "#e377c2",
"FCGR3A+ Monocytes": "#8c564b",
"Megakaryocytes": "#bcbd22",
"NK cells": "#9467bd",
}
}
Binary file not shown.
BIN
View File
Binary file not shown.
Binary file not shown.
Binary file not shown.
File diff suppressed because it is too large Load Diff
+16
View File
@@ -0,0 +1,16 @@
# Test fixture
gene_set_name, gene_set_description, gene_symbol, gene_description
first gene set name,,F5, a gene_description
first gene set name,a description, NO_SUCH_GENE, non-existent gene
first gene set name,a description, F5, duplicate gene
first gene set name, a description, SUMO3,
first gene set name,, SRM,
second gene set,,RER1
second gene set,,SIK1
third gene set,,NO_SUCH_GENE
fourth_gene_set,fourth description,,gene intentionally missing
fifth_dataset,,,
summary test,,ACD,
summary test,,AATF,
summary test,,F5,
summary test,,PIGU,
1 # Test fixture
2 gene_set_name, gene_set_description, gene_symbol, gene_description
3 first gene set name,,F5, a gene_description
4 first gene set name,a description, NO_SUCH_GENE, non-existent gene
5 first gene set name,a description, F5, duplicate gene
6 first gene set name, a description, SUMO3,
7 first gene set name,, SRM,
8 second gene set,,RER1
9 second gene set,,SIK1
10 third gene set,,NO_SUCH_GENE
11 fourth_gene_set,fourth description,,gene intentionally missing
12 fifth_dataset,,,
13 summary test,,ACD,
14 summary test,,AATF,
15 summary test,,F5,
16 summary test,,PIGU,
Binary file not shown.
View File
View File
Binary file not shown.
View File
View File
Binary file not shown.
View File
Binary file not shown.
View File
Binary file not shown.
View File
Binary file not shown.
View File
Binary file not shown.
View File
Binary file not shown.
View File
Binary file not shown.
View File
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.

Some files were not shown because too many files have changed in this diff Show More