mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-04 13:58:11 +08:00
Split out the local backend (#2052)
This splits the backend into two parts: the local backend for desktop cellxgene and the AWS backend for hosted cellxgene. The local backend is in local_server while the hosted remains in server. The general idea is to copy everything from server to local_server, pull unneeded stuff out of local_server, and keep server as-is for this PR. Not touching server means all the infra and deployment code will continue working just as it did before so we can make those changes incrementally.
This commit is contained in:
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,31 @@
|
||||
f"""
|
||||
dataset:
|
||||
app:
|
||||
scripts: {scripts} #list of strs (filenames) or dicts containing keys
|
||||
inline_scripts: {inline_scripts} #list of strs (filenames)
|
||||
|
||||
authentication_enable: {authentication_enable}
|
||||
|
||||
presentation:
|
||||
max_categories: {max_categories}
|
||||
custom_colors: {custom_colors}
|
||||
|
||||
user_annotations:
|
||||
enable: {enable_users_annotations}
|
||||
type: {annotation_type}
|
||||
local_file_csv:
|
||||
directory: {local_file_csv_directory}
|
||||
file: {local_file_csv_file}
|
||||
ontology:
|
||||
enable: {ontology_enabled}
|
||||
obo_location: {obo_location}
|
||||
|
||||
embeddings:
|
||||
names: {embedding_names}
|
||||
enable_reembedding: {enable_reembedding}
|
||||
|
||||
diffexp:
|
||||
enable: {enable_difexp}
|
||||
lfc_cutoff: {lfc_cutoff}
|
||||
top_n: {top_n}
|
||||
"""
|
||||
+12
@@ -0,0 +1,12 @@
|
||||
pbmc3k_colors = {
|
||||
"louvain": {
|
||||
"B cells": "#2ca02c",
|
||||
"CD14+ Monocytes": "#ff7f0e",
|
||||
"CD4 T cells": "#1f77b4",
|
||||
"CD8 T cells": "#d62728",
|
||||
"Dendritic cells": "#e377c2",
|
||||
"FCGR3A+ Monocytes": "#8c564b",
|
||||
"Megakaryocytes": "#bcbd22",
|
||||
"NK cells": "#9467bd",
|
||||
}
|
||||
}
|
||||
BIN
Binary file not shown.
Vendored
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
+2641
File diff suppressed because it is too large
Load Diff
+83
@@ -0,0 +1,83 @@
|
||||
{
|
||||
"dataframe": {
|
||||
"nObs": 2638,
|
||||
"nVar": 1838,
|
||||
"type": "float32"
|
||||
},
|
||||
"annotations": {
|
||||
"obs": {
|
||||
"index": "name_0",
|
||||
"columns": [
|
||||
{
|
||||
"name": "name_0",
|
||||
"type": "string",
|
||||
"writable": false
|
||||
},
|
||||
{
|
||||
"name": "n_genes",
|
||||
"type": "int32",
|
||||
"writable": false
|
||||
},
|
||||
{
|
||||
"name": "percent_mito",
|
||||
"type": "float32",
|
||||
"writable": false
|
||||
},
|
||||
{
|
||||
"name": "n_counts",
|
||||
"type": "float32",
|
||||
"writable": false
|
||||
},
|
||||
{
|
||||
"name": "louvain",
|
||||
"type": "categorical",
|
||||
"categories": [
|
||||
"CD4 T cells",
|
||||
"CD14+ Monocytes",
|
||||
"B cells",
|
||||
"CD8 T cells",
|
||||
"NK cells",
|
||||
"FCGR3A+ Monocytes",
|
||||
"Dendritic cells",
|
||||
"Megakaryocytes"
|
||||
],
|
||||
"writable": false
|
||||
}
|
||||
]
|
||||
},
|
||||
"var": {
|
||||
"index": "name_0",
|
||||
"columns": [
|
||||
{
|
||||
"name": "name_0",
|
||||
"type": "string",
|
||||
"writable": false
|
||||
},
|
||||
{
|
||||
"name": "n_cells",
|
||||
"type": "int32",
|
||||
"writable": false
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
"layout": {
|
||||
"obs": [
|
||||
{
|
||||
"name": "umap",
|
||||
"type": "float32",
|
||||
"dims": ["umap_0", "umap_1"]
|
||||
},
|
||||
{
|
||||
"name": "tsne",
|
||||
"type": "float32",
|
||||
"dims": ["tsne_0", "tsne_1"]
|
||||
},
|
||||
{
|
||||
"name": "pca",
|
||||
"type": "float32",
|
||||
"dims": ["pca_0", "pca_1"]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
+139
@@ -0,0 +1,139 @@
|
||||
#!/bin/bash
|
||||
wget "https://s3-us-west-2.amazonaws.com/10x.files/samples/cell/pbmc3k/pbmc3k_filtered_gene_bc_matrices.tar.gz"
|
||||
tar xf "pbmc3k_filtered_gene_bc_matrices.tar.gz"
|
||||
|
||||
python3 - <<MERGE_GENES
|
||||
import os
|
||||
from scipy.io import mmread, mmwrite
|
||||
import scipy.sparse
|
||||
import pandas as pd
|
||||
from local_server.converters.schema import gene_symbol
|
||||
|
||||
mat = mmread("filtered_gene_bc_matrices/hg19/matrix.mtx").todense()
|
||||
genes = pd.read_csv("filtered_gene_bc_matrices/hg19/genes.tsv", sep='\t', names=["gene_id", "gene_symbol"])
|
||||
|
||||
upgraded_genes = gene_symbol.get_upgraded_var_index(pd.DataFrame(index=genes["gene_symbol"]))
|
||||
df = pd.DataFrame(data=mat, index=upgraded_genes).T
|
||||
merged = df.sum(axis=1, level=0, skipna=False)
|
||||
|
||||
os.makedirs("merged")
|
||||
merged.columns.to_frame().to_csv("merged/genes.tsv", index=False, header=False)
|
||||
mmwrite("merged/matrix.mtx", scipy.sparse.coo_matrix(merged).T)
|
||||
MERGE_GENES
|
||||
|
||||
cp "filtered_gene_bc_matrices/hg19/barcodes.tsv" "merged/barcodes.tsv"
|
||||
awk '{print $1"\t"$1}' merged/genes.tsv > genes_tmp.tsv; mv genes_tmp.tsv merged/genes.tsv
|
||||
|
||||
echo -e "\n\n\nRunning tutorial on original\n\n\n"
|
||||
Rscript - <<TUTORIAL
|
||||
library(Seurat)
|
||||
|
||||
pbmc.data <- Read10X(data.dir = "filtered_gene_bc_matrices/hg19/")
|
||||
pbmc <- CreateSeuratObject(counts = pbmc.data, project = "pbmc3k", min.features = 200)
|
||||
pbmc <- NormalizeData(pbmc, normalization.method = "LogNormalize", scale.factor = 10000)
|
||||
pbmc <- FindVariableFeatures(pbmc, selection.method = "vst", nfeatures = 2000)
|
||||
pbmc[["percent.mt"]] <- PercentageFeatureSet(pbmc, pattern = "^MT-")
|
||||
all.genes <- rownames(pbmc)
|
||||
pbmc <- ScaleData(pbmc, features = all.genes)
|
||||
|
||||
pbmc <- RunPCA(pbmc, features = VariableFeatures(object = pbmc))
|
||||
pbmc <- FindNeighbors(pbmc, dims = 1:10)
|
||||
pbmc <- FindClusters(pbmc, resolution = 0.5)
|
||||
pbmc <- RunUMAP(pbmc, dims = 1:10)
|
||||
saveRDS(pbmc, file = "./seurat_tutorial.rds")
|
||||
TUTORIAL
|
||||
|
||||
echo -e "\n\n\nRunning tutorial on merged\n\n\n"
|
||||
Rscript - <<TUTORIAL_MERGED
|
||||
library(Seurat)
|
||||
|
||||
pbmc.data <- Read10X(data.dir = "merged/")
|
||||
pbmc <- CreateSeuratObject(counts = pbmc.data, project = "pbmc3k", min.features = 200)
|
||||
pbmc <- NormalizeData(pbmc, normalization.method = "LogNormalize", scale.factor = 10000)
|
||||
pbmc <- FindVariableFeatures(pbmc, selection.method = "vst", nfeatures = 2000)
|
||||
pbmc[["percent.mt"]] <- PercentageFeatureSet(pbmc, pattern = "^MT-")
|
||||
all.genes <- rownames(pbmc)
|
||||
pbmc <- ScaleData(pbmc, features = all.genes)
|
||||
|
||||
pbmc <- RunPCA(pbmc, features = VariableFeatures(object = pbmc))
|
||||
pbmc <- FindNeighbors(pbmc, dims = 1:10)
|
||||
pbmc <- FindClusters(pbmc, resolution = 0.5)
|
||||
pbmc <- RunUMAP(pbmc, dims = 1:10)
|
||||
saveRDS(pbmc, file = "./seurat_tutorial_merged.rds")
|
||||
TUTORIAL_MERGED
|
||||
|
||||
echo -e "\n\n\nRunning SCTransform on original\n\n\n"
|
||||
Rscript - <<SCTRANSFORM
|
||||
library(Seurat)
|
||||
library(sctransform)
|
||||
|
||||
pbmc.data <- Read10X(data.dir = "filtered_gene_bc_matrices/hg19/")
|
||||
pbmc <- CreateSeuratObject(counts = pbmc.data)
|
||||
pbmc <- PercentageFeatureSet(pbmc, pattern = "^MT-", col.name = "percent.mt")
|
||||
pbmc <- SCTransform(pbmc, vars.to.regress = "percent.mt", verbose = FALSE)
|
||||
pbmc <- RunPCA(pbmc, verbose = FALSE)
|
||||
pbmc <- RunUMAP(pbmc, dims = 1:30, verbose = FALSE)
|
||||
pbmc <- FindNeighbors(pbmc, dims = 1:30, verbose = FALSE)
|
||||
pbmc <- FindClusters(pbmc, verbose = FALSE)
|
||||
saveRDS(pbmc, file = "./sctransform.rds")
|
||||
SCTRANSFORM
|
||||
|
||||
echo -e "\n\n\nRunning SCTransform on merged\n\n\n"
|
||||
Rscript - <<SCTRANSFORM_MERGED
|
||||
library(Seurat)
|
||||
library(sctransform)
|
||||
|
||||
pbmc.data <- Read10X(data.dir = "merged/")
|
||||
pbmc <- CreateSeuratObject(counts = pbmc.data)
|
||||
pbmc <- PercentageFeatureSet(pbmc, pattern = "^MT-", col.name = "percent.mt")
|
||||
pbmc <- SCTransform(pbmc, vars.to.regress = "percent.mt", verbose = FALSE)
|
||||
pbmc <- RunPCA(pbmc, verbose = FALSE)
|
||||
pbmc <- RunUMAP(pbmc, dims = 1:30, verbose = FALSE)
|
||||
pbmc <- FindNeighbors(pbmc, dims = 1:30, verbose = FALSE)
|
||||
pbmc <- FindClusters(pbmc, verbose = FALSE)
|
||||
saveRDS(pbmc, file = "./sctransform_merged.rds")
|
||||
SCTRANSFORM_MERGED
|
||||
|
||||
echo -e "\n\n\nConverting\n\n\n"
|
||||
Rscript - <<SCEASY
|
||||
library(sceasy)
|
||||
srt <- readRDS("seurat_tutorial.rds")
|
||||
sceasy::convertFormat(srt,
|
||||
outFile = "seurat_tutorial.h5ad",
|
||||
from = "seurat",
|
||||
to = "anndata",
|
||||
assay = "RNA",
|
||||
main_layer = "data",
|
||||
transfer_layers = c("data", "counts", "scale.data"),
|
||||
drop_single_values = FALSE)
|
||||
|
||||
srt <- readRDS("seurat_tutorial_merged.rds")
|
||||
sceasy::convertFormat(srt,
|
||||
outFile = "seurat_tutorial_merged.h5ad",
|
||||
from = "seurat",
|
||||
to = "anndata",
|
||||
assay = "RNA",
|
||||
main_layer = "data",
|
||||
transfer_layers = c("data", "counts", "scale.data"),
|
||||
drop_single_values = FALSE)
|
||||
|
||||
srt <- readRDS("sctransform.rds")
|
||||
sceasy::convertFormat(srt,
|
||||
outFile = "sctransform.h5ad",
|
||||
from = "seurat",
|
||||
to = "anndata",
|
||||
assay = "SCT",
|
||||
main_layer = "data",
|
||||
transfer_layers = c("data", "counts", "scale.data"),
|
||||
drop_single_values = FALSE)
|
||||
|
||||
srt <- readRDS("sctransform_merged.rds")
|
||||
sceasy::convertFormat(srt,
|
||||
outFile = "sctransform_merged.h5ad",
|
||||
from = "seurat",
|
||||
to = "anndata",
|
||||
assay = "SCT",
|
||||
main_layer = "data",
|
||||
transfer_layers = c("data", "counts", "scale.data"),
|
||||
drop_single_values = FALSE)
|
||||
SCEASY
|
||||
@@ -0,0 +1,32 @@
|
||||
f"""server:
|
||||
app:
|
||||
verbose: {verbose}
|
||||
debug: {debug}
|
||||
host: {host}
|
||||
port: {port}
|
||||
open_browser: {open_browser}
|
||||
force_https: {force_https}
|
||||
flask_secret_key: {flask_secret_key}
|
||||
authentication:
|
||||
type: {auth_type}
|
||||
insecure_test_environment: {insecure_test_environment}
|
||||
|
||||
single_dataset:
|
||||
datapath: {dataset_datapath}
|
||||
obs_names: {obs_names}
|
||||
var_names: {var_names}
|
||||
about: {about}
|
||||
title: {title}
|
||||
|
||||
data_locator:
|
||||
s3:
|
||||
region_name: {data_locater_region_name}
|
||||
|
||||
adaptor:
|
||||
anndata_adaptor:
|
||||
backed: {anndata_backed}
|
||||
|
||||
limits:
|
||||
column_request_max: {column_request_max}
|
||||
diffexp_cellcount_max: {diffexp_cellcount_max}
|
||||
"""
|
||||
+34
@@ -0,0 +1,34 @@
|
||||
fixup_gene_symbols:
|
||||
X: log1p
|
||||
obs:
|
||||
cell_type_ontology_term_id:
|
||||
louvain:
|
||||
CD4 T cells: CL:00001
|
||||
B cells: CL:00002
|
||||
CD14+ Monocytes: CL:00003
|
||||
NK cells: CL:00004
|
||||
CD8 T cells: CL:00005
|
||||
FCGR3A+ Monocytes: CL:00006
|
||||
Dendritic cells: CL:00007
|
||||
Megakaryocytes: CL:00008
|
||||
tissue_ontology_term_id: UBERON:12345
|
||||
assay_ontology_term_id: EFO:12345
|
||||
disease_ontology_term_id: MONDO:12345
|
||||
ethnicity_ontology_term_id: MANCESTRO:12345
|
||||
development_stage_ontology_term_id: HsapDv:12345
|
||||
sex: other
|
||||
uns:
|
||||
version:
|
||||
corpora_schema_version: 1.0.0
|
||||
corpora_encoding_version: 0.1.0
|
||||
organism_ontology_term_id: NCBITaxon:9606
|
||||
title: Test dataset
|
||||
contributors:
|
||||
- name: Marcus
|
||||
institution: CZI
|
||||
layer_descriptions:
|
||||
X: raw
|
||||
project_links:
|
||||
- link_url: https://chanzuckerberg.com/
|
||||
link_name: CZI
|
||||
link_type: SUMMARY
|
||||
+34
@@ -0,0 +1,34 @@
|
||||
fixup_gene_symbols:
|
||||
X: log1p
|
||||
obs:
|
||||
cell_type_ontology_term_id:
|
||||
louvain:
|
||||
CD4 T cells: CL:00001
|
||||
B cells: CL:00002
|
||||
CD14+ Monocytes: CL:00003
|
||||
NK cells: CL:00004
|
||||
CD8 T cells: CL:00005
|
||||
FCGR3A+ Monocytes: CL:00006
|
||||
Dendritic cells: CL:00007
|
||||
Megakaryocytes: CL:00008
|
||||
tissue_ontology_term_id: UBERON:12345
|
||||
assay_ontology_term_id: EFO:12345
|
||||
disease_ontology_term_id: MONDO:12345
|
||||
ethnicity_ontology_term_id: HANCESTRO:12345
|
||||
development_stage_ontology_term_id: HsapDv:12345
|
||||
sex: other
|
||||
uns:
|
||||
version:
|
||||
corpora_schema_version: 1.0.0
|
||||
corpora_encoding_version: 0.1.0
|
||||
organism_ontology_term_id: NCBITaxon:9606
|
||||
title: Test dataset
|
||||
contributors:
|
||||
- name: Marcus
|
||||
institution: CZI
|
||||
layer_descriptions:
|
||||
X: raw
|
||||
project_links:
|
||||
- link_url: https://chanzuckerberg.com/
|
||||
link_name: CZI
|
||||
link_type: SUMMARY
|
||||
Reference in New Issue
Block a user