mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-02 21:08:12 +08:00
sparse column shift encoding. (#1502)
Many of our matrices are log normalized, which tends to eliminate the number of non zero values (if there were any). This prevents the matrix from being stored as a sparse matrix. The solution here is to use a simple transformation to make it sparse again. The most common value from each column is subtracted from that column. These values that were subtracted are saved in an array called X_col_shift. The cellxgene code needs to understand how to undo the transformation when operating over the X matrix. - added script to create a synthetic dataset for testing - added a script to convert an existing CXG dataset to a sparse CXG dataset
This commit is contained in:
@@ -133,6 +133,10 @@ class CxgAdaptor(DataAdaptor):
|
||||
return False
|
||||
return True
|
||||
|
||||
def has_array(self, name):
|
||||
a_type = tiledb.object_type(path_join(self.url, name), ctx=self.tiledb_ctx)
|
||||
return a_type == "array"
|
||||
|
||||
def _validate_and_initialize(self):
|
||||
"""
|
||||
remember, preload_validation() has already been called, so
|
||||
@@ -147,13 +151,7 @@ class CxgAdaptor(DataAdaptor):
|
||||
* version 0.1 -- metadata attache to cxg_group_metadata array.
|
||||
Same as 0, except it adds group metadata.
|
||||
"""
|
||||
a_type = tiledb.object_type(path_join(self.url, "cxg_group_metadata"), ctx=self.tiledb_ctx)
|
||||
if a_type is None:
|
||||
# version 0
|
||||
cxg_version = "0.0"
|
||||
title = None
|
||||
about = None
|
||||
elif a_type == "array":
|
||||
if self.has_array("cxg_group_metadata"):
|
||||
# version >0
|
||||
gmd = self.open_array("cxg_group_metadata")
|
||||
cxg_version = gmd.meta["cxg_version"]
|
||||
@@ -161,6 +159,11 @@ class CxgAdaptor(DataAdaptor):
|
||||
cxg_properties = json.loads(gmd.meta["cxg_properties"])
|
||||
title = cxg_properties.get("title", None)
|
||||
about = cxg_properties.get("about", None)
|
||||
else:
|
||||
# version 0
|
||||
cxg_version = "0.0"
|
||||
title = None
|
||||
about = None
|
||||
|
||||
if cxg_version not in ["0.0", "0.1"]:
|
||||
raise DatasetAccessError(f"cxg matrix is not valid: {self.url}")
|
||||
@@ -251,10 +254,18 @@ class CxgAdaptor(DataAdaptor):
|
||||
data = X[:, :]
|
||||
else:
|
||||
data = X.multi_index[obs_items, var_items]
|
||||
nrows, obsindices = self.__remap_indices(X.shape[0], obs_mask, data["obs"])
|
||||
ncols, varindices = self.__remap_indices(X.shape[1], var_mask, data["var"])
|
||||
|
||||
nrows, obsindices = self.__remap_indices(X.shape[0], obs_mask, data.get("coords", data)["obs"])
|
||||
ncols, varindices = self.__remap_indices(X.shape[1], var_mask, data.get("coords", data)["var"])
|
||||
densedata = np.zeros((nrows, ncols), dtype=self.get_X_array_dtype())
|
||||
densedata[obsindices, varindices] = data[""]
|
||||
if self.has_array("X_col_shift"):
|
||||
X_col_shift = self.open_array("X_col_shift")
|
||||
if var_items == slice(None):
|
||||
densedata += X_col_shift[:]
|
||||
else:
|
||||
densedata += X_col_shift.multi_index[var_items][""]
|
||||
|
||||
return densedata
|
||||
|
||||
else:
|
||||
|
||||
Reference in New Issue
Block a user