mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-30 03:58:11 +08:00
config refactor (#1854)
* split out config * add tests for base and app config, refactor client config out of app config * refactor default config retrieval * create config test class and helper functions * move default_config into server to fix import issue
This commit is contained in:
@@ -8,8 +8,12 @@ import numpy as np
|
||||
import tiledb
|
||||
from pandas import Series, DataFrame
|
||||
|
||||
from server.common.utils.cxg_generation_utils import (convert_dictionary_to_cxg_group, convert_dataframe_to_cxg_array,
|
||||
convert_ndarray_to_cxg_dense_array, convert_matrix_to_cxg_array)
|
||||
from server.common.utils.cxg_generation_utils import (
|
||||
convert_dictionary_to_cxg_group,
|
||||
convert_dataframe_to_cxg_array,
|
||||
convert_ndarray_to_cxg_dense_array,
|
||||
convert_matrix_to_cxg_array,
|
||||
)
|
||||
|
||||
PROJECT_ROOT = popen("git rev-parse --show-toplevel").read().strip()
|
||||
|
||||
@@ -28,8 +32,9 @@ class TestCxgGenerationUtils(unittest.TestCase):
|
||||
dictionary_name = "favorite_desserts"
|
||||
expected_array_directory = f"{self.testing_cxg_temp_directory}/{dictionary_name}"
|
||||
|
||||
convert_dictionary_to_cxg_group(self.testing_cxg_temp_directory, random_dictionary,
|
||||
group_metadata_name=dictionary_name)
|
||||
convert_dictionary_to_cxg_group(
|
||||
self.testing_cxg_temp_directory, random_dictionary, group_metadata_name=dictionary_name
|
||||
)
|
||||
|
||||
array = tiledb.open(expected_array_directory)
|
||||
actual_stored_metadata = dict(array.meta.items())
|
||||
@@ -44,13 +49,16 @@ class TestCxgGenerationUtils(unittest.TestCase):
|
||||
random_dataframe_name = f"random_dataframe_{uuid4()}"
|
||||
random_dataframe = DataFrame(data={"int_category": random_int_category, "bool_category": random_bool_category})
|
||||
|
||||
convert_dataframe_to_cxg_array(self.testing_cxg_temp_directory, random_dataframe_name, random_dataframe,
|
||||
"int_category", tiledb.Ctx())
|
||||
convert_dataframe_to_cxg_array(
|
||||
self.testing_cxg_temp_directory, random_dataframe_name, random_dataframe, "int_category", tiledb.Ctx()
|
||||
)
|
||||
|
||||
expected_array_directory = f"{self.testing_cxg_temp_directory}/{random_dataframe_name}"
|
||||
expected_array_metadata = {
|
||||
"cxg_schema": json.dumps({"int_category": {"type": "int32"}, "bool_category": {"type": "boolean"},
|
||||
"index": "int_category"})}
|
||||
"cxg_schema": json.dumps(
|
||||
{"int_category": {"type": "int32"}, "bool_category": {"type": "boolean"}, "index": "int_category"}
|
||||
)
|
||||
}
|
||||
|
||||
actual_stored_dataframe_array = tiledb.open(expected_array_directory)
|
||||
actual_stored_dataframe_metadata = dict(actual_stored_dataframe_array.meta.items())
|
||||
@@ -95,7 +103,7 @@ class TestCxgGenerationUtils(unittest.TestCase):
|
||||
|
||||
self.assertTrue(path.isdir(matrix_name))
|
||||
self.assertTrue(isinstance(actual_stored_array, tiledb.SparseArray))
|
||||
self.assertTrue(actual_stored_array[:, :][''].size == 0)
|
||||
self.assertTrue(actual_stored_array[:, :][""].size == 0)
|
||||
|
||||
def test__convert_matrix_to_cxg_array__sparse_array_only_store_nonzeros(self):
|
||||
matrix = np.zeros([3, 3])
|
||||
@@ -110,10 +118,10 @@ class TestCxgGenerationUtils(unittest.TestCase):
|
||||
|
||||
self.assertTrue(path.isdir(matrix_name))
|
||||
self.assertTrue(isinstance(actual_stored_array, tiledb.SparseArray))
|
||||
self.assertTrue(actual_stored_array[0, 0][''] == 1)
|
||||
self.assertTrue(actual_stored_array[1, 1][''] == 1)
|
||||
self.assertTrue(actual_stored_array[2, 2][''] == 2)
|
||||
self.assertTrue(actual_stored_array[:, :][''].size == 3)
|
||||
self.assertTrue(actual_stored_array[0, 0][""] == 1)
|
||||
self.assertTrue(actual_stored_array[1, 1][""] == 1)
|
||||
self.assertTrue(actual_stored_array[2, 2][""] == 2)
|
||||
self.assertTrue(actual_stored_array[:, :][""].size == 3)
|
||||
|
||||
def test__convert_matrix_to_cxg_array__sparse_array_with_column_encoding_empty_array(self):
|
||||
matrix_name = f"{self.testing_cxg_temp_directory}/awesome_column_shift_matrix_{uuid4()}"
|
||||
@@ -122,14 +130,15 @@ class TestCxgGenerationUtils(unittest.TestCase):
|
||||
# a matrix of zeros which is sparse.
|
||||
column_shift = np.ones((3, 2))
|
||||
|
||||
convert_matrix_to_cxg_array(matrix_name, matrix, True, tiledb.Ctx(),
|
||||
column_shift_for_sparse_encoding=column_shift)
|
||||
convert_matrix_to_cxg_array(
|
||||
matrix_name, matrix, True, tiledb.Ctx(), column_shift_for_sparse_encoding=column_shift
|
||||
)
|
||||
|
||||
actual_stored_array = tiledb.open(matrix_name)
|
||||
|
||||
self.assertTrue(path.isdir(matrix_name))
|
||||
self.assertTrue(isinstance(actual_stored_array, tiledb.SparseArray))
|
||||
self.assertTrue(actual_stored_array[:, :][''].size == 0)
|
||||
self.assertTrue(actual_stored_array[:, :][""].size == 0)
|
||||
|
||||
def test__convert_matrix_to_cxg_array__sparse_array_with_column_encoding_partial_array(self):
|
||||
matrix_name = f"{self.testing_cxg_temp_directory}/awesome_column_shift_matrix_{uuid4()}"
|
||||
@@ -137,13 +146,14 @@ class TestCxgGenerationUtils(unittest.TestCase):
|
||||
# Only column shift the first column of ones.
|
||||
column_shift = np.array([[1, 0], [1, 0]])
|
||||
|
||||
convert_matrix_to_cxg_array(matrix_name, matrix, True, tiledb.Ctx(),
|
||||
column_shift_for_sparse_encoding=column_shift)
|
||||
convert_matrix_to_cxg_array(
|
||||
matrix_name, matrix, True, tiledb.Ctx(), column_shift_for_sparse_encoding=column_shift
|
||||
)
|
||||
|
||||
actual_stored_array = tiledb.open(matrix_name)
|
||||
|
||||
self.assertTrue(path.isdir(matrix_name))
|
||||
self.assertTrue(isinstance(actual_stored_array, tiledb.SparseArray))
|
||||
self.assertTrue(actual_stored_array[0, 1][''] == 1)
|
||||
self.assertTrue(actual_stored_array[1, 1][''] == 1)
|
||||
self.assertTrue(actual_stored_array[:, :][''].size == 2)
|
||||
self.assertTrue(actual_stored_array[0, 1][""] == 1)
|
||||
self.assertTrue(actual_stored_array[1, 1][""] == 1)
|
||||
self.assertTrue(actual_stored_array[:, :][""].size == 2)
|
||||
|
||||
@@ -6,7 +6,6 @@ from server.common.utils.matrix_utils import is_matrix_sparse, get_column_shift_
|
||||
|
||||
|
||||
class TestMatrixUtils(unittest.TestCase):
|
||||
|
||||
def test__is_matrix_sparse__zero_and_one_hundred_percent_threshold(self):
|
||||
matrix = np.array([1, 2, 3])
|
||||
|
||||
|
||||
@@ -4,7 +4,6 @@ from server.common.utils.sanitization_utils import sanitize_values_in_list, sani
|
||||
|
||||
|
||||
class TestSanitizationUtils(unittest.TestCase):
|
||||
|
||||
def test__sanitize_values_in_list__not_strings_raises_exception(self):
|
||||
keys_to_sanitize = [1, 2, 3]
|
||||
|
||||
|
||||
@@ -5,12 +5,17 @@ from unittest.mock import patch
|
||||
import numpy as np
|
||||
from pandas import Series, DataFrame
|
||||
|
||||
from server.common.utils.type_conversion_utils import can_cast_to_float32, can_cast_to_int32, get_dtype_of_array, \
|
||||
get_schema_type_hint_of_array, get_dtypes_and_schemas_of_dataframe, convert_pandas_series_to_numpy
|
||||
from server.common.utils.type_conversion_utils import (
|
||||
can_cast_to_float32,
|
||||
can_cast_to_int32,
|
||||
get_dtype_of_array,
|
||||
get_schema_type_hint_of_array,
|
||||
get_dtypes_and_schemas_of_dataframe,
|
||||
convert_pandas_series_to_numpy,
|
||||
)
|
||||
|
||||
|
||||
class TestTypeConversionUtils(unittest.TestCase):
|
||||
|
||||
def test__can_cast_to_float32__string_is_false(self):
|
||||
array_to_convert = Series(data=["1", "2", "3"], dtype=str)
|
||||
|
||||
@@ -97,8 +102,9 @@ class TestTypeConversionUtils(unittest.TestCase):
|
||||
expected_dtypes = [np.float32, np.int32, np.uint8, np.unicode]
|
||||
|
||||
for test_type_index in range(len(types)):
|
||||
with self.subTest(f"Testing get_dtype_of_array with type {types[test_type_index].__name__}",
|
||||
i=test_type_index):
|
||||
with self.subTest(
|
||||
f"Testing get_dtype_of_array with type {types[test_type_index].__name__}", i=test_type_index
|
||||
):
|
||||
array = Series(data=[], dtype=types[test_type_index])
|
||||
self.assertEqual(get_dtype_of_array(array), expected_dtypes[test_type_index])
|
||||
|
||||
@@ -123,8 +129,9 @@ class TestTypeConversionUtils(unittest.TestCase):
|
||||
expected_dtypes = [np.float32, np.int32]
|
||||
|
||||
for test_type_index in range(len(types)):
|
||||
with self.subTest(f"Testing get_dtype_of_array with castable type {types[test_type_index].__name__}",
|
||||
i=test_type_index):
|
||||
with self.subTest(
|
||||
f"Testing get_dtype_of_array with castable type {types[test_type_index].__name__}", i=test_type_index
|
||||
):
|
||||
array = Series(data=[], dtype=types[test_type_index])
|
||||
self.assertEqual(get_dtype_of_array(array), expected_dtypes[test_type_index])
|
||||
|
||||
@@ -141,8 +148,9 @@ class TestTypeConversionUtils(unittest.TestCase):
|
||||
expected_schema_hints = [{"type": "float32"}, {"type": "int32"}, {"type": "boolean"}, {"type": "string"}]
|
||||
|
||||
for test_type_index in range(len(types)):
|
||||
with self.subTest(f"Testing get_schema_type_hint_of_array with type {types[test_type_index].__name__}",
|
||||
i=test_type_index):
|
||||
with self.subTest(
|
||||
f"Testing get_schema_type_hint_of_array with type {types[test_type_index].__name__}", i=test_type_index
|
||||
):
|
||||
array = Series(data=[], dtype=types[test_type_index])
|
||||
self.assertEqual(get_schema_type_hint_of_array(array), expected_schema_hints[test_type_index])
|
||||
|
||||
@@ -160,8 +168,9 @@ class TestTypeConversionUtils(unittest.TestCase):
|
||||
|
||||
for test_type_index in range(len(types)):
|
||||
with self.subTest(
|
||||
f"Testing get_schema_type_hint_of_array with castable type {types[test_type_index].__name__}",
|
||||
i=test_type_index):
|
||||
f"Testing get_schema_type_hint_of_array with castable type {types[test_type_index].__name__}",
|
||||
i=test_type_index,
|
||||
):
|
||||
array = Series(data=[], dtype=types[test_type_index])
|
||||
self.assertEqual(get_schema_type_hint_of_array(array), expected_schema_hints[test_type_index])
|
||||
|
||||
@@ -171,8 +180,10 @@ class TestTypeConversionUtils(unittest.TestCase):
|
||||
dataframe = DataFrame({"float_array": float_array, "category_array": category_array})
|
||||
|
||||
expected_data_types_dict = {"float_array": np.float32, "category_array": np.unicode}
|
||||
expected_schema_type_hints_dict = {"float_array": {"type": "float32"},
|
||||
"category_array": {"type": "categorical", "categories": ["a", "b"]}}
|
||||
expected_schema_type_hints_dict = {
|
||||
"float_array": {"type": "float32"},
|
||||
"category_array": {"type": "categorical", "categories": ["a", "b"]},
|
||||
}
|
||||
|
||||
actual_dataframe_data_types, actual_dataframe_schema_type_hints = get_dtypes_and_schemas_of_dataframe(dataframe)
|
||||
|
||||
@@ -201,5 +212,6 @@ class TestTypeConversionUtils(unittest.TestCase):
|
||||
with self.assertLogs(level="ERROR") as logger:
|
||||
convert_pandas_series_to_numpy(int_series, np.int32)
|
||||
|
||||
self.assertIn("Cannot convert a pandas Series object to an integer dtype if it contains NaNs",
|
||||
logger.output[0])
|
||||
self.assertIn(
|
||||
"Cannot convert a pandas Series object to an integer dtype if it contains NaNs", logger.output[0]
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user