create helper file for controls reducer (#615)

* initial dataframe commit

* initial dataframe port of core app

* rename variables for clarity

* remove unused import

* comment out unused code

* fix array handling bug in crossfilter dimension creation

* allow creation of empty dataframes

* handle non-existent columns

* handle non-existent columns

* revise tests for new dataframe

* comments for clarity

* comments for clarity

* generate bulk add placeholder with real gene names

* fix bug in gene name adding

* more dataframe unit tests

* fix bug - subset from current world, not universe

* put cut and pasted code into a single function

* improve caching of crossfilter

* remove cascading update bug from graph

* more performance work

* improve state handling for scatterplot

* performance optimization of critical path

* add column summarization

* dataframe utils

* add callOnceLazy

* fix tests

* minor updates found during review

* fix misspelling

* remove RESTv02 from function names

* comment cleanup

* cut/icut col parameter defaults to null

* break up large test

* improve tests and comments on dataframe at/has functions

* add Dataframe withCol/dropCol

* expression varData now stored in a dataframe

* dead code cleanup

* use dataframe.summarize()

* test cases for Dataframe.col.summarize

* update test cases for new dataframe summarize

* improve naming

* use new hasCol API

* add comments

* add more Dataframe.withCol tests

* add ability to specify row index in cut operation

* retire subsetVarData function

* correctly handle expression subsetting

* lint and improve comments

* rename cut to subset

* create helper file for controls reducer
This commit is contained in:
Bruce Martin
2019-02-28 09:21:19 -08:00
committed by GitHub
parent e7ad6f5d1c
commit a876740a3c
4 changed files with 192 additions and 159 deletions
+27 -158
View File
@@ -2,7 +2,7 @@
import _ from "lodash";
import { World, WorldUtil } from "../util/stateManager";
import { World, WorldUtil, ControlsHelper } from "../util/stateManager";
import parseRGB from "../util/parseRGB";
import Crossfilter from "../util/typedCrossfilter";
import * as globals from "../globals";
@@ -13,153 +13,6 @@ import {
diffexpDimensionName,
makeContinuousDimensionName
} from "../util/nameCreators";
import { fillRange } from "../util/typedCrossfilter/util";
/*
Selection state for categoricals are tracked in an Object that
has two main components for each category:
1. mapping of option value to an index
2. array of bool selection state by index
Remember that option values can be ANY js type, except undefined/null.
{
_category_name_1: {
// map of option value to index
categoryIndices: Map([
catval1: index,
...
])
// index->selection true/false state
categorySelected: [ true/false, true/false, ... ]
// number of options
numCategories: number,
// isTruncated - true if the options for selection has
// been truncated (ie, was too large to implement)
}
}
*/
function topNCategories(summary) {
const counts = _.map(summary.categories, cat =>
summary.categoryCounts.get(cat)
);
const sortIndex = fillRange(new Array(summary.numCategories)).sort(
(a, b) => counts[b] - counts[a]
);
const sortedCategories = _.map(sortIndex, i => summary.categories[i]);
const sortedCounts = _.map(sortIndex, i => counts[i]);
const N = globals.maxCategoricalOptionsToDisplay;
if (sortedCategories.length < N) {
return [sortedCategories, sortedCounts];
}
return [sortedCategories.slice(0, N), sortedCounts.slice(0, N)];
}
function createCategoricalSelectionState(state, world) {
const res = {};
_.forEach(world.obsAnnotations.colIndex.keys(), key => {
const summary = world.obsAnnotations.col(key).summarize();
if (summary.categories) {
const isColorField = key.includes("color") || key.includes("Color");
const isSelectableCategory =
!isColorField &&
key !== "name" &&
summary.categories.length < state.maxCategoryItems;
if (isSelectableCategory) {
const [categoryValues, categoryCounts] = topNCategories(summary);
const categoryIndices = new Map(categoryValues.map((v, i) => [v, i]));
const numCategories = categoryIndices.size;
const categorySelected = new Array(numCategories).fill(true);
const isTruncated = categoryValues.length < summary.numCategories;
res[key] = {
categoryValues, // array: of natively typed category values
categoryIndices, // map: category value (native type) -> category index
categorySelected, // array: t/f selection state
numCategories, // number: of categories
isTruncated, // bool: true if list was truncated
categoryCounts // array: cardinality of each category
};
}
}
});
return res;
}
/*
given a categoricalSelectionState, return the list of all category values
where selection state is true (ie, they are selected).
*/
function selectedValuesForCategory(categorySelectionState) {
const selectedValues = _([...categorySelectionState.categoryIndices])
.filter(tuple => categorySelectionState.categorySelected[tuple[1]])
.map(tuple => tuple[0])
.value();
return selectedValues;
}
/*
build a crossfilter dimension map for all gene expression related dimensions.
*/
function createGenesDimMap(userDefinedGenes, diffexpGenes, world, crossfilter) {
function _createGenesDimMap(genes, nameCreator) {
return genes.reduce((acc, gene) => {
acc[nameCreator(gene)] = World.createVarDataDimension(
world,
crossfilter,
gene
);
return acc;
}, {});
}
return {
..._createGenesDimMap(userDefinedGenes, userDefinedDimensionName),
..._createGenesDimMap(diffexpGenes, diffexpDimensionName)
};
}
function pruneVarDataCache(varData, needed) {
/*
Remove any unneeded columns from the varData dataframe. Will only
prune / remove if the total column count exceeds VarDataCacheLowWatermark
Note: this code leverages the fact that dataframe offsets indicate
the order in which the columns were added. This crudely provides
LRU semantics, so we can delete "older" columns first.
*/
/*
VarDataCacheLowWatermark - this cofig value sets the minimum cache size,
in columns, below which we don't throw away data.
The value should be high enough so we are caching the maximum which will
"typically" be used in the UI (currently: 10 for diffexp, and N for user-
specified genes), and low enough to account for memory use (any single
column size is 4 bytes * numObs, so a column can be multi-megabyte in common
use cases).
*/
const VarDataCacheLowWatermark = 32;
const numOverWatermark = varData.dims[1] - VarDataCacheLowWatermark;
if (numOverWatermark <= 0) return varData;
const { colIndex } = varData;
const all = colIndex.keys();
const unused = _.difference(all, needed);
if (unused.length > 0) {
// sort by offset in the dataframe - ie, psuedo-LRU
unused.sort((a, b) => colIndex.getOffset(a) - colIndex.getOffset(b));
const numToDrop =
unused.length < numOverWatermark ? unused.length : numOverWatermark;
for (let i = 0; i < numToDrop; i += 1) {
varData = varData.dropCol(unused[i]);
}
}
return varData;
}
const Controls = (
state = {
@@ -233,7 +86,7 @@ const Controls = (
const colorRGB = new Array(universe.nObs).fill(
parseRGB(globals.defaultCellColor)
);
const categoricalSelectionState = createCategoricalSelectionState(
const categoricalSelectionState = ControlsHelper.createCategoricalSelectionState(
state,
world
);
@@ -269,7 +122,7 @@ const Controls = (
const colorRGB = new Array(universe.nObs).fill(
parseRGB(globals.defaultCellColor)
);
const categoricalSelectionState = createCategoricalSelectionState(
const categoricalSelectionState = ControlsHelper.createCategoricalSelectionState(
state,
world
);
@@ -282,7 +135,12 @@ const Controls = (
});
const dimensionMap = {
...fullUniverseCache.dimensionMap,
...createGenesDimMap(userDefinedGenes, diffexpGenes, world, crossfilter)
...ControlsHelper.createGenesDimMap(
userDefinedGenes,
diffexpGenes,
world,
crossfilter
)
};
WorldUtil.clearCaches();
@@ -309,14 +167,19 @@ const Controls = (
const colorRGB = new Array(world.nObs).fill(
parseRGB(globals.defaultCellColor)
);
const categoricalSelectionState = createCategoricalSelectionState(
const categoricalSelectionState = ControlsHelper.createCategoricalSelectionState(
state,
world
);
const crossfilter = Crossfilter(world.obsAnnotations);
const dimensionMap = {
...World.createObsDimensionMap(crossfilter, world),
...createGenesDimMap(userDefinedGenes, diffexpGenes, world, crossfilter)
...ControlsHelper.createGenesDimMap(
userDefinedGenes,
diffexpGenes,
world,
crossfilter
)
};
WorldUtil.clearCaches();
@@ -376,8 +239,14 @@ const Controls = (
Object.keys(action.expressionData)
)
);
universeVarData = pruneVarDataCache(universeVarData, allTheGenesWeNeed);
worldVarData = pruneVarDataCache(worldVarData, allTheGenesWeNeed);
universeVarData = ControlsHelper.pruneVarDataCache(
universeVarData,
allTheGenesWeNeed
);
worldVarData = ControlsHelper.pruneVarDataCache(
worldVarData,
allTheGenesWeNeed
);
return {
...state,
@@ -442,7 +311,7 @@ const Controls = (
};
}
case "clear differential expression": {
const { world, universe, dimensionMap } = state;
const { world, dimensionMap } = state;
const _dimensionMap = dimensionMap;
_.forEach(action.diffExp, values => {
const name = world.varAnnotations.at(values[0], "name");
@@ -607,7 +476,7 @@ const Controls = (
// update the filter to match all selected options
const cat = newCategoricalSelectionState[action.metadataField];
state.dimensionMap[obsAnnoDimensionName(action.metadataField)].filterEnum(
selectedValuesForCategory(cat)
ControlsHelper.selectedValuesForCategory(cat)
);
return {
@@ -631,7 +500,7 @@ const Controls = (
// update the filter to match all selected options
const cat = newCategoricalSelectionState[action.metadataField];
state.dimensionMap[obsAnnoDimensionName(action.metadataField)].filterEnum(
selectedValuesForCategory(cat)
ControlsHelper.selectedValuesForCategory(cat)
);
return {
@@ -0,0 +1,164 @@
/*
Helper functions for the controls reducer
*/
import _ from "lodash";
import * as globals from "../../globals";
import { fillRange } from "../typedCrossfilter/util";
import {
userDefinedDimensionName,
diffexpDimensionName
} from "../nameCreators";
import * as World from "./world";
/*
Selection state for categoricals are tracked in an Object that
has two main components for each category:
1. mapping of option value to an index
2. array of bool selection state by index
Remember that option values can be ANY js type, except undefined/null.
{
_category_name_1: {
// map of option value to index
categoryIndices: Map([
catval1: index,
...
])
// index->selection true/false state
categorySelected: [ true/false, true/false, ... ]
// number of options
numCategories: number,
// isTruncated - true if the options for selection has
// been truncated (ie, was too large to implement)
}
}
*/
function topNCategories(summary) {
const counts = _.map(summary.categories, cat =>
summary.categoryCounts.get(cat)
);
const sortIndex = fillRange(new Array(summary.numCategories)).sort(
(a, b) => counts[b] - counts[a]
);
const sortedCategories = _.map(sortIndex, i => summary.categories[i]);
const sortedCounts = _.map(sortIndex, i => counts[i]);
const N = globals.maxCategoricalOptionsToDisplay;
if (sortedCategories.length < N) {
return [sortedCategories, sortedCounts];
}
return [sortedCategories.slice(0, N), sortedCounts.slice(0, N)];
}
export function createCategoricalSelectionState(state, world) {
const res = {};
_.forEach(world.obsAnnotations.colIndex.keys(), key => {
const summary = world.obsAnnotations.col(key).summarize();
if (summary.categories) {
const isColorField = key.includes("color") || key.includes("Color");
const isSelectableCategory =
!isColorField &&
key !== "name" &&
summary.categories.length < state.maxCategoryItems;
if (isSelectableCategory) {
const [categoryValues, categoryCounts] = topNCategories(summary);
const categoryIndices = new Map(categoryValues.map((v, i) => [v, i]));
const numCategories = categoryIndices.size;
const categorySelected = new Array(numCategories).fill(true);
const isTruncated = categoryValues.length < summary.numCategories;
res[key] = {
categoryValues, // array: of natively typed category values
categoryIndices, // map: category value (native type) -> category index
categorySelected, // array: t/f selection state
numCategories, // number: of categories
isTruncated, // bool: true if list was truncated
categoryCounts // array: cardinality of each category
};
}
}
});
return res;
}
/*
given a categoricalSelectionState, return the list of all category values
where selection state is true (ie, they are selected).
*/
export function selectedValuesForCategory(categorySelectionState) {
const selectedValues = _([...categorySelectionState.categoryIndices])
.filter(tuple => categorySelectionState.categorySelected[tuple[1]])
.map(tuple => tuple[0])
.value();
return selectedValues;
}
/*
build a crossfilter dimension map for all gene expression related dimensions.
*/
export function createGenesDimMap(
userDefinedGenes,
diffexpGenes,
world,
crossfilter
) {
function _createGenesDimMap(genes, nameCreator) {
return genes.reduce((acc, gene) => {
acc[nameCreator(gene)] = World.createVarDataDimension(
world,
crossfilter,
gene
);
return acc;
}, {});
}
return {
..._createGenesDimMap(userDefinedGenes, userDefinedDimensionName),
..._createGenesDimMap(diffexpGenes, diffexpDimensionName)
};
}
export function pruneVarDataCache(varData, needed) {
/*
Remove any unneeded columns from the varData dataframe. Will only
prune / remove if the total column count exceeds VarDataCacheLowWatermark
Note: this code leverages the fact that dataframe offsets indicate
the order in which the columns were added. This crudely provides
LRU semantics, so we can delete "older" columns first.
*/
/*
VarDataCacheLowWatermark - this cofig value sets the minimum cache size,
in columns, below which we don't throw away data.
The value should be high enough so we are caching the maximum which will
"typically" be used in the UI (currently: 10 for diffexp, and N for user-
specified genes), and low enough to account for memory use (any single
column size is 4 bytes * numObs, so a column can be multi-megabyte in common
use cases).
*/
const VarDataCacheLowWatermark = 32;
const numOverWatermark = varData.dims[1] - VarDataCacheLowWatermark;
if (numOverWatermark <= 0) return varData;
const { colIndex } = varData;
const all = colIndex.keys();
const unused = _.difference(all, needed);
if (unused.length > 0) {
// sort by offset in the dataframe - ie, psuedo-LRU
unused.sort((a, b) => colIndex.getOffset(a) - colIndex.getOffset(b));
const numToDrop =
unused.length < numOverWatermark ? unused.length : numOverWatermark;
for (let i = 0; i < numToDrop; i += 1) {
varData = varData.dropCol(unused[i]);
}
}
return varData;
}
+1
View File
@@ -17,3 +17,4 @@ exists to support those concepts.
export * as Universe from "./universe";
export * as World from "./world";
export * as WorldUtil from "./worldUtil";
export * as ControlsHelper from "./controlsHelpers";
-1
View File
@@ -3,7 +3,6 @@
import _ from "lodash";
import { layoutDimensionName, obsAnnoDimensionName } from "../nameCreators";
import Crossfilter from "../typedCrossfilter";
import { sliceByIndex } from "../typedCrossfilter/util";
import * as Dataframe from "../dataframe";
/*