mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-28 18:28:12 +08:00
Dataframe, part deux - add varData and summarize() (#608)
* initial dataframe commit * initial dataframe port of core app * rename variables for clarity * remove unused import * comment out unused code * fix array handling bug in crossfilter dimension creation * allow creation of empty dataframes * handle non-existent columns * handle non-existent columns * revise tests for new dataframe * comments for clarity * comments for clarity * generate bulk add placeholder with real gene names * fix bug in gene name adding * more dataframe unit tests * fix bug - subset from current world, not universe * put cut and pasted code into a single function * improve caching of crossfilter * remove cascading update bug from graph * more performance work * improve state handling for scatterplot * performance optimization of critical path * add column summarization * dataframe utils * add callOnceLazy * fix tests * minor updates found during review * fix misspelling * remove RESTv02 from function names * comment cleanup * cut/icut col parameter defaults to null * break up large test * improve tests and comments on dataframe at/has functions * add Dataframe withCol/dropCol * expression varData now stored in a dataframe * dead code cleanup * use dataframe.summarize() * test cases for Dataframe.col.summarize * update test cases for new dataframe summarize * improve naming * use new hasCol API * add comments * add more Dataframe.withCol tests * add ability to specify row index in cut operation * retire subsetVarData function * correctly handle expression subsetting * lint and improve comments * rename cut to subset * changes based on PR review
This commit is contained in:
@@ -16,5 +16,4 @@ exists to support those concepts.
|
||||
|
||||
export * as Universe from "./universe";
|
||||
export * as World from "./world";
|
||||
export * as kvCache from "./keyvalcache";
|
||||
export * as WorldUtil from "./worldUtil";
|
||||
|
||||
@@ -1,122 +0,0 @@
|
||||
// jshint esversion: 6
|
||||
import _ from "lodash";
|
||||
|
||||
/*
|
||||
Very simple key/value cache for use by World & Universe. Cache keys must
|
||||
be a string, and values are any JS non-primitive value.
|
||||
|
||||
* constructor(lowWatermark, minTTL):
|
||||
- lowWatermark defines the number of cache elements below which
|
||||
flushing will not occur.
|
||||
- minTTL defines minimum time in milliseconds that cache entries will live.
|
||||
A value of -1 disables automatic flushing (flush() can still
|
||||
be called by external user).
|
||||
* set() - add a key/val pair.
|
||||
* get() - get a value or undefined if not present.
|
||||
* flush(minAgeMs) - flush cache entries in excess of lowWatermark if those
|
||||
entries are older than minAgeMs.
|
||||
|
||||
*/
|
||||
|
||||
const cachePrivateKey = "__kvcachekey__";
|
||||
const defaultLowWatermark = 32;
|
||||
const defaultMinTTL = 1000;
|
||||
|
||||
function create(lowWatermark = defaultLowWatermark, minTTL = defaultMinTTL) {
|
||||
if (typeof minTTL !== "number" || typeof lowWatermark !== "number") {
|
||||
throw new TypeError(
|
||||
"minTTL and lowWatermark parameters must be a primitive number"
|
||||
);
|
||||
}
|
||||
if (lowWatermark < 0 || minTTL < 0) {
|
||||
throw new RangeError(
|
||||
"minTTL and lowWatermark parameters must be number greater than zero"
|
||||
);
|
||||
}
|
||||
|
||||
return {
|
||||
[cachePrivateKey]: {
|
||||
lowWatermark,
|
||||
minTTL
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
function get(kvcache, key) {
|
||||
if (key === cachePrivateKey) {
|
||||
throw new RangeError(`key parameter may not have value ${cachePrivateKey}`);
|
||||
}
|
||||
|
||||
const val = kvcache[key];
|
||||
if (val) {
|
||||
val[cachePrivateKey] = Date.now();
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
function set(kvcache, key, val) {
|
||||
if (key === cachePrivateKey) {
|
||||
throw new RangeError(`key parameter may not have value ${cachePrivateKey}`);
|
||||
}
|
||||
|
||||
const newKvCache = { ...kvcache };
|
||||
newKvCache[key] = val;
|
||||
val[cachePrivateKey] = Date.now();
|
||||
flushInPlace(newKvCache);
|
||||
return newKvCache;
|
||||
}
|
||||
|
||||
function flush(kvcache) {
|
||||
const newKvCache = { ...kvcache };
|
||||
flushInPlace(newKvCache);
|
||||
return newKvCache;
|
||||
}
|
||||
|
||||
/*
|
||||
Flush elements from cache IF cache size is greater than lowWatermark, and
|
||||
those elements are older than minAgeMS
|
||||
*/
|
||||
function flushInPlace(kvCache) {
|
||||
const { lowWatermark, minTTL } = kvCache[cachePrivateKey];
|
||||
const eol = Date.now() - minTTL;
|
||||
const allKeys = _(kvCache)
|
||||
.keys()
|
||||
.filter(k => k !== cachePrivateKey)
|
||||
.sortBy([k => kvCache[k][cachePrivateKey]])
|
||||
.value();
|
||||
|
||||
if (allKeys.length > lowWatermark) {
|
||||
const keysToDelete = _(allKeys)
|
||||
.slice(0, allKeys.length - lowWatermark)
|
||||
.filter(k => kvCache[k][cachePrivateKey] <= eol)
|
||||
.value();
|
||||
_.forEach(keysToDelete, k => delete kvCache[k]);
|
||||
}
|
||||
|
||||
return kvCache;
|
||||
}
|
||||
|
||||
/*
|
||||
use to create a cache that is a transformation of another cache.
|
||||
*/
|
||||
function map(srcKvCache, cb, createOptions) {
|
||||
const keysInSrcKvCache = _(srcKvCache)
|
||||
.keys()
|
||||
.filter(k => k !== cachePrivateKey)
|
||||
.value();
|
||||
const lowWatermark = _.get(
|
||||
createOptions,
|
||||
"lowWatermark",
|
||||
defaultLowWatermark
|
||||
);
|
||||
const minTTL = _.get(createOptions, "minTTL", defaultMinTTL);
|
||||
const newKvCache = create(lowWatermark, minTTL);
|
||||
_.forEach(keysInSrcKvCache, key => {
|
||||
const val = cb(get(srcKvCache, key), key);
|
||||
newKvCache[key] = val;
|
||||
val[cachePrivateKey] = Date.now();
|
||||
});
|
||||
return newKvCache;
|
||||
}
|
||||
|
||||
export { create, get, set, flush, map };
|
||||
@@ -1,128 +0,0 @@
|
||||
import _ from "lodash";
|
||||
import finiteExtent from "../finiteExtent";
|
||||
|
||||
/*
|
||||
Build and return obs/var summary using any annotation in the schema
|
||||
|
||||
Summary information for each annotation, keyed by annotation name.
|
||||
Value will be an object, containing summary information.
|
||||
|
||||
For continuous annotations (int, float, etc):
|
||||
<annotation_name>: {
|
||||
categorical: false,
|
||||
range {
|
||||
min: <number>,
|
||||
max: <number>
|
||||
}
|
||||
}
|
||||
|
||||
For categorical annotations (boolean, string, category):
|
||||
<annotation_name>: {
|
||||
categorical: true,
|
||||
categories: [ <category1>, <category2>, ... ]
|
||||
categoryCounts: Map {
|
||||
<category1>: <number>,
|
||||
...
|
||||
},
|
||||
numCategories: <number>
|
||||
}
|
||||
|
||||
Summarize will be returned for BOTH obs and var annotations.
|
||||
|
||||
Example:
|
||||
{
|
||||
"Splice_sites_Annotated": {
|
||||
categorical: false,
|
||||
range: {
|
||||
"min": 26,
|
||||
"max": 1075869
|
||||
}
|
||||
},
|
||||
"Selection": {
|
||||
categorical: true,
|
||||
numCategories, 3,
|
||||
categories: [ "Astrocytes(HEPACAM)", "Endothelial(BSC)", "Unpanned" ],
|
||||
categoryCounts: Map {
|
||||
"Astrocytes(HEPACAM)": 714,
|
||||
"Endothelial(BSC)": 123,
|
||||
"Unpanned": 665
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
NOTE: will not summarize the required 'name' annotation, as that is
|
||||
specified as unique per element.
|
||||
*/
|
||||
function _summarizeAnnotations(_schema, df) {
|
||||
const summary = _(_schema) // lodash wrapping: https://lodash.com/docs/4.17.11#lodash
|
||||
.filter(v => v.name !== "name") // don't summarize name
|
||||
.keyBy("name")
|
||||
.mapValues(anno => {
|
||||
const { name, type } = anno;
|
||||
const continuous = type === "int32" || type === "float32";
|
||||
const numRows = df.length;
|
||||
const col = df.col(name) ? df.col(name).asArray() : null;
|
||||
|
||||
if (continuous) {
|
||||
let min;
|
||||
let max;
|
||||
let nan = 0;
|
||||
let pinf = 0;
|
||||
let ninf = 0;
|
||||
if (col) {
|
||||
for (let r = 0; r < numRows; r += 1) {
|
||||
const val = Number(col[r]);
|
||||
if (Number.isFinite(val)) {
|
||||
if (min === undefined) {
|
||||
min = val;
|
||||
max = val;
|
||||
} else {
|
||||
min = val < min ? val : min;
|
||||
max = val > max ? val : max;
|
||||
}
|
||||
} else if (Number.isNaN(val)) {
|
||||
nan += 1;
|
||||
} else if (val > 0) {
|
||||
pinf += 1;
|
||||
} else {
|
||||
ninf += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
return {
|
||||
categorical: false,
|
||||
range: { min, max, nan, pinf, ninf }
|
||||
};
|
||||
}
|
||||
|
||||
/* else categorical */
|
||||
const categoryCounts = new Map();
|
||||
if (col) {
|
||||
for (let r = 0; r < numRows; r += 1) {
|
||||
const val = col[r];
|
||||
let curCount = categoryCounts.get(val);
|
||||
if (curCount === undefined) curCount = 0;
|
||||
categoryCounts.set(val, curCount + 1);
|
||||
}
|
||||
}
|
||||
return {
|
||||
categorical: true,
|
||||
categories: [...categoryCounts.keys()],
|
||||
categoryCounts,
|
||||
numCategories: categoryCounts.size
|
||||
};
|
||||
})
|
||||
.value();
|
||||
return summary;
|
||||
}
|
||||
|
||||
export default function summarizeAnnotations(
|
||||
schema,
|
||||
obsAnnotations,
|
||||
varAnnotations
|
||||
) {
|
||||
return {
|
||||
obs: _summarizeAnnotations(schema.annotations.obs, obsAnnotations),
|
||||
var: _summarizeAnnotations(schema.annotations.var, varAnnotations)
|
||||
};
|
||||
}
|
||||
@@ -2,8 +2,6 @@
|
||||
|
||||
import _ from "lodash";
|
||||
|
||||
import * as kvCache from "./keyvalcache";
|
||||
import summarizeAnnotations from "./summarizeAnnotations";
|
||||
import decodeMatrixFBS from "./matrix";
|
||||
import * as Dataframe from "../dataframe";
|
||||
|
||||
@@ -12,15 +10,7 @@ Private helper function - create and return a template Universe
|
||||
*/
|
||||
function templateUniverse() {
|
||||
/* default universe template */
|
||||
|
||||
/* varDataCache config - see kvCache for semantics */
|
||||
const VarDataCacheLowWatermark = 32; // cache element count
|
||||
const VarDataCacheTTLMs = 1000; // min cache time in MS
|
||||
|
||||
return {
|
||||
api: null,
|
||||
finalized: false, // XXX: may not be needed
|
||||
|
||||
nObs: 0,
|
||||
nVar: 0,
|
||||
schema: {},
|
||||
@@ -28,18 +18,14 @@ function templateUniverse() {
|
||||
/*
|
||||
Annotations
|
||||
*/
|
||||
obsAnnotations: null,
|
||||
varAnnotations: null,
|
||||
obsLayout: null,
|
||||
summary: null /* derived data summaries. XXX: consider exploding in place */,
|
||||
obsAnnotations: Dataframe.Dataframe.empty(),
|
||||
varAnnotations: Dataframe.Dataframe.empty(),
|
||||
obsLayout: Dataframe.Dataframe.empty(),
|
||||
|
||||
/*
|
||||
Cache of var data (expression), by var annotation name. Data can be
|
||||
accesses as a POJO, but if you want caching semantics, use the kvCache
|
||||
API (eg., kvCache.get(), kvCache.set(), ...), which will maintain the
|
||||
LRU semantics.
|
||||
Var data columns - subset of all
|
||||
*/
|
||||
varDataCache: kvCache.create(VarDataCacheLowWatermark, VarDataCacheTTLMs)
|
||||
varData: Dataframe.Dataframe.empty(null, new Dataframe.KeyIndex())
|
||||
};
|
||||
}
|
||||
|
||||
@@ -51,29 +37,6 @@ These functions are used exclusively by the actions and reducers to
|
||||
build an internal POJO for use by the rendering components.
|
||||
*/
|
||||
|
||||
/*
|
||||
generate any client-side transformations or summarization that
|
||||
is independent of REST API response formats.
|
||||
*/
|
||||
function finalize(universe) {
|
||||
/* A bit of sanity checking! */
|
||||
const { nObs, nVar } = universe;
|
||||
if (
|
||||
nObs !== universe.obsLayout.length ||
|
||||
nObs !== universe.obsAnnotations.length ||
|
||||
nVar !== universe.varAnnotations.length
|
||||
) {
|
||||
throw new Error("Universe dimensionality mismatch - failed to load");
|
||||
}
|
||||
// TODO: add more sanity checks, such as:
|
||||
// - all annotations in the schema
|
||||
// - layout has supported number of dimensions
|
||||
// - ...
|
||||
|
||||
universe.finalized = true;
|
||||
return universe;
|
||||
}
|
||||
|
||||
function AnnotationsFBSToDataframe(arrayBuffer) {
|
||||
/*
|
||||
Convert a Matrix FBS to a Dataframe.
|
||||
@@ -118,7 +81,7 @@ function reconcileSchemaCategoriesWithSummary(universe) {
|
||||
) {
|
||||
const categories = _.union(
|
||||
_.get(s, "categories", []),
|
||||
_.get(universe.summary.obs[s.name], "categories", [])
|
||||
_.get(universe.obsAnnotations.col(s.name).summarize(), "categories", [])
|
||||
);
|
||||
s.categories = categories;
|
||||
}
|
||||
@@ -138,9 +101,6 @@ export function createUniverseFromResponse(
|
||||
const { schema } = schemaResponse;
|
||||
const universe = templateUniverse();
|
||||
|
||||
/* constants */
|
||||
universe.api = "0.2";
|
||||
|
||||
/* schema related */
|
||||
universe.schema = schema;
|
||||
universe.nObs = schema.dataframe.nObs;
|
||||
@@ -152,14 +112,17 @@ export function createUniverseFromResponse(
|
||||
/* layout */
|
||||
universe.obsLayout = LayoutFBSToDataframe(layoutFBSResponse);
|
||||
|
||||
universe.summary = summarizeAnnotations(
|
||||
universe.schema,
|
||||
universe.obsAnnotations,
|
||||
universe.varAnnotations
|
||||
);
|
||||
/* sanity check */
|
||||
if (
|
||||
universe.nObs !== universe.obsLayout.length ||
|
||||
universe.nObs !== universe.obsAnnotations.length ||
|
||||
universe.nVar !== universe.varAnnotations.length
|
||||
) {
|
||||
throw new Error("Universe dimensionality mismatch - failed to load");
|
||||
}
|
||||
|
||||
reconcileSchemaCategoriesWithSummary(universe);
|
||||
return finalize(universe);
|
||||
return universe;
|
||||
}
|
||||
|
||||
export function convertDataFBStoObject(universe, arrayBuffer) {
|
||||
|
||||
@@ -1,11 +1,10 @@
|
||||
// jshint esversion: 6
|
||||
|
||||
import _ from "lodash";
|
||||
import * as kvCache from "./keyvalcache";
|
||||
import summarizeAnnotations from "./summarizeAnnotations";
|
||||
import { layoutDimensionName, obsAnnoDimensionName } from "../nameCreators";
|
||||
import Crossfilter from "../typedCrossfilter";
|
||||
import { sliceByIndex } from "../typedCrossfilter/util";
|
||||
import * as Dataframe from "../dataframe";
|
||||
|
||||
/*
|
||||
|
||||
@@ -38,48 +37,33 @@ Notable keys in the world object:
|
||||
A dataframe containing the X/Y layout for all obs. Columns are named
|
||||
'X' and 'Y', and rows are indexed in the same way as obsAnnotation.
|
||||
|
||||
* summary: summary of each obsAnnotation column (eg, numeric extent for
|
||||
continuous data, category counts for categorical metadata)
|
||||
|
||||
* varDataCache: expression columns, in a kvCache. TODO: maybe move to a
|
||||
Dataframe in the future.
|
||||
* varData: a cache of expression columns, stored in a Dataframe. Cache
|
||||
managed by controls reducer.
|
||||
|
||||
*/
|
||||
|
||||
/* varDataCache config - see kvCache for semantics */
|
||||
const VarDataCacheLowWatermark = 32; // cache element count
|
||||
const VarDataCacheTTLMs = 1000; // min cache time in MS
|
||||
|
||||
function templateWorld() {
|
||||
return {
|
||||
/* schema/version related */
|
||||
api: null,
|
||||
schema: null,
|
||||
nObs: 0,
|
||||
nVar: 0,
|
||||
|
||||
/* annotations */
|
||||
obsAnnotations: null,
|
||||
varAnnotations: null,
|
||||
obsAnnotations: Dataframe.Dataframe.empty(),
|
||||
varAnnotations: Dataframe.Dataframe.empty(),
|
||||
|
||||
/* layout of graph. Dataframe. */
|
||||
obsLayout: null,
|
||||
obsLayout: Dataframe.Dataframe.empty(),
|
||||
|
||||
/* derived data summaries XXX: consider exploding in place */
|
||||
summary: null,
|
||||
|
||||
varDataCache: kvCache.create(
|
||||
VarDataCacheLowWatermark,
|
||||
VarDataCacheTTLMs
|
||||
) /* cache of var data (expression) */
|
||||
/*
|
||||
Var data columns - subset of all data (may be empty)
|
||||
*/
|
||||
varData: Dataframe.Dataframe.empty(null, new Dataframe.KeyIndex())
|
||||
};
|
||||
}
|
||||
|
||||
export function createWorldFromEntireUniverse(universe) {
|
||||
if (!universe.finalized) {
|
||||
throw new Error("World can't be created from an partial Universe");
|
||||
}
|
||||
|
||||
const world = templateWorld();
|
||||
|
||||
/*
|
||||
@@ -87,31 +71,21 @@ export function createWorldFromEntireUniverse(universe) {
|
||||
*/
|
||||
|
||||
/* Schema related */
|
||||
world.api = universe.api;
|
||||
world.schema = universe.schema;
|
||||
world.nObs = universe.nObs;
|
||||
world.nVar = universe.nVar;
|
||||
|
||||
/* annotations */
|
||||
/* annotation dataframes */
|
||||
world.obsAnnotations = universe.obsAnnotations;
|
||||
world.varAnnotations = universe.varAnnotations;
|
||||
|
||||
/* layout and display characteristics */
|
||||
/* layout and display characteristics dataframe */
|
||||
world.obsLayout = universe.obsLayout;
|
||||
|
||||
/* derived data & summaries */
|
||||
world.summary = summarizeAnnotations(
|
||||
world.schema,
|
||||
world.obsAnnotations,
|
||||
world.varAnnotations
|
||||
);
|
||||
|
||||
/* build the varDataCache */
|
||||
world.varDataCache = kvCache.map(
|
||||
universe.varDataCache,
|
||||
val => subsetVarData(world, universe, val),
|
||||
{ lowWatermark: VarDataCacheLowWatermark, minTTL: VarDataCacheTTLMs }
|
||||
);
|
||||
/*
|
||||
Var data columns - subset of all
|
||||
*/
|
||||
world.varData = universe.varData.clone();
|
||||
|
||||
return world;
|
||||
}
|
||||
@@ -120,30 +94,24 @@ export function createWorldFromCurrentSelection(universe, world, crossfilter) {
|
||||
const newWorld = templateWorld();
|
||||
|
||||
/* these don't change as only OBS are selected in our current implementation */
|
||||
newWorld.api = universe.api;
|
||||
newWorld.nVar = universe.nVar;
|
||||
newWorld.schema = universe.schema;
|
||||
newWorld.varAnnotations = universe.varAnnotations;
|
||||
|
||||
/* now subset/cut obs */
|
||||
const mask = crossfilter.allFilteredMask();
|
||||
newWorld.obsAnnotations = world.obsAnnotations.icutByMask(mask);
|
||||
newWorld.obsLayout = world.obsLayout.icutByMask(mask);
|
||||
newWorld.obsAnnotations = world.obsAnnotations.isubsetMask(mask);
|
||||
newWorld.obsLayout = world.obsLayout.isubsetMask(mask);
|
||||
newWorld.nObs = newWorld.obsAnnotations.dims[0];
|
||||
|
||||
/* derived data & summaries */
|
||||
newWorld.summary = summarizeAnnotations(
|
||||
newWorld.schema,
|
||||
newWorld.obsAnnotations,
|
||||
newWorld.varAnnotations
|
||||
);
|
||||
|
||||
/* build the varDataCache */
|
||||
newWorld.varDataCache = kvCache.map(
|
||||
universe.varDataCache,
|
||||
val => subsetVarData(newWorld, universe, val),
|
||||
{ lowWatermark: VarDataCacheLowWatermark, minTTL: VarDataCacheTTLMs }
|
||||
);
|
||||
/*
|
||||
Var data columns - subset of all
|
||||
*/
|
||||
if (world.varData.isEmpty()) {
|
||||
newWorld.varData = world.varData.clone();
|
||||
} else {
|
||||
newWorld.varData = world.varData.isubsetMask(mask);
|
||||
}
|
||||
return newWorld;
|
||||
}
|
||||
|
||||
@@ -183,17 +151,10 @@ function deduceDimensionType(attributes, fieldName) {
|
||||
when it is no longer needed
|
||||
(it will not be garbage collected without this call)
|
||||
*/
|
||||
|
||||
export function createVarDimension(
|
||||
world,
|
||||
_worldVarDataCache,
|
||||
crossfilter,
|
||||
geneName
|
||||
) {
|
||||
// return crossfilter.dimension(_worldVarDataCache[geneName], Float32Array);
|
||||
export function createVarDataDimension(world, crossfilter, name) {
|
||||
return crossfilter.dimension(
|
||||
Crossfilter.ScalarDimension,
|
||||
_worldVarDataCache[geneName],
|
||||
world.varData.col(name).asArray(),
|
||||
Float32Array
|
||||
);
|
||||
}
|
||||
@@ -242,14 +203,6 @@ export function worldEqUniverse(world, universe) {
|
||||
return world.obsAnnotations === universe.obsAnnotations;
|
||||
}
|
||||
|
||||
export function subsetVarData(world, universe, varData) {
|
||||
// If world === universe, just return the entire varData array
|
||||
if (worldEqUniverse(world, universe)) {
|
||||
return varData;
|
||||
}
|
||||
return sliceByIndex(varData, world.obsAnnotations.rowIndex.keys());
|
||||
}
|
||||
|
||||
export function getSelectedByIndex(crossfilter) {
|
||||
/*
|
||||
return array of obsIndex, containing all selected obs/cells.
|
||||
|
||||
Reference in New Issue
Block a user