mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-10-01 21:48:12 +08:00
categorical vs continuous mini histograms (#827)
* comment * add histogram functionality to Dataframe; port category occupancy to use it * fix binning and create histogram for continous by catagorical * Remove unnecessary logs * Begin work on KDE * Replace broken KDE with working histogram * Define domain and range based on data from histogram * Fix occupancy * Add continuous obs and switch to canvas * Stop value from always rerendering * clear before render * Clear canvas on render * refactor categorical occupancy to canvas * Remove log * simplify finding max * refactor kde->histogram and occupancy->bins * refactor svg -> canvas * rename to occupancy stack * create popup * add metadata and categorical values to popup * fix overflow * remove zeros info * style graph * fix shouldComponentUpdate to look for world changes * change categorySelected -> categoryValueSelected * refactor out render * remove comment * conditionally have bottom border * remove diff comp * remove comments * remove unnecessary mapping * Add comments describing drawing functions * comments * flip comparison order * remove logging * move default to parameter * move defaults to parameter * disable popover if not showing histogram * fix wording and styling * add line break
This commit is contained in:
@@ -1,8 +1,19 @@
|
||||
import { IdentityInt32Index, isLabelIndex } from "./labelIndex";
|
||||
// weird cross-dependency that we should clean up someday...
|
||||
import { sortArray } from "../typedCrossfilter/sort";
|
||||
import { isTypedArray, isArrayOrTypedArray, callOnceLazy } from "./util";
|
||||
import {
|
||||
isTypedArray,
|
||||
isArrayOrTypedArray,
|
||||
callOnceLazy,
|
||||
memoize
|
||||
} from "./util";
|
||||
import { summarizeContinuous, summarizeCategorical } from "./summarize";
|
||||
import {
|
||||
histogramCategorical,
|
||||
hashCategorical,
|
||||
histogramContinuous,
|
||||
hashContinuous
|
||||
} from "./histogram";
|
||||
|
||||
/*
|
||||
Dataframe is an immutable 2D matrix similiar to Python Pandas Dataframe,
|
||||
@@ -59,6 +70,17 @@ Dataframe
|
||||
**/
|
||||
|
||||
class Dataframe {
|
||||
/**
|
||||
memoization helpers.
|
||||
**/
|
||||
static __DataframeId__ = 0;
|
||||
|
||||
static __getId() {
|
||||
const id = Dataframe.__DataframeId__;
|
||||
Dataframe.__DataframeId__ += 1;
|
||||
return id;
|
||||
}
|
||||
|
||||
/**
|
||||
Constructors & factories
|
||||
**/
|
||||
@@ -102,6 +124,7 @@ class Dataframe {
|
||||
this.length = nRows; // convenience accessor for row dimension
|
||||
this.rowIndex = rowIndex;
|
||||
this.colIndex = colIndex;
|
||||
this.__id = Dataframe.__getId();
|
||||
|
||||
this.__compile(__columnsAccessor);
|
||||
}
|
||||
@@ -144,7 +167,7 @@ class Dataframe {
|
||||
}
|
||||
}
|
||||
|
||||
static __compileColumn(column, getOffset, getLabel) {
|
||||
static __compileColumn(column, getRowByOffset, getRowByLabel) {
|
||||
/*
|
||||
Each column accessor is a function which will lookup data by
|
||||
index (ie, is equivalent to dataframe.get(row, col), where 'col'
|
||||
@@ -172,12 +195,15 @@ class Dataframe {
|
||||
|
||||
iget(offset) -- return the value at 'offset'
|
||||
|
||||
... and more ...
|
||||
|
||||
*/
|
||||
const { length } = column;
|
||||
const __id = Dataframe.__getId();
|
||||
|
||||
/* get value by row label */
|
||||
const get = function get(rlabel) {
|
||||
return column[getOffset(rlabel)];
|
||||
return column[getRowByOffset(rlabel)];
|
||||
};
|
||||
|
||||
/* get value by row offset */
|
||||
@@ -192,7 +218,7 @@ class Dataframe {
|
||||
|
||||
/* test for row label inclusion in column */
|
||||
const has = function has(rlabel) {
|
||||
const offset = getOffset(rlabel);
|
||||
const offset = getRowByOffset(rlabel);
|
||||
return offset >= 0 && offset < length;
|
||||
};
|
||||
|
||||
@@ -212,7 +238,7 @@ class Dataframe {
|
||||
if (offset === -1) {
|
||||
return undefined;
|
||||
}
|
||||
return getLabel(offset);
|
||||
return getRowByLabel(offset);
|
||||
};
|
||||
|
||||
/*
|
||||
@@ -224,12 +250,25 @@ class Dataframe {
|
||||
: summarizeCategorical(column)
|
||||
);
|
||||
|
||||
/*
|
||||
Create histogram bins for this column. Memoized.
|
||||
*/
|
||||
if (isTypedArray(column)) {
|
||||
const mFn = memoize(histogramContinuous, hashContinuous);
|
||||
get.histogram = (bins, domain, by) => mFn(get, bins, domain, by);
|
||||
} else {
|
||||
const mFn = memoize(histogramCategorical, hashCategorical);
|
||||
get.histogram = by => mFn(get, by);
|
||||
}
|
||||
|
||||
get.summarize = summarize;
|
||||
get.asArray = asArray;
|
||||
get.has = has;
|
||||
get.ihas = ihas;
|
||||
get.indexOf = indexOf;
|
||||
get.iget = iget;
|
||||
get.__id = __id;
|
||||
|
||||
return get;
|
||||
}
|
||||
|
||||
@@ -239,12 +278,15 @@ class Dataframe {
|
||||
|
||||
Use an existing accessor if provided, else compile a new one.
|
||||
*/
|
||||
const { getOffset, getLabel } = this.rowIndex;
|
||||
const {
|
||||
getOffset: getRowByOffset,
|
||||
getLabel: getRowByLabel
|
||||
} = this.rowIndex;
|
||||
this.__columnsAccessor = this.__columns.map((column, idx) => {
|
||||
if (accessors[idx]) {
|
||||
return accessors[idx];
|
||||
}
|
||||
return Dataframe.__compileColumn(column, getOffset, getLabel);
|
||||
return Dataframe.__compileColumn(column, getRowByOffset, getRowByLabel);
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
/*
|
||||
Dataframe histogram
|
||||
*/
|
||||
import { isTypedArray } from "./util";
|
||||
|
||||
function _histogramContinuous(column, bins, min, max) {
|
||||
const valBins = new Array(bins).fill(0);
|
||||
if (!column) {
|
||||
return valBins;
|
||||
}
|
||||
const binWidth = (max - min) / (bins - 1);
|
||||
const colArray = column.asArray();
|
||||
for (let r = 0, len = colArray.length; r < len; r += 1) {
|
||||
const val = colArray[r];
|
||||
if (val <= max && val >= min) {
|
||||
// ensure test excludes NaN values
|
||||
const valBin = (val - min) / binWidth;
|
||||
valBins[valBin] += 1;
|
||||
}
|
||||
}
|
||||
return valBins;
|
||||
}
|
||||
|
||||
function _histogramContinuousBy(column, bins, min, max, by) {
|
||||
const byMap = new Map();
|
||||
if (!column || !by) {
|
||||
return byMap;
|
||||
}
|
||||
const binWidth = (max - min) / (bins - 1);
|
||||
const byArray = by.asArray();
|
||||
const colArray = column.asArray();
|
||||
for (let r = 0, len = colArray.length; r < len; r += 1) {
|
||||
const byBin = byArray[r];
|
||||
let valBins = byMap.get(byBin);
|
||||
if (valBins === undefined) {
|
||||
valBins = new Array(bins).fill(0);
|
||||
byMap.set(byBin, valBins);
|
||||
}
|
||||
const val = colArray[r];
|
||||
if (val <= max && val >= min) {
|
||||
// ensure test excludes NaN values
|
||||
const valBin = (val - min) / binWidth;
|
||||
valBins[Math.floor(valBin)] += 1;
|
||||
}
|
||||
}
|
||||
return byMap;
|
||||
}
|
||||
|
||||
function _histogramCategorical(column) {
|
||||
const valMap = new Map();
|
||||
if (!column) {
|
||||
return valMap;
|
||||
}
|
||||
const colArray = column.asArray();
|
||||
for (let r = 0, len = colArray.length; r < len; r += 1) {
|
||||
const valBin = colArray[r];
|
||||
let curCount = valMap.get(valBin);
|
||||
if (curCount === undefined) {
|
||||
curCount = 0;
|
||||
}
|
||||
valMap.set(valBin, curCount + 1);
|
||||
}
|
||||
return valMap;
|
||||
}
|
||||
|
||||
function _histogramCategoricalBy(column, by) {
|
||||
const byMap = new Map();
|
||||
if (!column || !by) {
|
||||
return byMap;
|
||||
}
|
||||
const byArray = by.asArray();
|
||||
const colArray = column.asArray();
|
||||
for (let r = 0, len = colArray.length; r < len; r += 1) {
|
||||
const byBin = byArray[r];
|
||||
let valMap = byMap.get(byBin);
|
||||
if (valMap === undefined) {
|
||||
valMap = new Map();
|
||||
byMap.set(byBin, valMap);
|
||||
}
|
||||
const valBin = colArray[r];
|
||||
let curCount = valMap.get(valBin);
|
||||
if (curCount === undefined) {
|
||||
curCount = 0;
|
||||
}
|
||||
valMap.set(valBin, curCount + 1);
|
||||
}
|
||||
return byMap;
|
||||
}
|
||||
|
||||
/*
|
||||
Count category occupancy. Optional group-by category.
|
||||
*/
|
||||
export function histogramCategorical(column, by) {
|
||||
if (by && isTypedArray(by)) {
|
||||
throw new Error("Group by column must be categorical");
|
||||
}
|
||||
return by
|
||||
? _histogramCategoricalBy(column, by)
|
||||
: _histogramCategorical(column);
|
||||
}
|
||||
|
||||
/*
|
||||
Memoization hash for histogramCategorical()
|
||||
*/
|
||||
export function hashCategorical(column, by) {
|
||||
if (by) {
|
||||
return `${column.__id}:${by.__id}`;
|
||||
}
|
||||
return `${column.__id}:`;
|
||||
}
|
||||
|
||||
/*
|
||||
Bin counts for continuous/scalar values, with optional group-by category.
|
||||
Values outside domain are ignored.
|
||||
*/
|
||||
export function histogramContinuous(column, bins = 40, domain = [0, 1], by) {
|
||||
if (by && isTypedArray(by)) {
|
||||
throw new Error("Group by column must be categorical");
|
||||
}
|
||||
const [min, max] = domain;
|
||||
return by
|
||||
? _histogramContinuousBy(column, bins, min, max, by)
|
||||
: _histogramContinuous(column, bins, min, max);
|
||||
}
|
||||
|
||||
/*
|
||||
Memoization hash for histogramContinuous
|
||||
*/
|
||||
export function hashContinuous(column, bins = "", domain = [0, 0], by) {
|
||||
const [min, max] = domain;
|
||||
if (by) {
|
||||
return `${column.__id}:${bins}:${min}:${max}:${by.__id}`;
|
||||
}
|
||||
return `${column.__id}::${bins}:${min}:${max}`;
|
||||
}
|
||||
@@ -5,6 +5,10 @@ Private utility code for dataframe
|
||||
export { isTypedArray, isArrayOrTypedArray } from "../typeHelpers";
|
||||
|
||||
export function callOnceLazy(f) {
|
||||
/*
|
||||
call function once, and save the result, regardless of arguments (this is not
|
||||
the same as typical memoization).
|
||||
*/
|
||||
let value;
|
||||
let calledOnce = false;
|
||||
const result = function result(...args) {
|
||||
@@ -14,6 +18,25 @@ export function callOnceLazy(f) {
|
||||
}
|
||||
return value;
|
||||
};
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
export function memoize(fn, hashFn) {
|
||||
/*
|
||||
function memoization, with user-provided hash. hashFn must return a
|
||||
key which will be unique as a Map key (ie, obeys "sameValueZero" algorithm
|
||||
as defined in the JS spec). For more info on hash key, see:
|
||||
https://developer.mozilla.org/en-US/docs/Web/JavaScript/Reference/Global_Objects/Map#Key_equality
|
||||
*/
|
||||
const cache = new Map();
|
||||
const wrap = function wrap(...args) {
|
||||
const key = hashFn(...args);
|
||||
if (cache.has(key)) {
|
||||
return cache.get(key);
|
||||
}
|
||||
const result = fn(...args);
|
||||
cache.set(key, result);
|
||||
return result;
|
||||
};
|
||||
return wrap;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user