categorical vs continuous mini histograms (#827)

* comment

* add histogram functionality to Dataframe; port category occupancy to use it

* fix binning and create histogram for continous by catagorical

* Remove unnecessary logs

* Begin work on KDE

* Replace broken KDE with working histogram

* Define domain and range based on data from histogram

* Fix occupancy

* Add continuous obs and switch to canvas

* Stop value from always rerendering

* clear before render

* Clear canvas on render

* refactor categorical occupancy to canvas

* Remove log

* simplify finding max

* refactor kde->histogram and occupancy->bins

* refactor svg -> canvas

* rename to occupancy stack

* create popup

* add metadata and categorical values to popup

* fix overflow

* remove zeros info

* style graph

* fix shouldComponentUpdate to look for world changes

* change categorySelected -> categoryValueSelected

* refactor out render

* remove comment

* conditionally have bottom border

* remove diff comp

* remove comments

* remove unnecessary mapping

* Add comments describing drawing functions

* comments

* flip comparison order

* remove logging

* move default to parameter

* move defaults to parameter

* disable popover if not showing histogram

* fix wording and styling

* add line break
This commit is contained in:
Severiano Badajoz
2019-07-09 11:19:01 -07:00
committed by GitHub
parent 722a91f1d2
commit 941c297363
6 changed files with 465 additions and 64 deletions
@@ -0,0 +1,69 @@
import * as Dataframe from "../../../src/util/dataframe";
describe("Dataframe column histogram", () => {
test("categorical by categorical", () => {
const df = new Dataframe.Dataframe(
[3, 3],
[["n1", "n2", "n3"], ["c1", "c2", "c3"], new Int32Array([0, 1, 2])],
null,
new Dataframe.KeyIndex(["name", "cat", "value"])
);
const h1 = df.col("cat").histogram(df.col("name"));
expect(h1).toMatchObject(
new Map([
["n1", new Map([["c1", 1]])],
["n2", new Map([["c2", 1]])],
["n3", new Map([["c3", 1]])]
])
);
// memoized?
expect(df.col("cat").histogram(df.col("name"))).toMatchObject(h1);
});
test("continuous by categorical", () => {
const df = new Dataframe.Dataframe(
[3, 3],
[["n1", "n2", "n3"], ["c1", "c2", "c3"], new Int32Array([0, 1, 2])],
null,
new Dataframe.KeyIndex(["name", "cat", "value"])
);
const h1 = df.col("value").histogram(3, [0, 2], df.col("name"));
expect(h1).toMatchObject(
new Map([["n1", [1, 0, 0]], ["n2", [0, 1, 0]], ["n3", [0, 0, 1]]])
);
// memoized?
expect(df.col("value").histogram(3, [0, 2], df.col("name"))).toMatchObject(
h1
);
});
test("categorical", () => {
const df = new Dataframe.Dataframe(
[3, 3],
[["n1", "n2", "n3"], ["c1", "c2", "c3"], new Int32Array([0, 1, 2])],
null,
new Dataframe.KeyIndex(["name", "cat", "value"])
);
const h1 = df.col("cat").histogram();
expect(h1).toMatchObject(new Map([["c1", 1], ["c2", 1], ["c3", 1]]));
// memoized?
expect(df.col("value").histogram(3, [0, 2])).toMatchObject(h1);
});
test("continuous", () => {
const df = new Dataframe.Dataframe(
[3, 3],
[["n1", "n2", "n3"], ["c1", "c2", "c3"], new Int32Array([0, 1, 2])],
null,
new Dataframe.KeyIndex(["name", "cat", "value"])
);
const h1 = df.col("value").histogram(3, [0, 2]);
expect(h1).toMatchObject([1, 1, 1]);
// memoized?
expect(df.col("value").histogram(3, [0, 2])).toMatchObject(h1);
});
});
+157 -35
View File
@@ -2,60 +2,182 @@
import React from "react";
import { connect } from "react-redux";
import * as d3 from "d3";
import {
Popover,
PopoverInteractionKind,
Position,
Classes
} from "@blueprintjs/core";
@connect()
class Occupancy extends React.Component {
render() {
const { occupancy, colorScale, colorAccessor, schema, world } = this.props;
const width = 100;
const height = 11;
_WIDTH = 100;
const categories = schema.annotations.obsByName[colorAccessor]?.categories;
_HEIGHT = 11;
createHistogram = () => {
/*
Knowing that colorScale is based off continous data,
createHistogram fetches the continous data in relation to the cells releveant to the catagory value.
It then seperates that data into 50 bins for drawing the mini-histogram
*/
const {
world,
metadataField,
colorAccessor,
category,
categoryIndex
} = this.props;
if (!this.canvas) return;
const groupBy = world.obsAnnotations.col(metadataField);
const col =
world.obsAnnotations.col(colorAccessor) ||
world.varData.col(colorAccessor);
const range = col.summarize();
const histogramMap = col.histogram(
50,
[range.min, range.max],
groupBy
); /* Because the signature changes we really need different names for histogram to differentiate signatures */
const bins = histogramMap.get(category.categoryValues[categoryIndex]);
const xScale = d3
.scaleLinear()
.domain([0, bins.length])
.range([0, this._WIDTH]);
const largestBin = Math.max(...bins);
const yScale = d3
.scaleLinear()
.domain([0, largestBin])
.range([0, this._HEIGHT]);
const ctx = this.canvas.getContext("2d");
ctx.fillStyle = "#000";
let x;
let y;
const rectWidth = this._WIDTH / bins.length;
for (let i = 0, { length } = bins; i < length; i += 1) {
x = xScale(i);
y = yScale(bins[i]);
ctx.fillRect(x, this._HEIGHT - y, rectWidth, y);
}
};
createOccupancyStack = () => {
/*
Knowing that the color scale is based off of catagorical data,
createOccupancyStack obtains a map showing the number if cells per colored value
Using the colorScale a stack of colored bars is drawn representing the map
*/
const {
world,
metadataField,
colorAccessor,
category,
categoryIndex,
schema,
colorScale
} = this.props;
const ctx = this.canvas?.getContext("2d");
if (!ctx) return;
const groupBy = world.obsAnnotations.col(metadataField);
const occupancyMap = world.obsAnnotations
.col(colorAccessor)
.histogram(groupBy);
const occupancy = occupancyMap.get(category.categoryValues[categoryIndex]);
const x = d3
.scaleLinear()
/* get all the keys d[1] as an array, then find the sum */
.domain([0, d3.sum(Array.from(occupancy, d => d[1]))])
.range([0, width]);
.domain([0, d3.sum(Array.from(occupancy.values()))])
.range([0, this._WIDTH]);
const categories = schema.annotations.obsByName[colorAccessor]?.categories;
let currentOffset = 0;
const dfColumn = world.obsAnnotations.col(colorAccessor);
const categoryValues = dfColumn.summarize().categories;
const stacks = categoryValues.map(d => {
const o = occupancy.get(d);
const scaledValue = x(o);
let o;
let scaledValue;
let value;
const stackItem = {
key: d,
value: o || 0,
rectWidth: o ? scaledValue : 0,
offset: currentOffset,
fill: o ? colorScale(categories.indexOf(d)) : "rgb(255,255,255)"
};
for (let i = 0, { length } = categoryValues; i < length; i += 1) {
value = categoryValues[i];
o = occupancy.get(value);
scaledValue = x(o);
ctx.fillStyle = o
? colorScale(categories.indexOf(value))
: "rgb(255,255,255)";
ctx.fillRect(currentOffset, 0, o ? scaledValue : 0, this._HEIGHT);
currentOffset += o ? scaledValue : 0;
return stackItem;
});
}
};
render() {
const { colorAccessor, categoricalSelection } = this.props;
this.canvas?.getContext("2d").clearRect(0, 0, this._WIDTH, this._HEIGHT);
const colorByIsCatagoricalData = !!categoricalSelection[colorAccessor];
return (
<svg
style={{
marginRight: 5,
width,
height
<Popover
interactionKind={PopoverInteractionKind.HOVER}
hoverOpenDelay={1000}
position={Position.LEFT}
modifiers={{
preventOverflow: { enabled: false },
hide: { enabled: false }
}}
lazy
usePortal
disabled={colorByIsCatagoricalData}
popoverClassName={Classes.POPOVER_CONTENT_SIZING}
>
{stacks.map(d => (
<rect
key={d.key}
width={d.rectWidth}
height={height}
x={d.offset}
title={d.metadataField}
fill={d.fill}
/>
))}
</svg>
<canvas
className="bp3-popover-targer"
style={{
marginRight: 5,
width: this._WIDTH,
height: this._HEIGHT,
borderBottom: colorByIsCatagoricalData
? ""
: "solid rgb(230, 230, 230) 0.25px"
}}
width={this._WIDTH}
height={this._HEIGHT}
ref={ref => {
this.canvas = ref;
if (colorByIsCatagoricalData) this.createOccupancyStack();
else this.createHistogram();
}}
/>
<div key="text" style={{ fontFamily: "Roboto", fontSize: "14px" }}>
<p style={{ margin: "0" }}>
These histograms show the distribution of{" "}
<strong>{colorAccessor}</strong> within each category.
<br />
The x axis is the same for each histogram, while the y axis is
scaled to the highest bin within each histogram.
</p>
</div>
</Popover>
);
}
}
+31 -21
View File
@@ -2,7 +2,6 @@
import { connect } from "react-redux";
import React from "react";
import Occupancy from "./occupancy";
import { countCategoryValues2D } from "../../util/stateManager/worldUtil";
import * as globals from "../../globals";
@connect(state => ({
@@ -22,6 +21,33 @@ class CategoryValue extends React.Component {
});
};
shouldComponentUpdate = nextProps => {
/*
Checks to see if at least one of the following changed:
* world state
* the color accessor (what is currently being colored by)
* if this catagorical value's selection status has changed
If and only if true, update the component
*/
const { props } = this;
const { metadataField, categoryIndex, categoricalSelection } = props;
const { categoricalSelection: newCategoricalSelection } = nextProps;
const valueSelectionChange =
categoricalSelection[metadataField].categoryValueSelected[
categoryIndex
] !==
newCategoricalSelection[metadataField].categoryValueSelected[
categoryIndex
];
const worldChange = props.world !== nextProps.world;
const colorAccessorChange = props.colorAccessor !== nextProps.colorAccessor;
return valueSelectionChange || worldChange || colorAccessorChange;
};
toggleOn = () => {
const { dispatch, metadataField, categoryIndex } = this.props;
dispatch({
@@ -57,8 +83,7 @@ class CategoryValue extends React.Component {
colorAccessor,
colorScale,
i,
schema,
world
schema
} = this.props;
if (!categoricalSelection) return null;
@@ -74,20 +99,11 @@ class CategoryValue extends React.Component {
/* this is the color scale, so add swatches below */
const isColorBy = metadataField === colorAccessor;
let categories = null;
let occupancy = null;
if (isColorBy && schema) {
categories = schema.annotations.obsByName[colorAccessor]?.categories;
}
if (colorAccessor && !isColorBy && categoricalSelection[colorAccessor]) {
occupancy = countCategoryValues2D(
metadataField,
colorAccessor,
world.obsAnnotations
);
}
return (
<div
key={i}
@@ -122,20 +138,14 @@ class CategoryValue extends React.Component {
<span
data-testid={`categorical-value-${metadataField}-${displayString}`}
data-testclass="categorical-value"
style={{ wordBreak: "break-all" }}
>
{displayString}
</span>
</label>
<span style={{ flexShrink: 0 }}>
{colorAccessor &&
!isColorBy &&
categoricalSelection[colorAccessor] ? (
<Occupancy
occupancy={occupancy.get(
category.categoryValues[categoryIndex]
)}
{...this.props}
/>
{colorAccessor && !isColorBy ? (
<Occupancy category={category} {...this.props} />
) : null}
</span>
</div>
+49 -7
View File
@@ -1,8 +1,19 @@
import { IdentityInt32Index, isLabelIndex } from "./labelIndex";
// weird cross-dependency that we should clean up someday...
import { sortArray } from "../typedCrossfilter/sort";
import { isTypedArray, isArrayOrTypedArray, callOnceLazy } from "./util";
import {
isTypedArray,
isArrayOrTypedArray,
callOnceLazy,
memoize
} from "./util";
import { summarizeContinuous, summarizeCategorical } from "./summarize";
import {
histogramCategorical,
hashCategorical,
histogramContinuous,
hashContinuous
} from "./histogram";
/*
Dataframe is an immutable 2D matrix similiar to Python Pandas Dataframe,
@@ -59,6 +70,17 @@ Dataframe
**/
class Dataframe {
/**
memoization helpers.
**/
static __DataframeId__ = 0;
static __getId() {
const id = Dataframe.__DataframeId__;
Dataframe.__DataframeId__ += 1;
return id;
}
/**
Constructors & factories
**/
@@ -102,6 +124,7 @@ class Dataframe {
this.length = nRows; // convenience accessor for row dimension
this.rowIndex = rowIndex;
this.colIndex = colIndex;
this.__id = Dataframe.__getId();
this.__compile(__columnsAccessor);
}
@@ -144,7 +167,7 @@ class Dataframe {
}
}
static __compileColumn(column, getOffset, getLabel) {
static __compileColumn(column, getRowByOffset, getRowByLabel) {
/*
Each column accessor is a function which will lookup data by
index (ie, is equivalent to dataframe.get(row, col), where 'col'
@@ -172,12 +195,15 @@ class Dataframe {
iget(offset) -- return the value at 'offset'
... and more ...
*/
const { length } = column;
const __id = Dataframe.__getId();
/* get value by row label */
const get = function get(rlabel) {
return column[getOffset(rlabel)];
return column[getRowByOffset(rlabel)];
};
/* get value by row offset */
@@ -192,7 +218,7 @@ class Dataframe {
/* test for row label inclusion in column */
const has = function has(rlabel) {
const offset = getOffset(rlabel);
const offset = getRowByOffset(rlabel);
return offset >= 0 && offset < length;
};
@@ -212,7 +238,7 @@ class Dataframe {
if (offset === -1) {
return undefined;
}
return getLabel(offset);
return getRowByLabel(offset);
};
/*
@@ -224,12 +250,25 @@ class Dataframe {
: summarizeCategorical(column)
);
/*
Create histogram bins for this column. Memoized.
*/
if (isTypedArray(column)) {
const mFn = memoize(histogramContinuous, hashContinuous);
get.histogram = (bins, domain, by) => mFn(get, bins, domain, by);
} else {
const mFn = memoize(histogramCategorical, hashCategorical);
get.histogram = by => mFn(get, by);
}
get.summarize = summarize;
get.asArray = asArray;
get.has = has;
get.ihas = ihas;
get.indexOf = indexOf;
get.iget = iget;
get.__id = __id;
return get;
}
@@ -239,12 +278,15 @@ class Dataframe {
Use an existing accessor if provided, else compile a new one.
*/
const { getOffset, getLabel } = this.rowIndex;
const {
getOffset: getRowByOffset,
getLabel: getRowByLabel
} = this.rowIndex;
this.__columnsAccessor = this.__columns.map((column, idx) => {
if (accessors[idx]) {
return accessors[idx];
}
return Dataframe.__compileColumn(column, getOffset, getLabel);
return Dataframe.__compileColumn(column, getRowByOffset, getRowByLabel);
});
}
+135
View File
@@ -0,0 +1,135 @@
/*
Dataframe histogram
*/
import { isTypedArray } from "./util";
function _histogramContinuous(column, bins, min, max) {
const valBins = new Array(bins).fill(0);
if (!column) {
return valBins;
}
const binWidth = (max - min) / (bins - 1);
const colArray = column.asArray();
for (let r = 0, len = colArray.length; r < len; r += 1) {
const val = colArray[r];
if (val <= max && val >= min) {
// ensure test excludes NaN values
const valBin = (val - min) / binWidth;
valBins[valBin] += 1;
}
}
return valBins;
}
function _histogramContinuousBy(column, bins, min, max, by) {
const byMap = new Map();
if (!column || !by) {
return byMap;
}
const binWidth = (max - min) / (bins - 1);
const byArray = by.asArray();
const colArray = column.asArray();
for (let r = 0, len = colArray.length; r < len; r += 1) {
const byBin = byArray[r];
let valBins = byMap.get(byBin);
if (valBins === undefined) {
valBins = new Array(bins).fill(0);
byMap.set(byBin, valBins);
}
const val = colArray[r];
if (val <= max && val >= min) {
// ensure test excludes NaN values
const valBin = (val - min) / binWidth;
valBins[Math.floor(valBin)] += 1;
}
}
return byMap;
}
function _histogramCategorical(column) {
const valMap = new Map();
if (!column) {
return valMap;
}
const colArray = column.asArray();
for (let r = 0, len = colArray.length; r < len; r += 1) {
const valBin = colArray[r];
let curCount = valMap.get(valBin);
if (curCount === undefined) {
curCount = 0;
}
valMap.set(valBin, curCount + 1);
}
return valMap;
}
function _histogramCategoricalBy(column, by) {
const byMap = new Map();
if (!column || !by) {
return byMap;
}
const byArray = by.asArray();
const colArray = column.asArray();
for (let r = 0, len = colArray.length; r < len; r += 1) {
const byBin = byArray[r];
let valMap = byMap.get(byBin);
if (valMap === undefined) {
valMap = new Map();
byMap.set(byBin, valMap);
}
const valBin = colArray[r];
let curCount = valMap.get(valBin);
if (curCount === undefined) {
curCount = 0;
}
valMap.set(valBin, curCount + 1);
}
return byMap;
}
/*
Count category occupancy. Optional group-by category.
*/
export function histogramCategorical(column, by) {
if (by && isTypedArray(by)) {
throw new Error("Group by column must be categorical");
}
return by
? _histogramCategoricalBy(column, by)
: _histogramCategorical(column);
}
/*
Memoization hash for histogramCategorical()
*/
export function hashCategorical(column, by) {
if (by) {
return `${column.__id}:${by.__id}`;
}
return `${column.__id}:`;
}
/*
Bin counts for continuous/scalar values, with optional group-by category.
Values outside domain are ignored.
*/
export function histogramContinuous(column, bins = 40, domain = [0, 1], by) {
if (by && isTypedArray(by)) {
throw new Error("Group by column must be categorical");
}
const [min, max] = domain;
return by
? _histogramContinuousBy(column, bins, min, max, by)
: _histogramContinuous(column, bins, min, max);
}
/*
Memoization hash for histogramContinuous
*/
export function hashContinuous(column, bins = "", domain = [0, 0], by) {
const [min, max] = domain;
if (by) {
return `${column.__id}:${bins}:${min}:${max}:${by.__id}`;
}
return `${column.__id}::${bins}:${min}:${max}`;
}
+24 -1
View File
@@ -5,6 +5,10 @@ Private utility code for dataframe
export { isTypedArray, isArrayOrTypedArray } from "../typeHelpers";
export function callOnceLazy(f) {
/*
call function once, and save the result, regardless of arguments (this is not
the same as typical memoization).
*/
let value;
let calledOnce = false;
const result = function result(...args) {
@@ -14,6 +18,25 @@ export function callOnceLazy(f) {
}
return value;
};
return result;
}
export function memoize(fn, hashFn) {
/*
function memoization, with user-provided hash. hashFn must return a
key which will be unique as a Map key (ie, obeys "sameValueZero" algorithm
as defined in the JS spec). For more info on hash key, see:
https://developer.mozilla.org/en-US/docs/Web/JavaScript/Reference/Global_Objects/Map#Key_equality
*/
const cache = new Map();
const wrap = function wrap(...args) {
const key = hashFn(...args);
if (cache.has(key)) {
return cache.get(key);
}
const result = fn(...args);
cache.set(key, result);
return result;
};
return wrap;
}