diff --git a/client/__tests__/util/dataframe/histogram.test.js b/client/__tests__/util/dataframe/histogram.test.js new file mode 100644 index 00000000..45dc152c --- /dev/null +++ b/client/__tests__/util/dataframe/histogram.test.js @@ -0,0 +1,69 @@ +import * as Dataframe from "../../../src/util/dataframe"; + +describe("Dataframe column histogram", () => { + test("categorical by categorical", () => { + const df = new Dataframe.Dataframe( + [3, 3], + [["n1", "n2", "n3"], ["c1", "c2", "c3"], new Int32Array([0, 1, 2])], + null, + new Dataframe.KeyIndex(["name", "cat", "value"]) + ); + + const h1 = df.col("cat").histogram(df.col("name")); + expect(h1).toMatchObject( + new Map([ + ["n1", new Map([["c1", 1]])], + ["n2", new Map([["c2", 1]])], + ["n3", new Map([["c3", 1]])] + ]) + ); + // memoized? + expect(df.col("cat").histogram(df.col("name"))).toMatchObject(h1); + }); + + test("continuous by categorical", () => { + const df = new Dataframe.Dataframe( + [3, 3], + [["n1", "n2", "n3"], ["c1", "c2", "c3"], new Int32Array([0, 1, 2])], + null, + new Dataframe.KeyIndex(["name", "cat", "value"]) + ); + + const h1 = df.col("value").histogram(3, [0, 2], df.col("name")); + expect(h1).toMatchObject( + new Map([["n1", [1, 0, 0]], ["n2", [0, 1, 0]], ["n3", [0, 0, 1]]]) + ); + // memoized? + expect(df.col("value").histogram(3, [0, 2], df.col("name"))).toMatchObject( + h1 + ); + }); + + test("categorical", () => { + const df = new Dataframe.Dataframe( + [3, 3], + [["n1", "n2", "n3"], ["c1", "c2", "c3"], new Int32Array([0, 1, 2])], + null, + new Dataframe.KeyIndex(["name", "cat", "value"]) + ); + + const h1 = df.col("cat").histogram(); + expect(h1).toMatchObject(new Map([["c1", 1], ["c2", 1], ["c3", 1]])); + // memoized? + expect(df.col("value").histogram(3, [0, 2])).toMatchObject(h1); + }); + + test("continuous", () => { + const df = new Dataframe.Dataframe( + [3, 3], + [["n1", "n2", "n3"], ["c1", "c2", "c3"], new Int32Array([0, 1, 2])], + null, + new Dataframe.KeyIndex(["name", "cat", "value"]) + ); + + const h1 = df.col("value").histogram(3, [0, 2]); + expect(h1).toMatchObject([1, 1, 1]); + // memoized? + expect(df.col("value").histogram(3, [0, 2])).toMatchObject(h1); + }); +}); diff --git a/client/src/components/categorical/occupancy.js b/client/src/components/categorical/occupancy.js index 6ea86917..d0d8b148 100644 --- a/client/src/components/categorical/occupancy.js +++ b/client/src/components/categorical/occupancy.js @@ -2,60 +2,182 @@ import React from "react"; import { connect } from "react-redux"; import * as d3 from "d3"; +import { + Popover, + PopoverInteractionKind, + Position, + Classes +} from "@blueprintjs/core"; @connect() class Occupancy extends React.Component { - render() { - const { occupancy, colorScale, colorAccessor, schema, world } = this.props; - const width = 100; - const height = 11; + _WIDTH = 100; - const categories = schema.annotations.obsByName[colorAccessor]?.categories; + _HEIGHT = 11; + + createHistogram = () => { + /* + Knowing that colorScale is based off continous data, + createHistogram fetches the continous data in relation to the cells releveant to the catagory value. + It then seperates that data into 50 bins for drawing the mini-histogram + */ + const { + world, + metadataField, + colorAccessor, + category, + categoryIndex + } = this.props; + + if (!this.canvas) return; + + const groupBy = world.obsAnnotations.col(metadataField); + + const col = + world.obsAnnotations.col(colorAccessor) || + world.varData.col(colorAccessor); + + const range = col.summarize(); + + const histogramMap = col.histogram( + 50, + [range.min, range.max], + groupBy + ); /* Because the signature changes we really need different names for histogram to differentiate signatures */ + + const bins = histogramMap.get(category.categoryValues[categoryIndex]); + + const xScale = d3 + .scaleLinear() + .domain([0, bins.length]) + .range([0, this._WIDTH]); + + const largestBin = Math.max(...bins); + + const yScale = d3 + .scaleLinear() + .domain([0, largestBin]) + .range([0, this._HEIGHT]); + + const ctx = this.canvas.getContext("2d"); + + ctx.fillStyle = "#000"; + + let x; + let y; + + const rectWidth = this._WIDTH / bins.length; + + for (let i = 0, { length } = bins; i < length; i += 1) { + x = xScale(i); + y = yScale(bins[i]); + ctx.fillRect(x, this._HEIGHT - y, rectWidth, y); + } + }; + + createOccupancyStack = () => { + /* + Knowing that the color scale is based off of catagorical data, + createOccupancyStack obtains a map showing the number if cells per colored value + Using the colorScale a stack of colored bars is drawn representing the map + */ + const { + world, + metadataField, + colorAccessor, + category, + categoryIndex, + schema, + colorScale + } = this.props; + + const ctx = this.canvas?.getContext("2d"); + + if (!ctx) return; + + const groupBy = world.obsAnnotations.col(metadataField); + const occupancyMap = world.obsAnnotations + .col(colorAccessor) + .histogram(groupBy); + + const occupancy = occupancyMap.get(category.categoryValues[categoryIndex]); const x = d3 .scaleLinear() /* get all the keys d[1] as an array, then find the sum */ - .domain([0, d3.sum(Array.from(occupancy, d => d[1]))]) - .range([0, width]); + .domain([0, d3.sum(Array.from(occupancy.values()))]) + .range([0, this._WIDTH]); + const categories = schema.annotations.obsByName[colorAccessor]?.categories; let currentOffset = 0; const dfColumn = world.obsAnnotations.col(colorAccessor); const categoryValues = dfColumn.summarize().categories; - const stacks = categoryValues.map(d => { - const o = occupancy.get(d); - const scaledValue = x(o); + let o; + let scaledValue; + let value; - const stackItem = { - key: d, - value: o || 0, - rectWidth: o ? scaledValue : 0, - offset: currentOffset, - fill: o ? colorScale(categories.indexOf(d)) : "rgb(255,255,255)" - }; + for (let i = 0, { length } = categoryValues; i < length; i += 1) { + value = categoryValues[i]; + o = occupancy.get(value); + scaledValue = x(o); + ctx.fillStyle = o + ? colorScale(categories.indexOf(value)) + : "rgb(255,255,255)"; + ctx.fillRect(currentOffset, 0, o ? scaledValue : 0, this._HEIGHT); currentOffset += o ? scaledValue : 0; - return stackItem; - }); + } + }; + + render() { + const { colorAccessor, categoricalSelection } = this.props; + + this.canvas?.getContext("2d").clearRect(0, 0, this._WIDTH, this._HEIGHT); + + const colorByIsCatagoricalData = !!categoricalSelection[colorAccessor]; return ( - - {stacks.map(d => ( - - ))} - + { + this.canvas = ref; + if (colorByIsCatagoricalData) this.createOccupancyStack(); + else this.createHistogram(); + }} + /> +
+

+ These histograms show the distribution of{" "} + {colorAccessor} within each category. +
+ The x axis is the same for each histogram, while the y axis is + scaled to the highest bin within each histogram. +

+
+ ); } } diff --git a/client/src/components/categorical/value.js b/client/src/components/categorical/value.js index b538e611..9a50698b 100644 --- a/client/src/components/categorical/value.js +++ b/client/src/components/categorical/value.js @@ -2,7 +2,6 @@ import { connect } from "react-redux"; import React from "react"; import Occupancy from "./occupancy"; -import { countCategoryValues2D } from "../../util/stateManager/worldUtil"; import * as globals from "../../globals"; @connect(state => ({ @@ -22,6 +21,33 @@ class CategoryValue extends React.Component { }); }; + shouldComponentUpdate = nextProps => { + /* + Checks to see if at least one of the following changed: + * world state + * the color accessor (what is currently being colored by) + * if this catagorical value's selection status has changed + + If and only if true, update the component + */ + const { props } = this; + const { metadataField, categoryIndex, categoricalSelection } = props; + const { categoricalSelection: newCategoricalSelection } = nextProps; + + const valueSelectionChange = + categoricalSelection[metadataField].categoryValueSelected[ + categoryIndex + ] !== + newCategoricalSelection[metadataField].categoryValueSelected[ + categoryIndex + ]; + + const worldChange = props.world !== nextProps.world; + const colorAccessorChange = props.colorAccessor !== nextProps.colorAccessor; + + return valueSelectionChange || worldChange || colorAccessorChange; + }; + toggleOn = () => { const { dispatch, metadataField, categoryIndex } = this.props; dispatch({ @@ -57,8 +83,7 @@ class CategoryValue extends React.Component { colorAccessor, colorScale, i, - schema, - world + schema } = this.props; if (!categoricalSelection) return null; @@ -74,20 +99,11 @@ class CategoryValue extends React.Component { /* this is the color scale, so add swatches below */ const isColorBy = metadataField === colorAccessor; let categories = null; - let occupancy = null; if (isColorBy && schema) { categories = schema.annotations.obsByName[colorAccessor]?.categories; } - if (colorAccessor && !isColorBy && categoricalSelection[colorAccessor]) { - occupancy = countCategoryValues2D( - metadataField, - colorAccessor, - world.obsAnnotations - ); - } - return (
{displayString} - {colorAccessor && - !isColorBy && - categoricalSelection[colorAccessor] ? ( - + {colorAccessor && !isColorBy ? ( + ) : null}
diff --git a/client/src/util/dataframe/dataframe.js b/client/src/util/dataframe/dataframe.js index d7ad7f52..2072c9e2 100644 --- a/client/src/util/dataframe/dataframe.js +++ b/client/src/util/dataframe/dataframe.js @@ -1,8 +1,19 @@ import { IdentityInt32Index, isLabelIndex } from "./labelIndex"; // weird cross-dependency that we should clean up someday... import { sortArray } from "../typedCrossfilter/sort"; -import { isTypedArray, isArrayOrTypedArray, callOnceLazy } from "./util"; +import { + isTypedArray, + isArrayOrTypedArray, + callOnceLazy, + memoize +} from "./util"; import { summarizeContinuous, summarizeCategorical } from "./summarize"; +import { + histogramCategorical, + hashCategorical, + histogramContinuous, + hashContinuous +} from "./histogram"; /* Dataframe is an immutable 2D matrix similiar to Python Pandas Dataframe, @@ -59,6 +70,17 @@ Dataframe **/ class Dataframe { + /** + memoization helpers. + **/ + static __DataframeId__ = 0; + + static __getId() { + const id = Dataframe.__DataframeId__; + Dataframe.__DataframeId__ += 1; + return id; + } + /** Constructors & factories **/ @@ -102,6 +124,7 @@ class Dataframe { this.length = nRows; // convenience accessor for row dimension this.rowIndex = rowIndex; this.colIndex = colIndex; + this.__id = Dataframe.__getId(); this.__compile(__columnsAccessor); } @@ -144,7 +167,7 @@ class Dataframe { } } - static __compileColumn(column, getOffset, getLabel) { + static __compileColumn(column, getRowByOffset, getRowByLabel) { /* Each column accessor is a function which will lookup data by index (ie, is equivalent to dataframe.get(row, col), where 'col' @@ -172,12 +195,15 @@ class Dataframe { iget(offset) -- return the value at 'offset' + ... and more ... + */ const { length } = column; + const __id = Dataframe.__getId(); /* get value by row label */ const get = function get(rlabel) { - return column[getOffset(rlabel)]; + return column[getRowByOffset(rlabel)]; }; /* get value by row offset */ @@ -192,7 +218,7 @@ class Dataframe { /* test for row label inclusion in column */ const has = function has(rlabel) { - const offset = getOffset(rlabel); + const offset = getRowByOffset(rlabel); return offset >= 0 && offset < length; }; @@ -212,7 +238,7 @@ class Dataframe { if (offset === -1) { return undefined; } - return getLabel(offset); + return getRowByLabel(offset); }; /* @@ -224,12 +250,25 @@ class Dataframe { : summarizeCategorical(column) ); + /* + Create histogram bins for this column. Memoized. + */ + if (isTypedArray(column)) { + const mFn = memoize(histogramContinuous, hashContinuous); + get.histogram = (bins, domain, by) => mFn(get, bins, domain, by); + } else { + const mFn = memoize(histogramCategorical, hashCategorical); + get.histogram = by => mFn(get, by); + } + get.summarize = summarize; get.asArray = asArray; get.has = has; get.ihas = ihas; get.indexOf = indexOf; get.iget = iget; + get.__id = __id; + return get; } @@ -239,12 +278,15 @@ class Dataframe { Use an existing accessor if provided, else compile a new one. */ - const { getOffset, getLabel } = this.rowIndex; + const { + getOffset: getRowByOffset, + getLabel: getRowByLabel + } = this.rowIndex; this.__columnsAccessor = this.__columns.map((column, idx) => { if (accessors[idx]) { return accessors[idx]; } - return Dataframe.__compileColumn(column, getOffset, getLabel); + return Dataframe.__compileColumn(column, getRowByOffset, getRowByLabel); }); } diff --git a/client/src/util/dataframe/histogram.js b/client/src/util/dataframe/histogram.js new file mode 100644 index 00000000..09f0d17d --- /dev/null +++ b/client/src/util/dataframe/histogram.js @@ -0,0 +1,135 @@ +/* +Dataframe histogram +*/ +import { isTypedArray } from "./util"; + +function _histogramContinuous(column, bins, min, max) { + const valBins = new Array(bins).fill(0); + if (!column) { + return valBins; + } + const binWidth = (max - min) / (bins - 1); + const colArray = column.asArray(); + for (let r = 0, len = colArray.length; r < len; r += 1) { + const val = colArray[r]; + if (val <= max && val >= min) { + // ensure test excludes NaN values + const valBin = (val - min) / binWidth; + valBins[valBin] += 1; + } + } + return valBins; +} + +function _histogramContinuousBy(column, bins, min, max, by) { + const byMap = new Map(); + if (!column || !by) { + return byMap; + } + const binWidth = (max - min) / (bins - 1); + const byArray = by.asArray(); + const colArray = column.asArray(); + for (let r = 0, len = colArray.length; r < len; r += 1) { + const byBin = byArray[r]; + let valBins = byMap.get(byBin); + if (valBins === undefined) { + valBins = new Array(bins).fill(0); + byMap.set(byBin, valBins); + } + const val = colArray[r]; + if (val <= max && val >= min) { + // ensure test excludes NaN values + const valBin = (val - min) / binWidth; + valBins[Math.floor(valBin)] += 1; + } + } + return byMap; +} + +function _histogramCategorical(column) { + const valMap = new Map(); + if (!column) { + return valMap; + } + const colArray = column.asArray(); + for (let r = 0, len = colArray.length; r < len; r += 1) { + const valBin = colArray[r]; + let curCount = valMap.get(valBin); + if (curCount === undefined) { + curCount = 0; + } + valMap.set(valBin, curCount + 1); + } + return valMap; +} + +function _histogramCategoricalBy(column, by) { + const byMap = new Map(); + if (!column || !by) { + return byMap; + } + const byArray = by.asArray(); + const colArray = column.asArray(); + for (let r = 0, len = colArray.length; r < len; r += 1) { + const byBin = byArray[r]; + let valMap = byMap.get(byBin); + if (valMap === undefined) { + valMap = new Map(); + byMap.set(byBin, valMap); + } + const valBin = colArray[r]; + let curCount = valMap.get(valBin); + if (curCount === undefined) { + curCount = 0; + } + valMap.set(valBin, curCount + 1); + } + return byMap; +} + +/* +Count category occupancy. Optional group-by category. +*/ +export function histogramCategorical(column, by) { + if (by && isTypedArray(by)) { + throw new Error("Group by column must be categorical"); + } + return by + ? _histogramCategoricalBy(column, by) + : _histogramCategorical(column); +} + +/* +Memoization hash for histogramCategorical() +*/ +export function hashCategorical(column, by) { + if (by) { + return `${column.__id}:${by.__id}`; + } + return `${column.__id}:`; +} + +/* +Bin counts for continuous/scalar values, with optional group-by category. +Values outside domain are ignored. +*/ +export function histogramContinuous(column, bins = 40, domain = [0, 1], by) { + if (by && isTypedArray(by)) { + throw new Error("Group by column must be categorical"); + } + const [min, max] = domain; + return by + ? _histogramContinuousBy(column, bins, min, max, by) + : _histogramContinuous(column, bins, min, max); +} + +/* +Memoization hash for histogramContinuous +*/ +export function hashContinuous(column, bins = "", domain = [0, 0], by) { + const [min, max] = domain; + if (by) { + return `${column.__id}:${bins}:${min}:${max}:${by.__id}`; + } + return `${column.__id}::${bins}:${min}:${max}`; +} diff --git a/client/src/util/dataframe/util.js b/client/src/util/dataframe/util.js index 8fa1353e..bb674037 100644 --- a/client/src/util/dataframe/util.js +++ b/client/src/util/dataframe/util.js @@ -5,6 +5,10 @@ Private utility code for dataframe export { isTypedArray, isArrayOrTypedArray } from "../typeHelpers"; export function callOnceLazy(f) { + /* + call function once, and save the result, regardless of arguments (this is not + the same as typical memoization). + */ let value; let calledOnce = false; const result = function result(...args) { @@ -14,6 +18,25 @@ export function callOnceLazy(f) { } return value; }; - return result; } + +export function memoize(fn, hashFn) { + /* + function memoization, with user-provided hash. hashFn must return a + key which will be unique as a Map key (ie, obeys "sameValueZero" algorithm + as defined in the JS spec). For more info on hash key, see: + https://developer.mozilla.org/en-US/docs/Web/JavaScript/Reference/Global_Objects/Map#Key_equality + */ + const cache = new Map(); + const wrap = function wrap(...args) { + const key = hashFn(...args); + if (cache.has(key)) { + return cache.get(key); + } + const result = fn(...args); + cache.set(key, result); + return result; + }; + return wrap; +}