Dataframe (#576)

* initial dataframe commit

* initial dataframe port of core app

* rename variables for clarity

* remove unused import

* comment out unused code

* fix array handling bug in crossfilter dimension creation

* allow creation of empty dataframes

* handle non-existent columns

* handle non-existent columns

* revise tests for new dataframe

* comments for clarity

* comments for clarity

* generate bulk add placeholder with real gene names

* fix bug in gene name adding

* more dataframe unit tests

* fix bug - subset from current world, not universe

* put cut and pasted code into a single function

* improve caching of crossfilter

* remove cascading update bug from graph

* more performance work

* improve state handling for scatterplot

* performance optimization of critical path

* add column summarization

* dataframe utils

* add callOnceLazy

* fix tests

* minor updates found during review

* fix misspelling

* remove RESTv02 from function names

* comment cleanup

* cut/icut col parameter defaults to null

* break up large test

* improve tests and comments on dataframe at/has functions
This commit is contained in:
Bruce Martin
2019-02-22 11:31:34 -08:00
committed by GitHub
parent 57c4e9ff33
commit 6b33315cbe
29 changed files with 1798 additions and 611 deletions
@@ -157,16 +157,6 @@ const anAnnotationsVarFBSResponse = (() => {
return encodeMatrix(columns, anAnnotationsVarJSONResponse.names);
})();
const aLayoutJSONResponse = {
layout: {
ndims: 2,
coordinates: _()
.range(nObs)
.map(idx => [idx, Math.random(), Math.random()])
.value()
}
};
const aLayoutFBSResponse = (() => {
const coords = [
new Float32Array(nObs).fill(Math.random()),
@@ -190,7 +180,7 @@ const aLayoutFBSResponse = (() => {
NetEncoding.Matrix.startMatrix(builder);
NetEncoding.Matrix.addNRows(builder, nObs);
NetEncoding.Matrix.addNCols(builder, nVar);
NetEncoding.Matrix.addNCols(builder, coords.length);
NetEncoding.Matrix.addColumns(builder, columns);
const matrix = NetEncoding.Matrix.endMatrix(builder);
builder.finish(matrix);
@@ -1,4 +1,9 @@
import summarizeAnnotations from "../../../src/util/stateManager/summarizeAnnotations";
import * as Dataframe from "../../../src/util/dataframe";
function float32Conversion(f) {
return new Float32Array([39.3])[0];
}
describe("summarizeAnnotations", () => {
const schema = {
@@ -20,7 +25,8 @@ describe("summarizeAnnotations", () => {
};
test("empty test", () => {
const summary = summarizeAnnotations(schema, [], []);
const df = Dataframe.Dataframe.empty();
const summary = summarizeAnnotations(schema, df, df.clone());
expect(summary).toEqual(
expect.objectContaining({
obs: {
@@ -69,18 +75,27 @@ describe("summarizeAnnotations", () => {
});
test("simple test", () => {
const obsAnnotations = [
{
__index__: 0,
name: "n1",
nameString: "hi",
nameBoolean: true,
nameFloat32: 39.3,
nameInt32: 99,
nameCategorical: 1
}
];
const varAnnotations = [];
const obsAnnotations = new Dataframe.Dataframe(
[1, 6],
[
["n1"],
["hi"],
[true],
new Float32Array([39.3]),
new Int32Array([99]),
[1]
],
null,
new Dataframe.KeyIndex([
"name",
"nameString",
"nameBoolean",
"nameFloat32",
"nameInt32",
"nameCategorical"
])
);
const varAnnotations = Dataframe.Dataframe.empty();
const summary = summarizeAnnotations(
schema,
@@ -105,7 +120,13 @@ describe("summarizeAnnotations", () => {
},
nameFloat32: {
categorical: false,
range: { min: 39.3, max: 39.3, nan: 0, ninf: 0, pinf: 0 }
range: {
min: float32Conversion(39.3),
max: float32Conversion(39.3),
nan: 0,
ninf: 0,
pinf: 0
}
},
nameInt32: {
categorical: false,
@@ -124,36 +145,27 @@ describe("summarizeAnnotations", () => {
});
test("multi test", () => {
const obsAnnotations = [
{
__index__: 0,
name: "n0",
nameString: "hi",
nameBoolean: false,
nameFloat32: 39.3,
nameInt32: 99,
nameCategorical: 1
},
{
__index__: 1,
name: "n1",
nameString: "hi",
nameBoolean: true,
nameFloat32: 39.3,
nameInt32: 99,
nameCategorical: false
},
{
__index__: 2,
name: "n2",
nameString: "bye",
nameBoolean: true,
nameFloat32: 0,
nameInt32: 99,
nameCategorical: "0"
}
];
const varAnnotations = [];
const obsAnnotations = new Dataframe.Dataframe(
[3, 6],
[
["n0", "n1", "n2"],
["hi", "hi", "bye"],
[false, true, true],
new Float32Array([39.3, 39.3, 0]),
new Int32Array([99, 99, 99]),
[1, false, "0"]
],
null,
new Dataframe.KeyIndex([
"name",
"nameString",
"nameBoolean",
"nameFloat32",
"nameInt32",
"nameCategorical"
])
);
const varAnnotations = Dataframe.Dataframe.empty();
const summary = summarizeAnnotations(
schema,
@@ -178,7 +190,13 @@ describe("summarizeAnnotations", () => {
},
nameFloat32: {
categorical: false,
range: { min: 0, max: 39.3, nan: 0, ninf: 0, pinf: 0 }
range: {
min: 0,
max: float32Conversion(39.3),
nan: 0,
ninf: 0,
pinf: 0
}
},
nameInt32: {
categorical: false,
@@ -197,45 +215,32 @@ describe("summarizeAnnotations", () => {
});
test("non-finite numbers", () => {
const obsAnnotations = [
{
__index__: 0,
name: "n0",
nameString: "hi",
nameBoolean: false,
nameFloat32: 39.3,
nameInt32: 99,
nameCategorical: 1
},
{
__index__: 1,
name: "n1",
nameString: "hi",
nameBoolean: true,
nameFloat32: Number.NEGATIVE_INFINITY,
nameInt32: 99,
nameCategorical: false
},
{
__index__: 2,
name: "n2",
nameString: "bye",
nameBoolean: true,
nameFloat32: Number.NaN,
nameInt32: 99,
nameCategorical: "0"
},
{
__index__: 3,
name: "n2",
nameString: "bye",
nameBoolean: true,
nameFloat32: Number.POSITIVE_INFINITY,
nameInt32: 99,
nameCategorical: "0"
}
];
const varAnnotations = [];
const obsAnnotations = new Dataframe.Dataframe(
[4, 6],
[
["n0", "n1", "n2", "n2"],
["hi", "hi", "bye", "bye"],
[false, true, true, true],
new Float32Array([
39.3,
Number.NEGATIVE_INFINITY,
Number.NaN,
Number.POSITIVE_INFINITY
]),
new Int32Array([99, 99, 99, 99]),
[1, false, "0", "0"]
],
null,
new Dataframe.KeyIndex([
"name",
"nameString",
"nameBoolean",
"nameFloat32",
"nameInt32",
"nameCategorical"
])
);
const varAnnotations = Dataframe.Dataframe.empty();
const summary = summarizeAnnotations(
schema,
@@ -260,7 +265,13 @@ describe("summarizeAnnotations", () => {
},
nameFloat32: {
categorical: false,
range: { min: 39.3, max: 39.3, nan: 1, ninf: 1, pinf: 1 }
range: {
min: float32Conversion(39.3),
max: float32Conversion(39.3),
nan: 1,
ninf: 1,
pinf: 1
}
},
nameInt32: {
categorical: false,
@@ -1,13 +1,13 @@
import _ from "lodash";
import * as Universe from "../../../src/util/stateManager/universe";
import * as Dataframe from "../../../src/util/dataframe";
import * as REST from "./sampleResponses";
describe("createUniverseFromRestV02Response", () => {
describe("createUniverseFromResponse", () => {
/*
test createUniverseFromRestV02Response - this function converts
test createUniverseFromResponse - this function converts
a set of REST 0.2 responses into a "new" Universe.
createUniverseFromRestV02Response(
createUniverseFromResponse(
configResponse,
schemaResponse,
annotationsObsResponse,
@@ -30,7 +30,7 @@ describe("createUniverseFromRestV02Response", () => {
create a universe from sample data nad validate its shape & contents
*/
const { nObs, nVar } = REST.schema.schema.dataframe;
const universe = Universe.createUniverseFromRestV02Response(
const universe = Universe.createUniverseFromResponse(
REST.config,
REST.schema,
REST.annotationsObs,
@@ -45,23 +45,23 @@ describe("createUniverseFromRestV02Response", () => {
nObs,
nVar,
schema: REST.schema.schema,
obsAnnotations: expect.any(Array),
varAnnotations: expect.any(Array),
obsNameToIndexMap: expect.any(Object),
varNameToIndexMap: expect.any(Object),
obsLayout: expect.objectContaining({
X: expect.any(Float32Array),
Y: expect.any(Float32Array)
}),
obsAnnotations: expect.any(Dataframe.Dataframe),
varAnnotations: expect.any(Dataframe.Dataframe),
obsLayout: expect.any(Dataframe.Dataframe),
summary: expect.any(Object),
varDataCache: expect.any(Object)
})
);
expect(universe.obsAnnotations).toHaveLength(nObs);
expect(_.keys(universe.obsNameToIndexMap)).toHaveLength(nObs);
expect(universe.obsLayout.X).toHaveLength(nObs);
expect(universe.obsLayout.Y).toHaveLength(nObs);
expect(universe.varAnnotations).toHaveLength(nVar);
expect(_.keys(universe.varNameToIndexMap)).toHaveLength(nVar);
expect(universe.obsAnnotations.dims).toEqual([
nObs,
REST.schema.schema.annotations.obs.length
]);
expect(universe.obsLayout.dims).toEqual([nObs, 2]);
expect(universe.obsLayout.colIndex.keys()).toEqual(["X", "Y"]);
expect(universe.varAnnotations.dims).toEqual([
nVar,
REST.schema.schema.annotations.var.length
]);
});
});
@@ -1,6 +1,7 @@
import _ from "lodash";
import * as Universe from "../../../src/util/stateManager/universe";
import * as World from "../../../src/util/stateManager/world";
import * as Dataframe from "../../../src/util/dataframe";
import Crossfilter from "../../../src/util/typedCrossfilter";
import * as REST from "./sampleResponses";
import {
@@ -16,7 +17,7 @@ the default REST test response.
const defaultBigBang = () => {
/* create unverse, world, crossfilter and dimensionMap */
/* create universe */
const universe = Universe.createUniverseFromRestV02Response(
const universe = Universe.createUniverseFromResponse(
REST.config,
REST.schema,
REST.annotationsObs,
@@ -40,7 +41,7 @@ const defaultBigBang = () => {
describe("createWorldFromEntireUniverse", () => {
test("create from REST sample", () => {
const universe = Universe.createUniverseFromRestV02Response(
const universe = Universe.createUniverseFromResponse(
REST.config,
REST.schema,
REST.annotationsObs,
@@ -75,10 +76,7 @@ describe("createWorldFromEntireUniverse", () => {
.value()
}),
varDataCache: expect.any(Object),
obsIndex: null, // null indicating full universe
obsBackIndex: null
varDataCache: expect.any(Object)
})
);
});
@@ -111,51 +109,43 @@ describe("createWorldFromCurrentSelection", () => {
*/
/* matchFilter must match the dimension filters above */
const matchFilter = val => val.field1 >= 0 && val.field1 < 5 && !val.field3;
const universeIndices = _()
.range(universe.nObs)
.filter(idx => matchFilter(universe.obsAnnotations[idx]))
.value();
const expected = {
nObs: universeIndices.length,
obsAnnotations: _.map(universeIndices, i => universe.obsAnnotations[i]),
obsLayout: {
X: new Float32Array(
_.map(universeIndices, i => universe.obsLayout.X[i])
),
Y: new Float32Array(
_.map(universeIndices, i => universe.obsLayout.Y[i])
)
},
obsBackIndex: _.transform(
universeIndices,
(result, univIdx, worldIdx) => {
result[univIdx] = worldIdx;
},
new Uint32Array(universe.nObs).fill(-1)
),
obsIndex: new Uint32Array(universeIndices)
const matchFilter = (df, row) => {
const field1 = df.at(row, "field1");
const field3 = df.at(row, "field3");
return field1 >= 0 && field1 < 5 && !field3;
};
const matchingIndices = _()
.range(universe.nObs)
.filter(idx => matchFilter(universe.obsAnnotations, idx))
.value();
expect(world).toMatchObject(
expect.objectContaining({
api: "0.2",
nObs: expected.nObs,
nObs: matchingIndices.length,
nVar: universe.nVar,
schema: universe.schema,
obsAnnotations: expected.obsAnnotations,
obsAnnotations: expect.any(Dataframe.Dataframe),
varAnnotations: universe.varAnnotations,
obsLayout: expected.obsLayout,
obsLayout: expect.any(Dataframe.Dataframe),
summary: {
obs: expect.any(Object) /* we could do better! */,
var: expect.any(Object) /* we could do better! */
},
varDataCache: expect.any(Object),
obsIndex: expected.obsIndex,
obsBackIndex: expected.obsBackIndex
varDataCache: expect.any(Object)
})
);
expect(world.obsAnnotations.rowIndex.keys()).toEqual(
new Int32Array(matchingIndices)
);
expect(world.obsAnnotations.colIndex.keys()).toEqual(
universe.obsAnnotations.colIndex.keys()
);
expect(world.obsLayout.rowIndex.keys()).toEqual(
new Int32Array(matchingIndices)
);
expect(world.obsLayout.colIndex.keys()).toEqual(["X", "Y"]);
});
});
@@ -219,7 +209,9 @@ describe("subsetVarData", () => {
world,
crossfilter
);
expect(newWorld.obsIndex).toMatchObject(new Uint32Array([0, 2]));
expect(newWorld.obsAnnotations.rowIndex.keys()).toEqual(
new Int32Array([0, 2])
);
/* expect a subset */
const result = World.subsetVarData(newWorld, universe, sourceVarData);
@@ -2,16 +2,27 @@ import {
countCategoryValues2D,
clearCaches
} from "../../../src/util/stateManager/worldUtil";
import * as Dataframe from "../../../src/util/dataframe";
describe("WorldUtil cache management", () => {
test("empty", () => {
const count = countCategoryValues2D("a", "b", []);
const count = countCategoryValues2D(
"a",
"b",
new Dataframe.Dataframe([0, 0], [])
);
expect(count).toMatchObject(new Map());
expect(count.size).toBe(0);
});
test("simple couts", () => {
const rows = [{ a: 0, b: false }, { a: 0, b: true }, { a: 1, b: false }];
const count = countCategoryValues2D("a", "b", rows);
const df = new Dataframe.Dataframe(
[3, 2],
[[0, 0, 1], [false, true, false]],
null,
new Dataframe.KeyIndex(["a", "b"])
);
const count = countCategoryValues2D("a", "b", df);
expect(count).toMatchObject(
new Map([
[0, new Map([[true, 1], [false, 1]])],
@@ -22,16 +33,22 @@ describe("WorldUtil cache management", () => {
test("memo cache clear", () => {
clearCaches();
const row1 = [];
const row2 = [{ a: 0, b: false }, { a: 0, b: true }, { a: 1, b: false }];
const count1 = countCategoryValues2D("a", "b", row1);
const count2 = countCategoryValues2D("a", "b", row1);
const count3 = countCategoryValues2D("a", "b", []);
const count4 = countCategoryValues2D("a", "b", row2);
const df1 = new Dataframe.Dataframe([0, 0], []);
const df2 = new Dataframe.Dataframe(
[3, 2],
[[0, 0, 1], [false, true, false]],
null,
new Dataframe.KeyIndex(["a", "b"])
);
const count1 = countCategoryValues2D("a", "b", df1);
const count2 = countCategoryValues2D("a", "b", df1);
const count3 = countCategoryValues2D("a", "b", df1.clone());
const count4 = countCategoryValues2D("a", "b", df2);
clearCaches();
const count10 = countCategoryValues2D("a", "b", row1);
const count11 = countCategoryValues2D("a", "b", row2);
const count10 = countCategoryValues2D("a", "b", df1);
const count11 = countCategoryValues2D("a", "b", df2);
expect(count1).toEqual(count2);
expect(count1).toEqual(count3);