Experimental - manual annotations (#837)

* icons, partway

* redux for values

* onChange

* cancel

* annotations lifecycle for category names

* copy categorical

* edit category

* add Dataframe.withColsFrom

* render user annotations; default add/delete annotation category

* add label name to actions

* category name edit

* error checking improvements

* change schema field isUserAnnotation to writable

* always have an unassigned label; implement delete label

* implement add new label and edit label name

* label current cell selection

* fix select exact bug in crossfilter

* clean up categorical reducer

* fix tests

* remove debugging printf

* implement subset/reset for user annotations

* undo redo support for user annotations

* remove duplicate button from categories

* add modal

* remove obsolete duplicate annotation reducers

* remove old debugging printf

* connect modal to annotation create and dup

* initial full-stack wiring

* finish up end-to-end wiring

* fix existing unit tests

* fix pytests to match new schema API

* remove debugging printfs

* add label file rotation

* remove obsolete comment

* add fbs encode/decode tests

* add tests for writable annotations

* simplify code

* fix hashing bug with FBS encoding

* lint

* fix smoke tests

* improve error checking in Dataframe.withColsFrom

* add unit test for Dataframe.withColsFrom

* add unit test for Dataframe.columns and Dataframe.renameCol

* fix bug in FBS encode, add better error checks, refactor

* add FBS encode/decode test

* add clarifying comment

* clean up action type names; fix state inconsistency in crossfilter update

* change autosave timer to 2.5sec

* sort categorical metadata render order so it remains consistent

* add temporary autogenerated label for add-new-label operation

* fix hover-over label menu interference with cell highlighting

* remove debugging code

* add missing reducer cases & fix typo

* make dataframe memoize more general purpose

* add dev mode for annos

* fix error on select duplicate

* handle zero occupancy categories

* correctly maintain unclipped AND clipped world

* correctly handle zero length FBS matrix and label files

* ensure all writable categorical schema contains an unassigned category

* handle case where building occupancy stack for category with no members

* dialog for creating label, disable button if duplicate or empty

* visually separate writeable

* edit category

* fix edit category name

* remove debugging code

* fix edit annotation label

* visually define unassigned, change options

* Pull in requirements.txt from `master`

* label currently selected cells

* duplicate label

* lint

* fix pytest merge issues

* rename --label-file to --experimental-label-file

* remove debugging console log

* spelling error fix; fix bug found in PR review.

* lint
This commit is contained in:
Bruce Martin
2019-09-18 07:33:41 -04:00
committed by Colin Megill
parent ab2c423006
commit 3660a6cc27
51 changed files with 2823 additions and 337 deletions
+89
View File
@@ -345,12 +345,50 @@ class Dataframe {
);
}
withColsFrom(dataframe) {
/*
return a new dataframe containing all columns from both `this` and the
provided dataframe.
The row index from `this` will be used. Both dataframes must have identical
dimensionality, and no overlapping columns labels.
*/
const dims = [this.dims[0], this.dims[1] + dataframe.dims[1]];
const { rowIndex } = this;
const columns = [...this.__columns, ...dataframe.__columns];
const colIndex = this.colIndex.withLabels(dataframe.colIndex.keys());
const columnsAccessor = [
...this.__columnsAccessor,
...dataframe.__columnsAccessor
];
return new this.constructor(
dims,
columns,
rowIndex,
colIndex,
columnsAccessor
);
}
dropCol(label) {
/*
Create a new dataframe, omitting one columns.
const newDf = df.dropCol("colors");
Corner case to manage: if dropping the last column, return an empty dataframe.
*/
if (!this.hasCol(label)) {
throw new RangeError(`unknown label: ${label}`);
}
/*
Corner case to manage: if dropping the last column, return an empty dataframe.
*/
if (this.dims[1] === 1) {
return Dataframe.empty();
}
const dims = [this.dims[0], this.dims[1] - 1];
const coffset = this.colIndex.getOffset(label);
const columns = [...this.__columns];
@@ -367,6 +405,50 @@ class Dataframe {
);
}
renameCol(oldLabel, newLabel) {
/*
Accelerator for dropping a column and then adding it again with a new label
*/
const coffset = this.colIndex.getOffset(oldLabel);
const colIndex = this.colIndex.dropLabel(oldLabel).withLabel(newLabel);
const columns = [...this.__columns];
columns.push(columns[coffset]);
columns.splice(coffset, 1);
const columnsAccessor = [...this.__columnsAccessor];
columnsAccessor.push(columnsAccessor[coffset]);
columnsAccessor.splice(coffset, 1);
return new this.constructor(
this.dims,
columns,
this.rowIndex,
colIndex,
columnsAccessor
);
}
replaceColData(label, newColData) {
/*
Accelerator for dropping a column then adding it again with same
label and different values.
*/
const coffset = this.colIndex.getOffset(label);
const columns = [...this.__columns];
columns[coffset] = newColData;
const columnsAccessor = [...this.__columnsAccessor];
columnsAccessor[coffset] = null;
return new this.constructor(
this.dims,
columns,
this.rowIndex,
this.colIndex,
columnsAccessor
);
}
static empty(rowIndex = null, colIndex = null) {
return new Dataframe([0, 0], [], rowIndex, colIndex);
}
@@ -443,6 +525,8 @@ class Dataframe {
return newCol;
});
}
if (dims[0] === 0 || dims[1] === 0) return Dataframe.empty();
return new Dataframe(dims, columns, rowIndex, colIndex);
}
@@ -526,6 +610,11 @@ class Dataframe {
Data access with row/col.
**/
columns() {
/* return all column accessors as an array, in offset order */
return [...this.__columnsAccessor];
}
col(columnLabel) {
/*
Return accessor bound to a column. Allows random row access
+17
View File
@@ -80,6 +80,10 @@ class IdentityInt32Index {
return this.__promote([...this.keys(), label]);
}
withLabels(labels) {
return this.__promote([...this.keys(), ...labels]);
}
dropLabel(label) {
if (label === this.maxOffset - 1) {
return new IdentityInt32Index(label);
@@ -163,6 +167,10 @@ class DenseInt32Index {
return this.__promote([...this.keys(), label]);
}
withLabels(labels) {
return this.__promote([...this.keys(), ...labels]);
}
dropLabel(label) {
const labelArray = [...this.keys()];
labelArray.splice(labelArray.indexOf(label), 1);
@@ -187,6 +195,11 @@ class KeyIndex {
index.set(v, i);
});
if (index.size !== rindex.length) {
/* if true, there was a duplicate in the keys */
throw new Error("duplicate label provided to KeyIndex");
}
this.index = index;
this.rindex = rindex;
this.__compile();
@@ -218,6 +231,10 @@ class KeyIndex {
return new KeyIndex([...this.rindex, label]);
}
withLabels(labels) {
return new KeyIndex([...this.rindex, ...labels]);
}
dropLabel(label) {
const idx = this.rindex.indexOf(label);
const labelArray = [...this.rindex];
+13 -1
View File
@@ -21,7 +21,7 @@ export function callOnceLazy(f) {
return result;
}
export function memoize(fn, hashFn) {
export function memoize(fn, hashFn, maxResultsCached = -1) {
/*
function memoization, with user-provided hash. hashFn must return a
key which will be unique as a Map key (ie, obeys "sameValueZero" algorithm
@@ -36,7 +36,19 @@ export function memoize(fn, hashFn) {
}
const result = fn(...args);
cache.set(key, result);
if (maxResultsCached > -1 && cache.size > maxResultsCached) {
/* Least recent insertion deletion */
cache.delete(cache.keys().next().value);
}
return result;
};
wrap.clear = function clear() {
/* clear memoization cache */
cache.clear();
};
return wrap;
}
@@ -0,0 +1,168 @@
/*
Helper functions for user-editable nnotations state management.
See also reducers/annotations.js
*/
import { unassignedCategoryLabel } from "../../globals";
import * as SchemaHelpers from "./schemaHelpers";
import { obsAnnoDimensionName } from "../nameCreators";
/*
There are a number of state constraints assumed throughout the
application:
- all obs annotations are in {world|universe}.obsAnnotations,
regardless of whether or not they are user editable.
- the {world|universe}.schema is always up to date and matches
the data
- the schema flag `writable` correctly indicates whether
the annotation is editable/mutable.
In addition, the current state management only allows for
categorical annotations to be writable.
*/
export function isCategoricalAnnotation(schema, name) {
/* we treat any string, categorical or boolean as a categorical */
const { type } = schema.annotations.obsByName[name];
return type === "string" || type === "boolean" || type === "categorical";
}
export function isContinuousAnnotation(schema, name) {
return !isCategoricalAnnotation(schema, name);
}
function _isUserAnnotation(schema, name) {
return schema.annotations.obsByName[name]?.writable;
}
export function isUserAnnotation(worldOrUniverse, name) {
return _isUserAnnotation(worldOrUniverse.schema, name);
}
export function removeObsAnnoSchema(schema, name) {
/*
remove named annotation from obs annotation schema
*/
/* only remove if it exists and is a user annotation */
if (!_isUserAnnotation(schema, name))
throw new Error("removing non-user-defined schema");
return SchemaHelpers.removeObsAnnoColumn(schema, name);
}
export function addObsAnnoSchema(schema, name, colSchema) {
/*
add a categorical type to the obs annotation schema
*/
/* collision detection */
if (schema.annotations.obs.columns.some(v => v.name === name))
throw Error("annotations may not contain duplicate category names");
if (name !== colSchema.name) throw Error("column schema does not match");
return SchemaHelpers.addObsAnnoColumn(schema, name, colSchema);
}
export function dupObsAnnoSchema(schema, sourceName, dupName, defaultSchema) {
/*
duplicate the obs annotation `sourceName` schema, but with the name `dupName`
*/
const colSchema = {
...schema.annotations.obsByName[sourceName],
...defaultSchema,
name: dupName
};
/* existance check */
if (!colSchema) throw Error("source annotation does not exist");
/* collision detection */
if (schema.annotations.obs.columns.some(v => v.name === dupName))
throw Error("annotations may not contain duplicate category names");
return SchemaHelpers.addObsAnnoColumn(schema, dupName, colSchema);
}
export function removeObsAnnoCategory(schema, name, category) {
/* don't allow deletion of unassigned category on writable annotations */
if (!_isUserAnnotation(schema, name))
throw new Error("unable to modify read-only schema");
if (category === unassignedCategoryLabel)
throw new Error("may not remove unassigned category label");
return SchemaHelpers.removeObsAnnoCategory(schema, name, category);
}
export function addObsAnnoCategory(schema, name, category) {
if (!_isUserAnnotation(schema, name))
throw new Error("unable to modify read-only schema");
return SchemaHelpers.addObsAnnoCategory(schema, name, category);
}
export function setLabelByValue(df, colName, fromLabel, toLabel) {
/*
in the dataframe column `colName`, set any value of `fromLabel` to `toLabel`
*/
const keys = df.colIndex.keys();
const ndf = df.mapColumns((col, colIdx) => {
if (colName !== keys[colIdx]) return col;
/* clone data and return it. */
const newCol = col.slice();
for (let i = 0, l = newCol.length; i < l; i += 1) {
if (newCol[i] === fromLabel) newCol[i] = toLabel;
}
return newCol;
});
return ndf;
}
export function setLabelByMask(df, colName, mask, label) {
/*
in the dataframe column `colName`, set the masked rows to 'label'
*/
const keys = df.colIndex.keys();
const ndf = df.mapColumns((col, colIdx) => {
if (colName !== keys[colIdx]) return col;
/* clone data and return it. */
const newCol = col.slice();
for (let i = 0, l = newCol.length; i < l; i += 1) {
if (mask[i]) newCol[i] = label;
}
return newCol;
});
return ndf;
}
export function worldToUniverseMask(worldMask, worldObsAnnotations, nObs) {
/*
given world seleciton mask, return a selection mask for entire universe
that has same selection state.
*/
const mask = new Uint8Array(nObs);
const { rowIndex } = worldObsAnnotations;
for (let i = 0, l = worldMask.length; i < l; i += 1) {
if (worldMask[i]) {
const label = rowIndex.getLabel(i);
mask[label] = 1;
}
}
return mask;
}
export function createWritableAnnotationDimensions(world, crossfilter) {
const { obsAnnotations, schema } = world;
const writableAnnotations = schema.annotations.obs.columns
.filter(s => s.writable)
.map(s => s.name);
crossfilter = writableAnnotations.reduce((xflt, anno) => {
const dimName = obsAnnoDimensionName(anno);
if (xflt.hasDimension(dimName)) xflt = xflt.delDimension(dimName);
return xflt.addDimension(
dimName,
"enum",
obsAnnotations.col(anno).asArray()
);
}, crossfilter);
return crossfilter;
}
+3 -2
View File
@@ -50,8 +50,9 @@ function createColorsByCategoricalMetadata(world, accessor) {
}, {});
const rgb = new Array(world.nObs);
const data = world.obsAnnotations.col(accessor).asArray();
for (let i = 0, len = world.obsAnnotations.length; i < len; i += 1) {
const df = world.obsAnnotations;
const data = df.col(accessor).asArray();
for (let i = 0, len = df.length; i < len; i += 1) {
const cat = data[i];
rgb[i] = colors[cat];
}
+53 -63
View File
@@ -37,16 +37,14 @@ Remember that option values can be ANY js type, except undefined/null.
}
}
*/
function topNCategories(summary) {
const counts = _.map(summary.categories, cat =>
summary.categoryCounts.get(cat)
);
const sortIndex = fillRange(new Array(summary.numCategories)).sort(
function topNCategories(colSchema, summary, N) {
const { categories } = colSchema;
const counts = _.map(categories, cat => summary.categoryCounts.get(cat) ?? 0);
const sortIndex = fillRange(new Array(categories.length)).sort(
(a, b) => counts[b] - counts[a]
);
const sortedCategories = _.map(sortIndex, i => summary.categories[i]);
const sortedCategories = _.map(sortIndex, i => categories[i]);
const sortedCounts = _.map(sortIndex, i => counts[i]);
const N = globals.maxCategoricalOptionsToDisplay;
if (sortedCategories.length < N) {
return [sortedCategories, sortedCounts];
@@ -54,64 +52,56 @@ function topNCategories(summary) {
return [sortedCategories.slice(0, N), sortedCounts.slice(0, N)];
}
export function createCategoricalSelection(maxCategoryItems, world) {
const res = {};
const obsIndexName = world.schema.annotations.obs.index;
_.forEach(world.obsAnnotations.colIndex.keys(), key => {
const summary = world.obsAnnotations.col(key).summarize();
if (summary.categories) {
const isColorField = key.includes("color") || key.includes("Color");
const isSelectableCategory =
!isColorField &&
key !== obsIndexName &&
summary.categories.length < maxCategoryItems;
if (isSelectableCategory) {
const [categoryValues, categoryValueCounts] = topNCategories(summary);
const categoryValueIndices = new Map(
categoryValues.map((v, i) => [v, i])
);
const numCategoryValues = categoryValueIndices.size;
const categoryValueSelected = new Array(numCategoryValues).fill(true);
const isTruncated = categoryValues.length < summary.numCategories;
res[key] = {
categoryValues, // array: of natively typed category values
categoryValueIndices, // map: category value (native type) -> category index
categoryValueSelected, // array: t/f selection state
numCategoryValues, // number: of values in the category
isTruncated, // bool: true if list was truncated
categoryValueCounts, // array: cardinality of each category,
categorySelected: true // bool - default state for entire category
};
}
}
});
return res;
export function selectableCategoryNames(world, maxCategoryItems) {
const { schema } = world;
const { index, columns } = schema.annotations.obs;
return columns
.filter(colSchema => {
const { name, categories } = colSchema;
return (
categories && categories.length < maxCategoryItems && name !== index
);
})
.map(v => v.name);
}
/*
given a categoricalSelection, return the list of all category values
where selection state is true (ie, they are selected).
*/
export function selectedValuesForCategory(categorySelectionState, dfColumn) {
const {
categorySelected,
categoryValueSelected,
categoryValueIndices
} = categorySelectionState;
let selectedValues;
if (categorySelected) {
selectedValues = new Set(dfColumn.summarize().categories);
} else {
selectedValues = new Set();
}
categoryValueIndices.forEach((catIndex, catValue) => {
if (!categoryValueSelected[catIndex]) {
selectedValues.delete(catValue);
} else {
selectedValues.add(catValue);
}
});
return [...selectedValues.values()];
export function createCategoricalSelection(world, names) {
const N = globals.maxCategoricalOptionsToDisplay;
const { obsAnnotations, schema } = world;
const res = names.reduce((acc, name) => {
const colSchema = schema.annotations.obsByName[name];
const { writable: isUserAnno } = colSchema;
/*
Summarize the annotation data currently in world. Must return categoryValues
in sorted order, and must include all category values even if they are not
actively used in the current world.
*/
const summary = obsAnnotations.col(name).summarize();
const [categoryValues, categoryValueCounts] = topNCategories(
colSchema,
summary,
N
);
const categoryValueIndices = new Map(categoryValues.map((v, i) => [v, i]));
const numCategoryValues = categoryValueIndices.size;
const categoryValueSelected = new Array(numCategoryValues).fill(true);
const isTruncated = categoryValues.length < summary.numCategories;
acc[name] = {
categoryValues, // array: of natively typed category values
categoryValueIndices, // map: category value (native type) -> category index
categoryValueSelected, // array: t/f selection state
numCategoryValues, // number: of values in the category
isTruncated, // bool: true if list was truncated
categoryValueCounts, // array: cardinality of each category,
categorySelected: true, // bool - default state for entire category
isUserAnno // bool
};
return acc;
}, {});
return res;
}
/*
+3
View File
@@ -19,3 +19,6 @@ export * as Universe from "./universe";
export * as World from "./world";
export * as WorldUtil from "./worldUtil";
export * as ControlsHelpers from "./controlsHelpers";
export * as AnnotationsHelpers from "./annotationsHelpers";
export * as SchemaHelpers from "./schemaHelpers";
export * as MatrixFBS from "./matrix";
+98 -9
View File
@@ -1,5 +1,7 @@
import { flatbuffers } from "flatbuffers";
import { NetEncoding } from "./matrix_generated";
import { isTypedArray } from "../typeHelpers";
import { IdentityInt32Index, DenseInt32Index, KeyIndex } from "../dataframe";
const utf8Decoder = new TextDecoder("utf-8");
@@ -41,25 +43,25 @@ Returns: object containing decoded Matrix:
colIdx: []|null
}
*/
function decodeMatrixFBS(arrayBuffer, inplace = false) {
export function decodeMatrixFBS(arrayBuffer, inplace = false) {
const bb = new flatbuffers.ByteBuffer(new Uint8Array(arrayBuffer));
const df = NetEncoding.Matrix.getRootAsMatrix(bb);
const matrix = NetEncoding.Matrix.getRootAsMatrix(bb);
const nRows = df.nRows();
const nCols = df.nCols();
const nRows = matrix.nRows();
const nCols = matrix.nCols();
/* decode columns */
const columnsLength = df.columnsLength();
const columnsLength = matrix.columnsLength();
const columns = Array(columnsLength).fill(null);
for (let c = 0; c < columnsLength; c += 1) {
const col = df.columns(c);
const col = matrix.columns(c);
columns[c] = decodeTypedArray(col.uType(), col.u.bind(col), inplace);
}
/* decode col_idx */
const colIdx = decodeTypedArray(
df.colIndexType(),
df.colIndex.bind(df),
matrix.colIndexType(),
matrix.colIndex.bind(matrix),
inplace
);
@@ -72,4 +74,91 @@ function decodeMatrixFBS(arrayBuffer, inplace = false) {
};
}
export default decodeMatrixFBS;
function encodeTypedArray(builder, uType, uData) {
const uTypeName = NetEncoding.TypedArray[uType];
const ArrayType = NetEncoding[uTypeName];
const dv = ArrayType.createDataVector(builder, uData);
builder.startObject(1);
builder.addFieldOffset(0, dv, 0);
return builder.endObject();
}
export function encodeMatrixFBS(df) {
/*
encode the dataframe as an FBS Matrix
*/
/* row indexing not supported currently */
if (df.rowIndex.constructor !== IdentityInt32Index) {
throw new Error("FBS does not support row index encoding at this time");
}
const shape = df.dims;
const utf8Encoder = new TextEncoder("utf-8");
const builder = new flatbuffers.Builder(1024);
let encColIndex;
let encColIndexUType;
let encColumns;
if (shape[0] > 0 && shape[1] > 0) {
const columns = df.columns().map(col => col.asArray());
const cols = columns.map(carr => {
let uType;
let tarr;
if (isTypedArray(carr)) {
uType = NetEncoding.TypedArray[carr.constructor.name];
tarr = encodeTypedArray(builder, uType, carr);
} else {
uType = NetEncoding.TypedArray.JSONEncodedArray;
const json = JSON.stringify(carr);
const jsonUTF8 = utf8Encoder.encode(json);
tarr = encodeTypedArray(builder, uType, jsonUTF8);
}
NetEncoding.Column.startColumn(builder);
NetEncoding.Column.addUType(builder, uType);
NetEncoding.Column.addU(builder, tarr);
return NetEncoding.Column.endColumn(builder);
});
encColumns = NetEncoding.Matrix.createColumnsVector(builder, cols);
if (df.colIndex && shape[1] > 0) {
const colIndexType = df.colIndex.constructor;
if (colIndexType === IdentityInt32Index) {
encColIndex = undefined;
} else if (colIndexType === DenseInt32Index) {
encColIndexUType = NetEncoding.TypedArray.Int32Array;
encColIndex = encodeTypedArray(
builder,
encColIndexUType,
df.colIndex.keys()
);
} else if (colIndexType === KeyIndex) {
encColIndexUType = NetEncoding.TypedArray.JSONEncodedArray;
encColIndex = encodeTypedArray(
builder,
encColIndexUType,
utf8Encoder.encode(JSON.stringify(df.colIndex.keys()))
);
} else {
throw new Error("Index type FBS encoding unsupported");
}
}
}
NetEncoding.Matrix.startMatrix(builder);
NetEncoding.Matrix.addNRows(builder, shape[0]);
NetEncoding.Matrix.addNCols(builder, shape[1]);
if (encColumns) {
NetEncoding.Matrix.addColumns(builder, encColumns);
}
if (encColIndexUType) {
NetEncoding.Matrix.addColIndexType(builder, encColIndexUType);
NetEncoding.Matrix.addColIndex(builder, encColIndex);
}
const root = NetEncoding.Matrix.endMatrix(builder);
builder.finish(root);
return builder.asUint8Array();
}
@@ -0,0 +1,96 @@
/*
Helpers for schema management
*/
import _ from "lodash";
import fromEntries from "../fromEntries";
/*
System wide schema assumptions:
- schema and data wil be consistent (eg, for user-created annotations)
- schema will be internally self-consistent (eg, index matches columns)
- world & universe schema are same - only data is subset
*/
export function indexEntireSchema(schema) {
/* Index schema for ease of use */
schema.annotations.obsByName = fromEntries(
schema.annotations.obs.columns.map(v => [v.name, v])
);
schema.annotations.varByName = fromEntries(
schema.annotations.var.columns.map(v => [v.name, v])
);
schema.layout.obsByName = fromEntries(
schema.layout.obs.map(v => [v.name, v])
);
schema.layout.varByName = fromEntries(
schema.layout.var.map(v => [v.name, v])
);
return schema;
}
function _copy(schema) {
/* redux copy conventions - WARNING, only for modifyign obs annotations */
return {
...schema,
annotations: {
...schema.annotations,
obs: _.cloneDeep(schema.annotations.obs)
}
};
}
function _reindex(schema) {
/* reindex obs annotations ONLY */
schema.annotations.obsByName = fromEntries(
schema.annotations.obs.columns.map(v => [v.name, v])
);
return schema;
}
export function removeObsAnnoColumn(schema, name) {
const newSchema = _copy(schema);
newSchema.annotations.obs.columns = schema.annotations.obs.columns.filter(
v => v.name !== name
);
return _reindex(newSchema);
}
export function addObsAnnoColumn(schema, name, defn) {
const newSchema = _copy(schema);
newSchema.annotations.obs.columns.push(defn);
return _reindex(newSchema);
}
export function removeObsAnnoCategory(schema, name, category) {
/* remove a category from a categorical annotation */
const categories = schema.annotations.obsByName[name]?.categories;
if (!categories)
throw new Error("column does not exist or is not categorical");
const idx = categories.indexOf(category);
if (idx === -1) throw new Error("category does not exist");
const newSchema = _reindex(_copy(schema));
/* remove category */
newSchema.annotations.obsByName[name].categories.splice(idx, 1);
return newSchema;
}
export function addObsAnnoCategory(schema, name, category) {
/* add a category to a categorical annotation */
const categories = schema.annotations.obsByName[name]?.categories;
if (!categories)
throw new Error("column does not exist or is not categorical");
const idx = categories.indexOf(category);
if (idx !== -1) throw new Error("category already exists");
const newSchema = _reindex(_copy(schema));
/* remove category */
newSchema.annotations.obsByName[name].categories.push(category);
return newSchema;
}
+29 -19
View File
@@ -1,11 +1,11 @@
// jshint esversion: 6
import _ from "lodash";
import decodeMatrixFBS from "./matrix";
import { unassignedCategoryLabel } from "../../globals";
import { decodeMatrixFBS } from "./matrix";
import * as Dataframe from "../dataframe";
import fromEntries from "../fromEntries";
import { isFpTypedArray } from "../typeHelpers";
import { indexEntireSchema } from "./schemaHelpers";
import { isCategoricalAnnotation } from "./annotationsHelpers";
/*
Private helper function - create and return a template Universe
@@ -18,10 +18,13 @@ function templateUniverse() {
schema: {},
/*
Annotations
annotations
*/
obsAnnotations: Dataframe.Dataframe.empty(),
varAnnotations: Dataframe.Dataframe.empty(),
/*
layout
*/
obsLayout: Dataframe.Dataframe.empty(),
/*
@@ -122,6 +125,10 @@ function reconcileSchemaCategoriesWithSummary(universe) {
For example, boolean defined fields in the schema do not contain
explicit declaration of categories (nor do string fields). In these
cases, add a 'categories' field to the schema so it is accessible.
In addition, we have a client-side convention (UI) that all writable
annotations must have an 'unassigned' category, even if it is not currently
in use.
*/
universe.schema.annotations.obs.columns.forEach(s => {
@@ -136,6 +143,10 @@ function reconcileSchemaCategoriesWithSummary(universe) {
);
s.categories = categories;
}
if (s.writable && s.categories.indexOf(unassignedCategoryLabel) === -1) {
s.categories = s.categories.concat(unassignedCategoryLabel);
}
});
}
@@ -166,7 +177,7 @@ export function createUniverseFromResponse(
/* layout */
universe.obsLayout = LayoutFBSToDataframe(layoutFBSResponse);
/* sanity check */
/* sanity checks */
if (
universe.nObs !== universe.obsLayout.length ||
universe.nObs !== universe.obsAnnotations.length ||
@@ -176,20 +187,19 @@ export function createUniverseFromResponse(
}
reconcileSchemaCategoriesWithSummary(universe);
indexEntireSchema(universe.schema);
/* sanity checks */
if (
schema.annotations.obs.columns.some(
s => s.writable && !isCategoricalAnnotation(schema, s.name)
)
) {
throw new Error(
"Writable continuous obs annotations are not supproted - failed to laod"
);
}
/* Index schema for ease of use */
universe.schema.annotations.obsByName = fromEntries(
universe.schema.annotations.obs.columns.map(v => [v.name, v])
);
universe.schema.annotations.varByName = fromEntries(
universe.schema.annotations.var.columns.map(v => [v.name, v])
);
universe.schema.layout.obsByName = fromEntries(
universe.schema.layout.obs.map(v => [v.name, v])
);
universe.schema.layout.varByName = fromEntries(
universe.schema.layout.var.map(v => [v.name, v])
);
return universe;
}
+4 -11
View File
@@ -1,14 +1,7 @@
// jshint esversion: 6
import clip from "../clip";
import {
layoutDimensionName,
obsAnnoDimensionName,
diffexpDimensionName,
userDefinedDimensionName
} from "../nameCreators";
import { layoutDimensionName, obsAnnoDimensionName } from "../nameCreators";
import * as Dataframe from "../dataframe";
import ImmutableTypedCrossfilter from "../typedCrossfilter/crossfilter";
import { isContinuousAnnotation } from "./annotationsHelpers";
/*
@@ -158,7 +151,7 @@ and world.varData.
function setClippedDataframes(world) {
const { schema } = world;
const isContinuousObsAnnotation = (df, idx, label) =>
deduceDimensionType(schema.annotations.obsByName[label], label) !== "enum";
isContinuousAnnotation(schema, label);
const obsQuantile = (label, q) =>
world.unclipped.obsAnnotations.col(label).summarize().percentiles[100 * q];
world.obsAnnotations = clipDataframe(
@@ -183,7 +176,7 @@ function setClippedDataframes(world) {
/*
Subset the current world based upon the current selection, maintaining any existing
clip. Returns new world. Parameters:
* unvierse
* universe
* world - the current world
* crossfilter - the selection state
*/
@@ -5,6 +5,7 @@ import BitArray from "./bitArray";
import {
sortArray,
lowerBound,
binarySearch,
lowerBoundIndirect,
upperBoundIndirect
} from "./sort";
@@ -56,6 +57,14 @@ export default class ImmutableTypedCrossfilter {
return this.data;
}
setData(data) {
return new ImmutableTypedCrossfilter(
data,
this.dimensions,
this.selectionCache
);
}
dimensionNames() {
/* return array of all dimensions (by name) */
return Object.keys(this.dimensions);
@@ -113,6 +122,23 @@ export default class ImmutableTypedCrossfilter {
return new ImmutableTypedCrossfilter(data, dimensions, selectionCache);
}
renameDimension(oldName, newName) {
/*
rename a dimension
*/
const { [oldName]: dim, ...dimensions } = this.dimensions;
const { data, selectionCache } = this;
dim.dim.rename(newName);
return new ImmutableTypedCrossfilter(
data,
{
...dimensions,
[newName]: dim
},
selectionCache
);
}
select(name, spec) {
/*
select on named dimension, as indicated by `spec`. Spec is an object
@@ -288,6 +314,10 @@ class _ImmutableBaseDimension {
this.name = name;
}
rename(name) {
this.name = name;
}
select(spec) {
const { mode } = spec;
if (mode === undefined) {
@@ -436,7 +466,7 @@ class ImmutableEnumDimension extends ImmutableScalarDimension {
const { values } = spec;
return super.selectExact({
mode: spec.mode,
values: values.map(v => lowerBound(enumIndex, v, 0, enumIndex.length))
values: values.map(v => binarySearch(enumIndex, v, 0, enumIndex.length))
});
}
+13
View File
@@ -413,3 +413,16 @@ export function upperBoundIndirect(valueArray, indexArray, value, first, last) {
}
return upperBoundNonFloatIndirect(valueArray, indexArray, value, first, last);
}
// Search for `value` in the sorted array `arr`, in the range [first, last).
// Return the first index where arr[index] == value, OR if value not present,
// return `last`
//
// The same semantics/behavior as:
// C++: binary_search()
//
export function binarySearch(valueArray, value, first, last) {
const index = lowerBound(valueArray, value, first, last);
if (index !== last && value === valueArray[index]) return index;
return last;
}