Experimental - manual annotations (#837)

* icons, partway

* redux for values

* onChange

* cancel

* annotations lifecycle for category names

* copy categorical

* edit category

* add Dataframe.withColsFrom

* render user annotations; default add/delete annotation category

* add label name to actions

* category name edit

* error checking improvements

* change schema field isUserAnnotation to writable

* always have an unassigned label; implement delete label

* implement add new label and edit label name

* label current cell selection

* fix select exact bug in crossfilter

* clean up categorical reducer

* fix tests

* remove debugging printf

* implement subset/reset for user annotations

* undo redo support for user annotations

* remove duplicate button from categories

* add modal

* remove obsolete duplicate annotation reducers

* remove old debugging printf

* connect modal to annotation create and dup

* initial full-stack wiring

* finish up end-to-end wiring

* fix existing unit tests

* fix pytests to match new schema API

* remove debugging printfs

* add label file rotation

* remove obsolete comment

* add fbs encode/decode tests

* add tests for writable annotations

* simplify code

* fix hashing bug with FBS encoding

* lint

* fix smoke tests

* improve error checking in Dataframe.withColsFrom

* add unit test for Dataframe.withColsFrom

* add unit test for Dataframe.columns and Dataframe.renameCol

* fix bug in FBS encode, add better error checks, refactor

* add FBS encode/decode test

* add clarifying comment

* clean up action type names; fix state inconsistency in crossfilter update

* change autosave timer to 2.5sec

* sort categorical metadata render order so it remains consistent

* add temporary autogenerated label for add-new-label operation

* fix hover-over label menu interference with cell highlighting

* remove debugging code

* add missing reducer cases & fix typo

* make dataframe memoize more general purpose

* add dev mode for annos

* fix error on select duplicate

* handle zero occupancy categories

* correctly maintain unclipped AND clipped world

* correctly handle zero length FBS matrix and label files

* ensure all writable categorical schema contains an unassigned category

* handle case where building occupancy stack for category with no members

* dialog for creating label, disable button if duplicate or empty

* visually separate writeable

* edit category

* fix edit category name

* remove debugging code

* fix edit annotation label

* visually define unassigned, change options

* Pull in requirements.txt from `master`

* label currently selected cells

* duplicate label

* lint

* fix pytest merge issues

* rename --label-file to --experimental-label-file

* remove debugging console log

* spelling error fix; fix bug found in PR review.

* lint
This commit is contained in:
Bruce Martin
2019-09-18 07:33:41 -04:00
committed by Colin Megill
parent ab2c423006
commit 3660a6cc27
51 changed files with 2823 additions and 337 deletions
@@ -0,0 +1,168 @@
/*
Helper functions for user-editable nnotations state management.
See also reducers/annotations.js
*/
import { unassignedCategoryLabel } from "../../globals";
import * as SchemaHelpers from "./schemaHelpers";
import { obsAnnoDimensionName } from "../nameCreators";
/*
There are a number of state constraints assumed throughout the
application:
- all obs annotations are in {world|universe}.obsAnnotations,
regardless of whether or not they are user editable.
- the {world|universe}.schema is always up to date and matches
the data
- the schema flag `writable` correctly indicates whether
the annotation is editable/mutable.
In addition, the current state management only allows for
categorical annotations to be writable.
*/
export function isCategoricalAnnotation(schema, name) {
/* we treat any string, categorical or boolean as a categorical */
const { type } = schema.annotations.obsByName[name];
return type === "string" || type === "boolean" || type === "categorical";
}
export function isContinuousAnnotation(schema, name) {
return !isCategoricalAnnotation(schema, name);
}
function _isUserAnnotation(schema, name) {
return schema.annotations.obsByName[name]?.writable;
}
export function isUserAnnotation(worldOrUniverse, name) {
return _isUserAnnotation(worldOrUniverse.schema, name);
}
export function removeObsAnnoSchema(schema, name) {
/*
remove named annotation from obs annotation schema
*/
/* only remove if it exists and is a user annotation */
if (!_isUserAnnotation(schema, name))
throw new Error("removing non-user-defined schema");
return SchemaHelpers.removeObsAnnoColumn(schema, name);
}
export function addObsAnnoSchema(schema, name, colSchema) {
/*
add a categorical type to the obs annotation schema
*/
/* collision detection */
if (schema.annotations.obs.columns.some(v => v.name === name))
throw Error("annotations may not contain duplicate category names");
if (name !== colSchema.name) throw Error("column schema does not match");
return SchemaHelpers.addObsAnnoColumn(schema, name, colSchema);
}
export function dupObsAnnoSchema(schema, sourceName, dupName, defaultSchema) {
/*
duplicate the obs annotation `sourceName` schema, but with the name `dupName`
*/
const colSchema = {
...schema.annotations.obsByName[sourceName],
...defaultSchema,
name: dupName
};
/* existance check */
if (!colSchema) throw Error("source annotation does not exist");
/* collision detection */
if (schema.annotations.obs.columns.some(v => v.name === dupName))
throw Error("annotations may not contain duplicate category names");
return SchemaHelpers.addObsAnnoColumn(schema, dupName, colSchema);
}
export function removeObsAnnoCategory(schema, name, category) {
/* don't allow deletion of unassigned category on writable annotations */
if (!_isUserAnnotation(schema, name))
throw new Error("unable to modify read-only schema");
if (category === unassignedCategoryLabel)
throw new Error("may not remove unassigned category label");
return SchemaHelpers.removeObsAnnoCategory(schema, name, category);
}
export function addObsAnnoCategory(schema, name, category) {
if (!_isUserAnnotation(schema, name))
throw new Error("unable to modify read-only schema");
return SchemaHelpers.addObsAnnoCategory(schema, name, category);
}
export function setLabelByValue(df, colName, fromLabel, toLabel) {
/*
in the dataframe column `colName`, set any value of `fromLabel` to `toLabel`
*/
const keys = df.colIndex.keys();
const ndf = df.mapColumns((col, colIdx) => {
if (colName !== keys[colIdx]) return col;
/* clone data and return it. */
const newCol = col.slice();
for (let i = 0, l = newCol.length; i < l; i += 1) {
if (newCol[i] === fromLabel) newCol[i] = toLabel;
}
return newCol;
});
return ndf;
}
export function setLabelByMask(df, colName, mask, label) {
/*
in the dataframe column `colName`, set the masked rows to 'label'
*/
const keys = df.colIndex.keys();
const ndf = df.mapColumns((col, colIdx) => {
if (colName !== keys[colIdx]) return col;
/* clone data and return it. */
const newCol = col.slice();
for (let i = 0, l = newCol.length; i < l; i += 1) {
if (mask[i]) newCol[i] = label;
}
return newCol;
});
return ndf;
}
export function worldToUniverseMask(worldMask, worldObsAnnotations, nObs) {
/*
given world seleciton mask, return a selection mask for entire universe
that has same selection state.
*/
const mask = new Uint8Array(nObs);
const { rowIndex } = worldObsAnnotations;
for (let i = 0, l = worldMask.length; i < l; i += 1) {
if (worldMask[i]) {
const label = rowIndex.getLabel(i);
mask[label] = 1;
}
}
return mask;
}
export function createWritableAnnotationDimensions(world, crossfilter) {
const { obsAnnotations, schema } = world;
const writableAnnotations = schema.annotations.obs.columns
.filter(s => s.writable)
.map(s => s.name);
crossfilter = writableAnnotations.reduce((xflt, anno) => {
const dimName = obsAnnoDimensionName(anno);
if (xflt.hasDimension(dimName)) xflt = xflt.delDimension(dimName);
return xflt.addDimension(
dimName,
"enum",
obsAnnotations.col(anno).asArray()
);
}, crossfilter);
return crossfilter;
}
+3 -2
View File
@@ -50,8 +50,9 @@ function createColorsByCategoricalMetadata(world, accessor) {
}, {});
const rgb = new Array(world.nObs);
const data = world.obsAnnotations.col(accessor).asArray();
for (let i = 0, len = world.obsAnnotations.length; i < len; i += 1) {
const df = world.obsAnnotations;
const data = df.col(accessor).asArray();
for (let i = 0, len = df.length; i < len; i += 1) {
const cat = data[i];
rgb[i] = colors[cat];
}
+53 -63
View File
@@ -37,16 +37,14 @@ Remember that option values can be ANY js type, except undefined/null.
}
}
*/
function topNCategories(summary) {
const counts = _.map(summary.categories, cat =>
summary.categoryCounts.get(cat)
);
const sortIndex = fillRange(new Array(summary.numCategories)).sort(
function topNCategories(colSchema, summary, N) {
const { categories } = colSchema;
const counts = _.map(categories, cat => summary.categoryCounts.get(cat) ?? 0);
const sortIndex = fillRange(new Array(categories.length)).sort(
(a, b) => counts[b] - counts[a]
);
const sortedCategories = _.map(sortIndex, i => summary.categories[i]);
const sortedCategories = _.map(sortIndex, i => categories[i]);
const sortedCounts = _.map(sortIndex, i => counts[i]);
const N = globals.maxCategoricalOptionsToDisplay;
if (sortedCategories.length < N) {
return [sortedCategories, sortedCounts];
@@ -54,64 +52,56 @@ function topNCategories(summary) {
return [sortedCategories.slice(0, N), sortedCounts.slice(0, N)];
}
export function createCategoricalSelection(maxCategoryItems, world) {
const res = {};
const obsIndexName = world.schema.annotations.obs.index;
_.forEach(world.obsAnnotations.colIndex.keys(), key => {
const summary = world.obsAnnotations.col(key).summarize();
if (summary.categories) {
const isColorField = key.includes("color") || key.includes("Color");
const isSelectableCategory =
!isColorField &&
key !== obsIndexName &&
summary.categories.length < maxCategoryItems;
if (isSelectableCategory) {
const [categoryValues, categoryValueCounts] = topNCategories(summary);
const categoryValueIndices = new Map(
categoryValues.map((v, i) => [v, i])
);
const numCategoryValues = categoryValueIndices.size;
const categoryValueSelected = new Array(numCategoryValues).fill(true);
const isTruncated = categoryValues.length < summary.numCategories;
res[key] = {
categoryValues, // array: of natively typed category values
categoryValueIndices, // map: category value (native type) -> category index
categoryValueSelected, // array: t/f selection state
numCategoryValues, // number: of values in the category
isTruncated, // bool: true if list was truncated
categoryValueCounts, // array: cardinality of each category,
categorySelected: true // bool - default state for entire category
};
}
}
});
return res;
export function selectableCategoryNames(world, maxCategoryItems) {
const { schema } = world;
const { index, columns } = schema.annotations.obs;
return columns
.filter(colSchema => {
const { name, categories } = colSchema;
return (
categories && categories.length < maxCategoryItems && name !== index
);
})
.map(v => v.name);
}
/*
given a categoricalSelection, return the list of all category values
where selection state is true (ie, they are selected).
*/
export function selectedValuesForCategory(categorySelectionState, dfColumn) {
const {
categorySelected,
categoryValueSelected,
categoryValueIndices
} = categorySelectionState;
let selectedValues;
if (categorySelected) {
selectedValues = new Set(dfColumn.summarize().categories);
} else {
selectedValues = new Set();
}
categoryValueIndices.forEach((catIndex, catValue) => {
if (!categoryValueSelected[catIndex]) {
selectedValues.delete(catValue);
} else {
selectedValues.add(catValue);
}
});
return [...selectedValues.values()];
export function createCategoricalSelection(world, names) {
const N = globals.maxCategoricalOptionsToDisplay;
const { obsAnnotations, schema } = world;
const res = names.reduce((acc, name) => {
const colSchema = schema.annotations.obsByName[name];
const { writable: isUserAnno } = colSchema;
/*
Summarize the annotation data currently in world. Must return categoryValues
in sorted order, and must include all category values even if they are not
actively used in the current world.
*/
const summary = obsAnnotations.col(name).summarize();
const [categoryValues, categoryValueCounts] = topNCategories(
colSchema,
summary,
N
);
const categoryValueIndices = new Map(categoryValues.map((v, i) => [v, i]));
const numCategoryValues = categoryValueIndices.size;
const categoryValueSelected = new Array(numCategoryValues).fill(true);
const isTruncated = categoryValues.length < summary.numCategories;
acc[name] = {
categoryValues, // array: of natively typed category values
categoryValueIndices, // map: category value (native type) -> category index
categoryValueSelected, // array: t/f selection state
numCategoryValues, // number: of values in the category
isTruncated, // bool: true if list was truncated
categoryValueCounts, // array: cardinality of each category,
categorySelected: true, // bool - default state for entire category
isUserAnno // bool
};
return acc;
}, {});
return res;
}
/*
+3
View File
@@ -19,3 +19,6 @@ export * as Universe from "./universe";
export * as World from "./world";
export * as WorldUtil from "./worldUtil";
export * as ControlsHelpers from "./controlsHelpers";
export * as AnnotationsHelpers from "./annotationsHelpers";
export * as SchemaHelpers from "./schemaHelpers";
export * as MatrixFBS from "./matrix";
+98 -9
View File
@@ -1,5 +1,7 @@
import { flatbuffers } from "flatbuffers";
import { NetEncoding } from "./matrix_generated";
import { isTypedArray } from "../typeHelpers";
import { IdentityInt32Index, DenseInt32Index, KeyIndex } from "../dataframe";
const utf8Decoder = new TextDecoder("utf-8");
@@ -41,25 +43,25 @@ Returns: object containing decoded Matrix:
colIdx: []|null
}
*/
function decodeMatrixFBS(arrayBuffer, inplace = false) {
export function decodeMatrixFBS(arrayBuffer, inplace = false) {
const bb = new flatbuffers.ByteBuffer(new Uint8Array(arrayBuffer));
const df = NetEncoding.Matrix.getRootAsMatrix(bb);
const matrix = NetEncoding.Matrix.getRootAsMatrix(bb);
const nRows = df.nRows();
const nCols = df.nCols();
const nRows = matrix.nRows();
const nCols = matrix.nCols();
/* decode columns */
const columnsLength = df.columnsLength();
const columnsLength = matrix.columnsLength();
const columns = Array(columnsLength).fill(null);
for (let c = 0; c < columnsLength; c += 1) {
const col = df.columns(c);
const col = matrix.columns(c);
columns[c] = decodeTypedArray(col.uType(), col.u.bind(col), inplace);
}
/* decode col_idx */
const colIdx = decodeTypedArray(
df.colIndexType(),
df.colIndex.bind(df),
matrix.colIndexType(),
matrix.colIndex.bind(matrix),
inplace
);
@@ -72,4 +74,91 @@ function decodeMatrixFBS(arrayBuffer, inplace = false) {
};
}
export default decodeMatrixFBS;
function encodeTypedArray(builder, uType, uData) {
const uTypeName = NetEncoding.TypedArray[uType];
const ArrayType = NetEncoding[uTypeName];
const dv = ArrayType.createDataVector(builder, uData);
builder.startObject(1);
builder.addFieldOffset(0, dv, 0);
return builder.endObject();
}
export function encodeMatrixFBS(df) {
/*
encode the dataframe as an FBS Matrix
*/
/* row indexing not supported currently */
if (df.rowIndex.constructor !== IdentityInt32Index) {
throw new Error("FBS does not support row index encoding at this time");
}
const shape = df.dims;
const utf8Encoder = new TextEncoder("utf-8");
const builder = new flatbuffers.Builder(1024);
let encColIndex;
let encColIndexUType;
let encColumns;
if (shape[0] > 0 && shape[1] > 0) {
const columns = df.columns().map(col => col.asArray());
const cols = columns.map(carr => {
let uType;
let tarr;
if (isTypedArray(carr)) {
uType = NetEncoding.TypedArray[carr.constructor.name];
tarr = encodeTypedArray(builder, uType, carr);
} else {
uType = NetEncoding.TypedArray.JSONEncodedArray;
const json = JSON.stringify(carr);
const jsonUTF8 = utf8Encoder.encode(json);
tarr = encodeTypedArray(builder, uType, jsonUTF8);
}
NetEncoding.Column.startColumn(builder);
NetEncoding.Column.addUType(builder, uType);
NetEncoding.Column.addU(builder, tarr);
return NetEncoding.Column.endColumn(builder);
});
encColumns = NetEncoding.Matrix.createColumnsVector(builder, cols);
if (df.colIndex && shape[1] > 0) {
const colIndexType = df.colIndex.constructor;
if (colIndexType === IdentityInt32Index) {
encColIndex = undefined;
} else if (colIndexType === DenseInt32Index) {
encColIndexUType = NetEncoding.TypedArray.Int32Array;
encColIndex = encodeTypedArray(
builder,
encColIndexUType,
df.colIndex.keys()
);
} else if (colIndexType === KeyIndex) {
encColIndexUType = NetEncoding.TypedArray.JSONEncodedArray;
encColIndex = encodeTypedArray(
builder,
encColIndexUType,
utf8Encoder.encode(JSON.stringify(df.colIndex.keys()))
);
} else {
throw new Error("Index type FBS encoding unsupported");
}
}
}
NetEncoding.Matrix.startMatrix(builder);
NetEncoding.Matrix.addNRows(builder, shape[0]);
NetEncoding.Matrix.addNCols(builder, shape[1]);
if (encColumns) {
NetEncoding.Matrix.addColumns(builder, encColumns);
}
if (encColIndexUType) {
NetEncoding.Matrix.addColIndexType(builder, encColIndexUType);
NetEncoding.Matrix.addColIndex(builder, encColIndex);
}
const root = NetEncoding.Matrix.endMatrix(builder);
builder.finish(root);
return builder.asUint8Array();
}
@@ -0,0 +1,96 @@
/*
Helpers for schema management
*/
import _ from "lodash";
import fromEntries from "../fromEntries";
/*
System wide schema assumptions:
- schema and data wil be consistent (eg, for user-created annotations)
- schema will be internally self-consistent (eg, index matches columns)
- world & universe schema are same - only data is subset
*/
export function indexEntireSchema(schema) {
/* Index schema for ease of use */
schema.annotations.obsByName = fromEntries(
schema.annotations.obs.columns.map(v => [v.name, v])
);
schema.annotations.varByName = fromEntries(
schema.annotations.var.columns.map(v => [v.name, v])
);
schema.layout.obsByName = fromEntries(
schema.layout.obs.map(v => [v.name, v])
);
schema.layout.varByName = fromEntries(
schema.layout.var.map(v => [v.name, v])
);
return schema;
}
function _copy(schema) {
/* redux copy conventions - WARNING, only for modifyign obs annotations */
return {
...schema,
annotations: {
...schema.annotations,
obs: _.cloneDeep(schema.annotations.obs)
}
};
}
function _reindex(schema) {
/* reindex obs annotations ONLY */
schema.annotations.obsByName = fromEntries(
schema.annotations.obs.columns.map(v => [v.name, v])
);
return schema;
}
export function removeObsAnnoColumn(schema, name) {
const newSchema = _copy(schema);
newSchema.annotations.obs.columns = schema.annotations.obs.columns.filter(
v => v.name !== name
);
return _reindex(newSchema);
}
export function addObsAnnoColumn(schema, name, defn) {
const newSchema = _copy(schema);
newSchema.annotations.obs.columns.push(defn);
return _reindex(newSchema);
}
export function removeObsAnnoCategory(schema, name, category) {
/* remove a category from a categorical annotation */
const categories = schema.annotations.obsByName[name]?.categories;
if (!categories)
throw new Error("column does not exist or is not categorical");
const idx = categories.indexOf(category);
if (idx === -1) throw new Error("category does not exist");
const newSchema = _reindex(_copy(schema));
/* remove category */
newSchema.annotations.obsByName[name].categories.splice(idx, 1);
return newSchema;
}
export function addObsAnnoCategory(schema, name, category) {
/* add a category to a categorical annotation */
const categories = schema.annotations.obsByName[name]?.categories;
if (!categories)
throw new Error("column does not exist or is not categorical");
const idx = categories.indexOf(category);
if (idx !== -1) throw new Error("category already exists");
const newSchema = _reindex(_copy(schema));
/* remove category */
newSchema.annotations.obsByName[name].categories.push(category);
return newSchema;
}
+29 -19
View File
@@ -1,11 +1,11 @@
// jshint esversion: 6
import _ from "lodash";
import decodeMatrixFBS from "./matrix";
import { unassignedCategoryLabel } from "../../globals";
import { decodeMatrixFBS } from "./matrix";
import * as Dataframe from "../dataframe";
import fromEntries from "../fromEntries";
import { isFpTypedArray } from "../typeHelpers";
import { indexEntireSchema } from "./schemaHelpers";
import { isCategoricalAnnotation } from "./annotationsHelpers";
/*
Private helper function - create and return a template Universe
@@ -18,10 +18,13 @@ function templateUniverse() {
schema: {},
/*
Annotations
annotations
*/
obsAnnotations: Dataframe.Dataframe.empty(),
varAnnotations: Dataframe.Dataframe.empty(),
/*
layout
*/
obsLayout: Dataframe.Dataframe.empty(),
/*
@@ -122,6 +125,10 @@ function reconcileSchemaCategoriesWithSummary(universe) {
For example, boolean defined fields in the schema do not contain
explicit declaration of categories (nor do string fields). In these
cases, add a 'categories' field to the schema so it is accessible.
In addition, we have a client-side convention (UI) that all writable
annotations must have an 'unassigned' category, even if it is not currently
in use.
*/
universe.schema.annotations.obs.columns.forEach(s => {
@@ -136,6 +143,10 @@ function reconcileSchemaCategoriesWithSummary(universe) {
);
s.categories = categories;
}
if (s.writable && s.categories.indexOf(unassignedCategoryLabel) === -1) {
s.categories = s.categories.concat(unassignedCategoryLabel);
}
});
}
@@ -166,7 +177,7 @@ export function createUniverseFromResponse(
/* layout */
universe.obsLayout = LayoutFBSToDataframe(layoutFBSResponse);
/* sanity check */
/* sanity checks */
if (
universe.nObs !== universe.obsLayout.length ||
universe.nObs !== universe.obsAnnotations.length ||
@@ -176,20 +187,19 @@ export function createUniverseFromResponse(
}
reconcileSchemaCategoriesWithSummary(universe);
indexEntireSchema(universe.schema);
/* sanity checks */
if (
schema.annotations.obs.columns.some(
s => s.writable && !isCategoricalAnnotation(schema, s.name)
)
) {
throw new Error(
"Writable continuous obs annotations are not supproted - failed to laod"
);
}
/* Index schema for ease of use */
universe.schema.annotations.obsByName = fromEntries(
universe.schema.annotations.obs.columns.map(v => [v.name, v])
);
universe.schema.annotations.varByName = fromEntries(
universe.schema.annotations.var.columns.map(v => [v.name, v])
);
universe.schema.layout.obsByName = fromEntries(
universe.schema.layout.obs.map(v => [v.name, v])
);
universe.schema.layout.varByName = fromEntries(
universe.schema.layout.var.map(v => [v.name, v])
);
return universe;
}
+4 -11
View File
@@ -1,14 +1,7 @@
// jshint esversion: 6
import clip from "../clip";
import {
layoutDimensionName,
obsAnnoDimensionName,
diffexpDimensionName,
userDefinedDimensionName
} from "../nameCreators";
import { layoutDimensionName, obsAnnoDimensionName } from "../nameCreators";
import * as Dataframe from "../dataframe";
import ImmutableTypedCrossfilter from "../typedCrossfilter/crossfilter";
import { isContinuousAnnotation } from "./annotationsHelpers";
/*
@@ -158,7 +151,7 @@ and world.varData.
function setClippedDataframes(world) {
const { schema } = world;
const isContinuousObsAnnotation = (df, idx, label) =>
deduceDimensionType(schema.annotations.obsByName[label], label) !== "enum";
isContinuousAnnotation(schema, label);
const obsQuantile = (label, q) =>
world.unclipped.obsAnnotations.col(label).summarize().percentiles[100 * q];
world.obsAnnotations = clipDataframe(
@@ -183,7 +176,7 @@ function setClippedDataframes(world) {
/*
Subset the current world based upon the current selection, maintaining any existing
clip. Returns new world. Parameters:
* unvierse
* universe
* world - the current world
* crossfilter - the selection state
*/