mirror of
https://github.com/chanzuckerberg/cellxgene.git
synced 2026-09-30 03:08:12 +08:00
Experimental - manual annotations (#837)
* icons, partway * redux for values * onChange * cancel * annotations lifecycle for category names * copy categorical * edit category * add Dataframe.withColsFrom * render user annotations; default add/delete annotation category * add label name to actions * category name edit * error checking improvements * change schema field isUserAnnotation to writable * always have an unassigned label; implement delete label * implement add new label and edit label name * label current cell selection * fix select exact bug in crossfilter * clean up categorical reducer * fix tests * remove debugging printf * implement subset/reset for user annotations * undo redo support for user annotations * remove duplicate button from categories * add modal * remove obsolete duplicate annotation reducers * remove old debugging printf * connect modal to annotation create and dup * initial full-stack wiring * finish up end-to-end wiring * fix existing unit tests * fix pytests to match new schema API * remove debugging printfs * add label file rotation * remove obsolete comment * add fbs encode/decode tests * add tests for writable annotations * simplify code * fix hashing bug with FBS encoding * lint * fix smoke tests * improve error checking in Dataframe.withColsFrom * add unit test for Dataframe.withColsFrom * add unit test for Dataframe.columns and Dataframe.renameCol * fix bug in FBS encode, add better error checks, refactor * add FBS encode/decode test * add clarifying comment * clean up action type names; fix state inconsistency in crossfilter update * change autosave timer to 2.5sec * sort categorical metadata render order so it remains consistent * add temporary autogenerated label for add-new-label operation * fix hover-over label menu interference with cell highlighting * remove debugging code * add missing reducer cases & fix typo * make dataframe memoize more general purpose * add dev mode for annos * fix error on select duplicate * handle zero occupancy categories * correctly maintain unclipped AND clipped world * correctly handle zero length FBS matrix and label files * ensure all writable categorical schema contains an unassigned category * handle case where building occupancy stack for category with no members * dialog for creating label, disable button if duplicate or empty * visually separate writeable * edit category * fix edit category name * remove debugging code * fix edit annotation label * visually define unassigned, change options * Pull in requirements.txt from `master` * label currently selected cells * duplicate label * lint * fix pytest merge issues * rename --label-file to --experimental-label-file * remove debugging console log * spelling error fix; fix bug found in PR review. * lint
This commit is contained in:
committed by
Colin Megill
parent
ab2c423006
commit
3660a6cc27
@@ -345,12 +345,50 @@ class Dataframe {
|
||||
);
|
||||
}
|
||||
|
||||
withColsFrom(dataframe) {
|
||||
/*
|
||||
return a new dataframe containing all columns from both `this` and the
|
||||
provided dataframe.
|
||||
|
||||
The row index from `this` will be used. Both dataframes must have identical
|
||||
dimensionality, and no overlapping columns labels.
|
||||
*/
|
||||
const dims = [this.dims[0], this.dims[1] + dataframe.dims[1]];
|
||||
const { rowIndex } = this;
|
||||
const columns = [...this.__columns, ...dataframe.__columns];
|
||||
const colIndex = this.colIndex.withLabels(dataframe.colIndex.keys());
|
||||
const columnsAccessor = [
|
||||
...this.__columnsAccessor,
|
||||
...dataframe.__columnsAccessor
|
||||
];
|
||||
return new this.constructor(
|
||||
dims,
|
||||
columns,
|
||||
rowIndex,
|
||||
colIndex,
|
||||
columnsAccessor
|
||||
);
|
||||
}
|
||||
|
||||
dropCol(label) {
|
||||
/*
|
||||
Create a new dataframe, omitting one columns.
|
||||
|
||||
const newDf = df.dropCol("colors");
|
||||
|
||||
Corner case to manage: if dropping the last column, return an empty dataframe.
|
||||
*/
|
||||
if (!this.hasCol(label)) {
|
||||
throw new RangeError(`unknown label: ${label}`);
|
||||
}
|
||||
|
||||
/*
|
||||
Corner case to manage: if dropping the last column, return an empty dataframe.
|
||||
*/
|
||||
if (this.dims[1] === 1) {
|
||||
return Dataframe.empty();
|
||||
}
|
||||
|
||||
const dims = [this.dims[0], this.dims[1] - 1];
|
||||
const coffset = this.colIndex.getOffset(label);
|
||||
const columns = [...this.__columns];
|
||||
@@ -367,6 +405,50 @@ class Dataframe {
|
||||
);
|
||||
}
|
||||
|
||||
renameCol(oldLabel, newLabel) {
|
||||
/*
|
||||
Accelerator for dropping a column and then adding it again with a new label
|
||||
*/
|
||||
const coffset = this.colIndex.getOffset(oldLabel);
|
||||
const colIndex = this.colIndex.dropLabel(oldLabel).withLabel(newLabel);
|
||||
|
||||
const columns = [...this.__columns];
|
||||
columns.push(columns[coffset]);
|
||||
columns.splice(coffset, 1);
|
||||
|
||||
const columnsAccessor = [...this.__columnsAccessor];
|
||||
columnsAccessor.push(columnsAccessor[coffset]);
|
||||
columnsAccessor.splice(coffset, 1);
|
||||
|
||||
return new this.constructor(
|
||||
this.dims,
|
||||
columns,
|
||||
this.rowIndex,
|
||||
colIndex,
|
||||
columnsAccessor
|
||||
);
|
||||
}
|
||||
|
||||
replaceColData(label, newColData) {
|
||||
/*
|
||||
Accelerator for dropping a column then adding it again with same
|
||||
label and different values.
|
||||
*/
|
||||
const coffset = this.colIndex.getOffset(label);
|
||||
const columns = [...this.__columns];
|
||||
columns[coffset] = newColData;
|
||||
const columnsAccessor = [...this.__columnsAccessor];
|
||||
columnsAccessor[coffset] = null;
|
||||
|
||||
return new this.constructor(
|
||||
this.dims,
|
||||
columns,
|
||||
this.rowIndex,
|
||||
this.colIndex,
|
||||
columnsAccessor
|
||||
);
|
||||
}
|
||||
|
||||
static empty(rowIndex = null, colIndex = null) {
|
||||
return new Dataframe([0, 0], [], rowIndex, colIndex);
|
||||
}
|
||||
@@ -443,6 +525,8 @@ class Dataframe {
|
||||
return newCol;
|
||||
});
|
||||
}
|
||||
|
||||
if (dims[0] === 0 || dims[1] === 0) return Dataframe.empty();
|
||||
return new Dataframe(dims, columns, rowIndex, colIndex);
|
||||
}
|
||||
|
||||
@@ -526,6 +610,11 @@ class Dataframe {
|
||||
Data access with row/col.
|
||||
**/
|
||||
|
||||
columns() {
|
||||
/* return all column accessors as an array, in offset order */
|
||||
return [...this.__columnsAccessor];
|
||||
}
|
||||
|
||||
col(columnLabel) {
|
||||
/*
|
||||
Return accessor bound to a column. Allows random row access
|
||||
|
||||
@@ -80,6 +80,10 @@ class IdentityInt32Index {
|
||||
return this.__promote([...this.keys(), label]);
|
||||
}
|
||||
|
||||
withLabels(labels) {
|
||||
return this.__promote([...this.keys(), ...labels]);
|
||||
}
|
||||
|
||||
dropLabel(label) {
|
||||
if (label === this.maxOffset - 1) {
|
||||
return new IdentityInt32Index(label);
|
||||
@@ -163,6 +167,10 @@ class DenseInt32Index {
|
||||
return this.__promote([...this.keys(), label]);
|
||||
}
|
||||
|
||||
withLabels(labels) {
|
||||
return this.__promote([...this.keys(), ...labels]);
|
||||
}
|
||||
|
||||
dropLabel(label) {
|
||||
const labelArray = [...this.keys()];
|
||||
labelArray.splice(labelArray.indexOf(label), 1);
|
||||
@@ -187,6 +195,11 @@ class KeyIndex {
|
||||
index.set(v, i);
|
||||
});
|
||||
|
||||
if (index.size !== rindex.length) {
|
||||
/* if true, there was a duplicate in the keys */
|
||||
throw new Error("duplicate label provided to KeyIndex");
|
||||
}
|
||||
|
||||
this.index = index;
|
||||
this.rindex = rindex;
|
||||
this.__compile();
|
||||
@@ -218,6 +231,10 @@ class KeyIndex {
|
||||
return new KeyIndex([...this.rindex, label]);
|
||||
}
|
||||
|
||||
withLabels(labels) {
|
||||
return new KeyIndex([...this.rindex, ...labels]);
|
||||
}
|
||||
|
||||
dropLabel(label) {
|
||||
const idx = this.rindex.indexOf(label);
|
||||
const labelArray = [...this.rindex];
|
||||
|
||||
@@ -21,7 +21,7 @@ export function callOnceLazy(f) {
|
||||
return result;
|
||||
}
|
||||
|
||||
export function memoize(fn, hashFn) {
|
||||
export function memoize(fn, hashFn, maxResultsCached = -1) {
|
||||
/*
|
||||
function memoization, with user-provided hash. hashFn must return a
|
||||
key which will be unique as a Map key (ie, obeys "sameValueZero" algorithm
|
||||
@@ -36,7 +36,19 @@ export function memoize(fn, hashFn) {
|
||||
}
|
||||
const result = fn(...args);
|
||||
cache.set(key, result);
|
||||
|
||||
if (maxResultsCached > -1 && cache.size > maxResultsCached) {
|
||||
/* Least recent insertion deletion */
|
||||
cache.delete(cache.keys().next().value);
|
||||
}
|
||||
|
||||
return result;
|
||||
};
|
||||
|
||||
wrap.clear = function clear() {
|
||||
/* clear memoization cache */
|
||||
cache.clear();
|
||||
};
|
||||
|
||||
return wrap;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
/*
|
||||
Helper functions for user-editable nnotations state management.
|
||||
See also reducers/annotations.js
|
||||
*/
|
||||
import { unassignedCategoryLabel } from "../../globals";
|
||||
import * as SchemaHelpers from "./schemaHelpers";
|
||||
import { obsAnnoDimensionName } from "../nameCreators";
|
||||
|
||||
/*
|
||||
There are a number of state constraints assumed throughout the
|
||||
application:
|
||||
- all obs annotations are in {world|universe}.obsAnnotations,
|
||||
regardless of whether or not they are user editable.
|
||||
- the {world|universe}.schema is always up to date and matches
|
||||
the data
|
||||
- the schema flag `writable` correctly indicates whether
|
||||
the annotation is editable/mutable.
|
||||
|
||||
In addition, the current state management only allows for
|
||||
categorical annotations to be writable.
|
||||
*/
|
||||
|
||||
export function isCategoricalAnnotation(schema, name) {
|
||||
/* we treat any string, categorical or boolean as a categorical */
|
||||
const { type } = schema.annotations.obsByName[name];
|
||||
return type === "string" || type === "boolean" || type === "categorical";
|
||||
}
|
||||
|
||||
export function isContinuousAnnotation(schema, name) {
|
||||
return !isCategoricalAnnotation(schema, name);
|
||||
}
|
||||
|
||||
function _isUserAnnotation(schema, name) {
|
||||
return schema.annotations.obsByName[name]?.writable;
|
||||
}
|
||||
|
||||
export function isUserAnnotation(worldOrUniverse, name) {
|
||||
return _isUserAnnotation(worldOrUniverse.schema, name);
|
||||
}
|
||||
|
||||
export function removeObsAnnoSchema(schema, name) {
|
||||
/*
|
||||
remove named annotation from obs annotation schema
|
||||
*/
|
||||
|
||||
/* only remove if it exists and is a user annotation */
|
||||
if (!_isUserAnnotation(schema, name))
|
||||
throw new Error("removing non-user-defined schema");
|
||||
return SchemaHelpers.removeObsAnnoColumn(schema, name);
|
||||
}
|
||||
|
||||
export function addObsAnnoSchema(schema, name, colSchema) {
|
||||
/*
|
||||
add a categorical type to the obs annotation schema
|
||||
*/
|
||||
|
||||
/* collision detection */
|
||||
if (schema.annotations.obs.columns.some(v => v.name === name))
|
||||
throw Error("annotations may not contain duplicate category names");
|
||||
if (name !== colSchema.name) throw Error("column schema does not match");
|
||||
return SchemaHelpers.addObsAnnoColumn(schema, name, colSchema);
|
||||
}
|
||||
|
||||
export function dupObsAnnoSchema(schema, sourceName, dupName, defaultSchema) {
|
||||
/*
|
||||
duplicate the obs annotation `sourceName` schema, but with the name `dupName`
|
||||
*/
|
||||
const colSchema = {
|
||||
...schema.annotations.obsByName[sourceName],
|
||||
...defaultSchema,
|
||||
name: dupName
|
||||
};
|
||||
/* existance check */
|
||||
if (!colSchema) throw Error("source annotation does not exist");
|
||||
/* collision detection */
|
||||
if (schema.annotations.obs.columns.some(v => v.name === dupName))
|
||||
throw Error("annotations may not contain duplicate category names");
|
||||
return SchemaHelpers.addObsAnnoColumn(schema, dupName, colSchema);
|
||||
}
|
||||
|
||||
export function removeObsAnnoCategory(schema, name, category) {
|
||||
/* don't allow deletion of unassigned category on writable annotations */
|
||||
|
||||
if (!_isUserAnnotation(schema, name))
|
||||
throw new Error("unable to modify read-only schema");
|
||||
if (category === unassignedCategoryLabel)
|
||||
throw new Error("may not remove unassigned category label");
|
||||
|
||||
return SchemaHelpers.removeObsAnnoCategory(schema, name, category);
|
||||
}
|
||||
|
||||
export function addObsAnnoCategory(schema, name, category) {
|
||||
if (!_isUserAnnotation(schema, name))
|
||||
throw new Error("unable to modify read-only schema");
|
||||
|
||||
return SchemaHelpers.addObsAnnoCategory(schema, name, category);
|
||||
}
|
||||
|
||||
export function setLabelByValue(df, colName, fromLabel, toLabel) {
|
||||
/*
|
||||
in the dataframe column `colName`, set any value of `fromLabel` to `toLabel`
|
||||
*/
|
||||
const keys = df.colIndex.keys();
|
||||
const ndf = df.mapColumns((col, colIdx) => {
|
||||
if (colName !== keys[colIdx]) return col;
|
||||
|
||||
/* clone data and return it. */
|
||||
const newCol = col.slice();
|
||||
for (let i = 0, l = newCol.length; i < l; i += 1) {
|
||||
if (newCol[i] === fromLabel) newCol[i] = toLabel;
|
||||
}
|
||||
return newCol;
|
||||
});
|
||||
return ndf;
|
||||
}
|
||||
|
||||
export function setLabelByMask(df, colName, mask, label) {
|
||||
/*
|
||||
in the dataframe column `colName`, set the masked rows to 'label'
|
||||
*/
|
||||
const keys = df.colIndex.keys();
|
||||
const ndf = df.mapColumns((col, colIdx) => {
|
||||
if (colName !== keys[colIdx]) return col;
|
||||
|
||||
/* clone data and return it. */
|
||||
const newCol = col.slice();
|
||||
for (let i = 0, l = newCol.length; i < l; i += 1) {
|
||||
if (mask[i]) newCol[i] = label;
|
||||
}
|
||||
return newCol;
|
||||
});
|
||||
return ndf;
|
||||
}
|
||||
|
||||
export function worldToUniverseMask(worldMask, worldObsAnnotations, nObs) {
|
||||
/*
|
||||
given world seleciton mask, return a selection mask for entire universe
|
||||
that has same selection state.
|
||||
*/
|
||||
const mask = new Uint8Array(nObs);
|
||||
const { rowIndex } = worldObsAnnotations;
|
||||
|
||||
for (let i = 0, l = worldMask.length; i < l; i += 1) {
|
||||
if (worldMask[i]) {
|
||||
const label = rowIndex.getLabel(i);
|
||||
mask[label] = 1;
|
||||
}
|
||||
}
|
||||
return mask;
|
||||
}
|
||||
|
||||
export function createWritableAnnotationDimensions(world, crossfilter) {
|
||||
const { obsAnnotations, schema } = world;
|
||||
const writableAnnotations = schema.annotations.obs.columns
|
||||
.filter(s => s.writable)
|
||||
.map(s => s.name);
|
||||
|
||||
crossfilter = writableAnnotations.reduce((xflt, anno) => {
|
||||
const dimName = obsAnnoDimensionName(anno);
|
||||
if (xflt.hasDimension(dimName)) xflt = xflt.delDimension(dimName);
|
||||
return xflt.addDimension(
|
||||
dimName,
|
||||
"enum",
|
||||
obsAnnotations.col(anno).asArray()
|
||||
);
|
||||
}, crossfilter);
|
||||
return crossfilter;
|
||||
}
|
||||
@@ -50,8 +50,9 @@ function createColorsByCategoricalMetadata(world, accessor) {
|
||||
}, {});
|
||||
|
||||
const rgb = new Array(world.nObs);
|
||||
const data = world.obsAnnotations.col(accessor).asArray();
|
||||
for (let i = 0, len = world.obsAnnotations.length; i < len; i += 1) {
|
||||
const df = world.obsAnnotations;
|
||||
const data = df.col(accessor).asArray();
|
||||
for (let i = 0, len = df.length; i < len; i += 1) {
|
||||
const cat = data[i];
|
||||
rgb[i] = colors[cat];
|
||||
}
|
||||
|
||||
@@ -37,16 +37,14 @@ Remember that option values can be ANY js type, except undefined/null.
|
||||
}
|
||||
}
|
||||
*/
|
||||
function topNCategories(summary) {
|
||||
const counts = _.map(summary.categories, cat =>
|
||||
summary.categoryCounts.get(cat)
|
||||
);
|
||||
const sortIndex = fillRange(new Array(summary.numCategories)).sort(
|
||||
function topNCategories(colSchema, summary, N) {
|
||||
const { categories } = colSchema;
|
||||
const counts = _.map(categories, cat => summary.categoryCounts.get(cat) ?? 0);
|
||||
const sortIndex = fillRange(new Array(categories.length)).sort(
|
||||
(a, b) => counts[b] - counts[a]
|
||||
);
|
||||
const sortedCategories = _.map(sortIndex, i => summary.categories[i]);
|
||||
const sortedCategories = _.map(sortIndex, i => categories[i]);
|
||||
const sortedCounts = _.map(sortIndex, i => counts[i]);
|
||||
const N = globals.maxCategoricalOptionsToDisplay;
|
||||
|
||||
if (sortedCategories.length < N) {
|
||||
return [sortedCategories, sortedCounts];
|
||||
@@ -54,64 +52,56 @@ function topNCategories(summary) {
|
||||
return [sortedCategories.slice(0, N), sortedCounts.slice(0, N)];
|
||||
}
|
||||
|
||||
export function createCategoricalSelection(maxCategoryItems, world) {
|
||||
const res = {};
|
||||
const obsIndexName = world.schema.annotations.obs.index;
|
||||
_.forEach(world.obsAnnotations.colIndex.keys(), key => {
|
||||
const summary = world.obsAnnotations.col(key).summarize();
|
||||
if (summary.categories) {
|
||||
const isColorField = key.includes("color") || key.includes("Color");
|
||||
const isSelectableCategory =
|
||||
!isColorField &&
|
||||
key !== obsIndexName &&
|
||||
summary.categories.length < maxCategoryItems;
|
||||
if (isSelectableCategory) {
|
||||
const [categoryValues, categoryValueCounts] = topNCategories(summary);
|
||||
const categoryValueIndices = new Map(
|
||||
categoryValues.map((v, i) => [v, i])
|
||||
);
|
||||
const numCategoryValues = categoryValueIndices.size;
|
||||
const categoryValueSelected = new Array(numCategoryValues).fill(true);
|
||||
const isTruncated = categoryValues.length < summary.numCategories;
|
||||
res[key] = {
|
||||
categoryValues, // array: of natively typed category values
|
||||
categoryValueIndices, // map: category value (native type) -> category index
|
||||
categoryValueSelected, // array: t/f selection state
|
||||
numCategoryValues, // number: of values in the category
|
||||
isTruncated, // bool: true if list was truncated
|
||||
categoryValueCounts, // array: cardinality of each category,
|
||||
categorySelected: true // bool - default state for entire category
|
||||
};
|
||||
}
|
||||
}
|
||||
});
|
||||
return res;
|
||||
export function selectableCategoryNames(world, maxCategoryItems) {
|
||||
const { schema } = world;
|
||||
const { index, columns } = schema.annotations.obs;
|
||||
return columns
|
||||
.filter(colSchema => {
|
||||
const { name, categories } = colSchema;
|
||||
return (
|
||||
categories && categories.length < maxCategoryItems && name !== index
|
||||
);
|
||||
})
|
||||
.map(v => v.name);
|
||||
}
|
||||
|
||||
/*
|
||||
given a categoricalSelection, return the list of all category values
|
||||
where selection state is true (ie, they are selected).
|
||||
*/
|
||||
export function selectedValuesForCategory(categorySelectionState, dfColumn) {
|
||||
const {
|
||||
categorySelected,
|
||||
categoryValueSelected,
|
||||
categoryValueIndices
|
||||
} = categorySelectionState;
|
||||
let selectedValues;
|
||||
if (categorySelected) {
|
||||
selectedValues = new Set(dfColumn.summarize().categories);
|
||||
} else {
|
||||
selectedValues = new Set();
|
||||
}
|
||||
categoryValueIndices.forEach((catIndex, catValue) => {
|
||||
if (!categoryValueSelected[catIndex]) {
|
||||
selectedValues.delete(catValue);
|
||||
} else {
|
||||
selectedValues.add(catValue);
|
||||
}
|
||||
});
|
||||
return [...selectedValues.values()];
|
||||
export function createCategoricalSelection(world, names) {
|
||||
const N = globals.maxCategoricalOptionsToDisplay;
|
||||
const { obsAnnotations, schema } = world;
|
||||
|
||||
const res = names.reduce((acc, name) => {
|
||||
const colSchema = schema.annotations.obsByName[name];
|
||||
const { writable: isUserAnno } = colSchema;
|
||||
|
||||
/*
|
||||
Summarize the annotation data currently in world. Must return categoryValues
|
||||
in sorted order, and must include all category values even if they are not
|
||||
actively used in the current world.
|
||||
*/
|
||||
const summary = obsAnnotations.col(name).summarize();
|
||||
const [categoryValues, categoryValueCounts] = topNCategories(
|
||||
colSchema,
|
||||
summary,
|
||||
N
|
||||
);
|
||||
const categoryValueIndices = new Map(categoryValues.map((v, i) => [v, i]));
|
||||
const numCategoryValues = categoryValueIndices.size;
|
||||
const categoryValueSelected = new Array(numCategoryValues).fill(true);
|
||||
const isTruncated = categoryValues.length < summary.numCategories;
|
||||
|
||||
acc[name] = {
|
||||
categoryValues, // array: of natively typed category values
|
||||
categoryValueIndices, // map: category value (native type) -> category index
|
||||
categoryValueSelected, // array: t/f selection state
|
||||
numCategoryValues, // number: of values in the category
|
||||
isTruncated, // bool: true if list was truncated
|
||||
categoryValueCounts, // array: cardinality of each category,
|
||||
categorySelected: true, // bool - default state for entire category
|
||||
isUserAnno // bool
|
||||
};
|
||||
return acc;
|
||||
}, {});
|
||||
return res;
|
||||
}
|
||||
|
||||
/*
|
||||
|
||||
@@ -19,3 +19,6 @@ export * as Universe from "./universe";
|
||||
export * as World from "./world";
|
||||
export * as WorldUtil from "./worldUtil";
|
||||
export * as ControlsHelpers from "./controlsHelpers";
|
||||
export * as AnnotationsHelpers from "./annotationsHelpers";
|
||||
export * as SchemaHelpers from "./schemaHelpers";
|
||||
export * as MatrixFBS from "./matrix";
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
import { flatbuffers } from "flatbuffers";
|
||||
import { NetEncoding } from "./matrix_generated";
|
||||
import { isTypedArray } from "../typeHelpers";
|
||||
import { IdentityInt32Index, DenseInt32Index, KeyIndex } from "../dataframe";
|
||||
|
||||
const utf8Decoder = new TextDecoder("utf-8");
|
||||
|
||||
@@ -41,25 +43,25 @@ Returns: object containing decoded Matrix:
|
||||
colIdx: []|null
|
||||
}
|
||||
*/
|
||||
function decodeMatrixFBS(arrayBuffer, inplace = false) {
|
||||
export function decodeMatrixFBS(arrayBuffer, inplace = false) {
|
||||
const bb = new flatbuffers.ByteBuffer(new Uint8Array(arrayBuffer));
|
||||
const df = NetEncoding.Matrix.getRootAsMatrix(bb);
|
||||
const matrix = NetEncoding.Matrix.getRootAsMatrix(bb);
|
||||
|
||||
const nRows = df.nRows();
|
||||
const nCols = df.nCols();
|
||||
const nRows = matrix.nRows();
|
||||
const nCols = matrix.nCols();
|
||||
|
||||
/* decode columns */
|
||||
const columnsLength = df.columnsLength();
|
||||
const columnsLength = matrix.columnsLength();
|
||||
const columns = Array(columnsLength).fill(null);
|
||||
for (let c = 0; c < columnsLength; c += 1) {
|
||||
const col = df.columns(c);
|
||||
const col = matrix.columns(c);
|
||||
columns[c] = decodeTypedArray(col.uType(), col.u.bind(col), inplace);
|
||||
}
|
||||
|
||||
/* decode col_idx */
|
||||
const colIdx = decodeTypedArray(
|
||||
df.colIndexType(),
|
||||
df.colIndex.bind(df),
|
||||
matrix.colIndexType(),
|
||||
matrix.colIndex.bind(matrix),
|
||||
inplace
|
||||
);
|
||||
|
||||
@@ -72,4 +74,91 @@ function decodeMatrixFBS(arrayBuffer, inplace = false) {
|
||||
};
|
||||
}
|
||||
|
||||
export default decodeMatrixFBS;
|
||||
function encodeTypedArray(builder, uType, uData) {
|
||||
const uTypeName = NetEncoding.TypedArray[uType];
|
||||
const ArrayType = NetEncoding[uTypeName];
|
||||
const dv = ArrayType.createDataVector(builder, uData);
|
||||
builder.startObject(1);
|
||||
builder.addFieldOffset(0, dv, 0);
|
||||
return builder.endObject();
|
||||
}
|
||||
|
||||
export function encodeMatrixFBS(df) {
|
||||
/*
|
||||
encode the dataframe as an FBS Matrix
|
||||
*/
|
||||
|
||||
/* row indexing not supported currently */
|
||||
if (df.rowIndex.constructor !== IdentityInt32Index) {
|
||||
throw new Error("FBS does not support row index encoding at this time");
|
||||
}
|
||||
|
||||
const shape = df.dims;
|
||||
const utf8Encoder = new TextEncoder("utf-8");
|
||||
const builder = new flatbuffers.Builder(1024);
|
||||
|
||||
let encColIndex;
|
||||
let encColIndexUType;
|
||||
let encColumns;
|
||||
|
||||
if (shape[0] > 0 && shape[1] > 0) {
|
||||
const columns = df.columns().map(col => col.asArray());
|
||||
|
||||
const cols = columns.map(carr => {
|
||||
let uType;
|
||||
let tarr;
|
||||
if (isTypedArray(carr)) {
|
||||
uType = NetEncoding.TypedArray[carr.constructor.name];
|
||||
tarr = encodeTypedArray(builder, uType, carr);
|
||||
} else {
|
||||
uType = NetEncoding.TypedArray.JSONEncodedArray;
|
||||
const json = JSON.stringify(carr);
|
||||
const jsonUTF8 = utf8Encoder.encode(json);
|
||||
tarr = encodeTypedArray(builder, uType, jsonUTF8);
|
||||
}
|
||||
NetEncoding.Column.startColumn(builder);
|
||||
NetEncoding.Column.addUType(builder, uType);
|
||||
NetEncoding.Column.addU(builder, tarr);
|
||||
return NetEncoding.Column.endColumn(builder);
|
||||
});
|
||||
|
||||
encColumns = NetEncoding.Matrix.createColumnsVector(builder, cols);
|
||||
|
||||
if (df.colIndex && shape[1] > 0) {
|
||||
const colIndexType = df.colIndex.constructor;
|
||||
if (colIndexType === IdentityInt32Index) {
|
||||
encColIndex = undefined;
|
||||
} else if (colIndexType === DenseInt32Index) {
|
||||
encColIndexUType = NetEncoding.TypedArray.Int32Array;
|
||||
encColIndex = encodeTypedArray(
|
||||
builder,
|
||||
encColIndexUType,
|
||||
df.colIndex.keys()
|
||||
);
|
||||
} else if (colIndexType === KeyIndex) {
|
||||
encColIndexUType = NetEncoding.TypedArray.JSONEncodedArray;
|
||||
encColIndex = encodeTypedArray(
|
||||
builder,
|
||||
encColIndexUType,
|
||||
utf8Encoder.encode(JSON.stringify(df.colIndex.keys()))
|
||||
);
|
||||
} else {
|
||||
throw new Error("Index type FBS encoding unsupported");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
NetEncoding.Matrix.startMatrix(builder);
|
||||
NetEncoding.Matrix.addNRows(builder, shape[0]);
|
||||
NetEncoding.Matrix.addNCols(builder, shape[1]);
|
||||
if (encColumns) {
|
||||
NetEncoding.Matrix.addColumns(builder, encColumns);
|
||||
}
|
||||
if (encColIndexUType) {
|
||||
NetEncoding.Matrix.addColIndexType(builder, encColIndexUType);
|
||||
NetEncoding.Matrix.addColIndex(builder, encColIndex);
|
||||
}
|
||||
const root = NetEncoding.Matrix.endMatrix(builder);
|
||||
builder.finish(root);
|
||||
return builder.asUint8Array();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
/*
|
||||
Helpers for schema management
|
||||
*/
|
||||
import _ from "lodash";
|
||||
|
||||
import fromEntries from "../fromEntries";
|
||||
|
||||
/*
|
||||
System wide schema assumptions:
|
||||
- schema and data wil be consistent (eg, for user-created annotations)
|
||||
- schema will be internally self-consistent (eg, index matches columns)
|
||||
- world & universe schema are same - only data is subset
|
||||
*/
|
||||
|
||||
export function indexEntireSchema(schema) {
|
||||
/* Index schema for ease of use */
|
||||
schema.annotations.obsByName = fromEntries(
|
||||
schema.annotations.obs.columns.map(v => [v.name, v])
|
||||
);
|
||||
schema.annotations.varByName = fromEntries(
|
||||
schema.annotations.var.columns.map(v => [v.name, v])
|
||||
);
|
||||
schema.layout.obsByName = fromEntries(
|
||||
schema.layout.obs.map(v => [v.name, v])
|
||||
);
|
||||
schema.layout.varByName = fromEntries(
|
||||
schema.layout.var.map(v => [v.name, v])
|
||||
);
|
||||
|
||||
return schema;
|
||||
}
|
||||
|
||||
function _copy(schema) {
|
||||
/* redux copy conventions - WARNING, only for modifyign obs annotations */
|
||||
return {
|
||||
...schema,
|
||||
annotations: {
|
||||
...schema.annotations,
|
||||
obs: _.cloneDeep(schema.annotations.obs)
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
function _reindex(schema) {
|
||||
/* reindex obs annotations ONLY */
|
||||
schema.annotations.obsByName = fromEntries(
|
||||
schema.annotations.obs.columns.map(v => [v.name, v])
|
||||
);
|
||||
return schema;
|
||||
}
|
||||
|
||||
export function removeObsAnnoColumn(schema, name) {
|
||||
const newSchema = _copy(schema);
|
||||
newSchema.annotations.obs.columns = schema.annotations.obs.columns.filter(
|
||||
v => v.name !== name
|
||||
);
|
||||
return _reindex(newSchema);
|
||||
}
|
||||
|
||||
export function addObsAnnoColumn(schema, name, defn) {
|
||||
const newSchema = _copy(schema);
|
||||
newSchema.annotations.obs.columns.push(defn);
|
||||
return _reindex(newSchema);
|
||||
}
|
||||
|
||||
export function removeObsAnnoCategory(schema, name, category) {
|
||||
/* remove a category from a categorical annotation */
|
||||
const categories = schema.annotations.obsByName[name]?.categories;
|
||||
if (!categories)
|
||||
throw new Error("column does not exist or is not categorical");
|
||||
|
||||
const idx = categories.indexOf(category);
|
||||
if (idx === -1) throw new Error("category does not exist");
|
||||
|
||||
const newSchema = _reindex(_copy(schema));
|
||||
|
||||
/* remove category */
|
||||
newSchema.annotations.obsByName[name].categories.splice(idx, 1);
|
||||
return newSchema;
|
||||
}
|
||||
|
||||
export function addObsAnnoCategory(schema, name, category) {
|
||||
/* add a category to a categorical annotation */
|
||||
const categories = schema.annotations.obsByName[name]?.categories;
|
||||
if (!categories)
|
||||
throw new Error("column does not exist or is not categorical");
|
||||
|
||||
const idx = categories.indexOf(category);
|
||||
if (idx !== -1) throw new Error("category already exists");
|
||||
|
||||
const newSchema = _reindex(_copy(schema));
|
||||
|
||||
/* remove category */
|
||||
newSchema.annotations.obsByName[name].categories.push(category);
|
||||
return newSchema;
|
||||
}
|
||||
@@ -1,11 +1,11 @@
|
||||
// jshint esversion: 6
|
||||
|
||||
import _ from "lodash";
|
||||
|
||||
import decodeMatrixFBS from "./matrix";
|
||||
import { unassignedCategoryLabel } from "../../globals";
|
||||
import { decodeMatrixFBS } from "./matrix";
|
||||
import * as Dataframe from "../dataframe";
|
||||
import fromEntries from "../fromEntries";
|
||||
import { isFpTypedArray } from "../typeHelpers";
|
||||
import { indexEntireSchema } from "./schemaHelpers";
|
||||
import { isCategoricalAnnotation } from "./annotationsHelpers";
|
||||
|
||||
/*
|
||||
Private helper function - create and return a template Universe
|
||||
@@ -18,10 +18,13 @@ function templateUniverse() {
|
||||
schema: {},
|
||||
|
||||
/*
|
||||
Annotations
|
||||
annotations
|
||||
*/
|
||||
obsAnnotations: Dataframe.Dataframe.empty(),
|
||||
varAnnotations: Dataframe.Dataframe.empty(),
|
||||
/*
|
||||
layout
|
||||
*/
|
||||
obsLayout: Dataframe.Dataframe.empty(),
|
||||
|
||||
/*
|
||||
@@ -122,6 +125,10 @@ function reconcileSchemaCategoriesWithSummary(universe) {
|
||||
For example, boolean defined fields in the schema do not contain
|
||||
explicit declaration of categories (nor do string fields). In these
|
||||
cases, add a 'categories' field to the schema so it is accessible.
|
||||
|
||||
In addition, we have a client-side convention (UI) that all writable
|
||||
annotations must have an 'unassigned' category, even if it is not currently
|
||||
in use.
|
||||
*/
|
||||
|
||||
universe.schema.annotations.obs.columns.forEach(s => {
|
||||
@@ -136,6 +143,10 @@ function reconcileSchemaCategoriesWithSummary(universe) {
|
||||
);
|
||||
s.categories = categories;
|
||||
}
|
||||
|
||||
if (s.writable && s.categories.indexOf(unassignedCategoryLabel) === -1) {
|
||||
s.categories = s.categories.concat(unassignedCategoryLabel);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
@@ -166,7 +177,7 @@ export function createUniverseFromResponse(
|
||||
/* layout */
|
||||
universe.obsLayout = LayoutFBSToDataframe(layoutFBSResponse);
|
||||
|
||||
/* sanity check */
|
||||
/* sanity checks */
|
||||
if (
|
||||
universe.nObs !== universe.obsLayout.length ||
|
||||
universe.nObs !== universe.obsAnnotations.length ||
|
||||
@@ -176,20 +187,19 @@ export function createUniverseFromResponse(
|
||||
}
|
||||
|
||||
reconcileSchemaCategoriesWithSummary(universe);
|
||||
indexEntireSchema(universe.schema);
|
||||
|
||||
/* sanity checks */
|
||||
if (
|
||||
schema.annotations.obs.columns.some(
|
||||
s => s.writable && !isCategoricalAnnotation(schema, s.name)
|
||||
)
|
||||
) {
|
||||
throw new Error(
|
||||
"Writable continuous obs annotations are not supproted - failed to laod"
|
||||
);
|
||||
}
|
||||
|
||||
/* Index schema for ease of use */
|
||||
universe.schema.annotations.obsByName = fromEntries(
|
||||
universe.schema.annotations.obs.columns.map(v => [v.name, v])
|
||||
);
|
||||
universe.schema.annotations.varByName = fromEntries(
|
||||
universe.schema.annotations.var.columns.map(v => [v.name, v])
|
||||
);
|
||||
universe.schema.layout.obsByName = fromEntries(
|
||||
universe.schema.layout.obs.map(v => [v.name, v])
|
||||
);
|
||||
universe.schema.layout.varByName = fromEntries(
|
||||
universe.schema.layout.var.map(v => [v.name, v])
|
||||
);
|
||||
return universe;
|
||||
}
|
||||
|
||||
|
||||
@@ -1,14 +1,7 @@
|
||||
// jshint esversion: 6
|
||||
|
||||
import clip from "../clip";
|
||||
import {
|
||||
layoutDimensionName,
|
||||
obsAnnoDimensionName,
|
||||
diffexpDimensionName,
|
||||
userDefinedDimensionName
|
||||
} from "../nameCreators";
|
||||
import { layoutDimensionName, obsAnnoDimensionName } from "../nameCreators";
|
||||
import * as Dataframe from "../dataframe";
|
||||
import ImmutableTypedCrossfilter from "../typedCrossfilter/crossfilter";
|
||||
import { isContinuousAnnotation } from "./annotationsHelpers";
|
||||
|
||||
/*
|
||||
|
||||
@@ -158,7 +151,7 @@ and world.varData.
|
||||
function setClippedDataframes(world) {
|
||||
const { schema } = world;
|
||||
const isContinuousObsAnnotation = (df, idx, label) =>
|
||||
deduceDimensionType(schema.annotations.obsByName[label], label) !== "enum";
|
||||
isContinuousAnnotation(schema, label);
|
||||
const obsQuantile = (label, q) =>
|
||||
world.unclipped.obsAnnotations.col(label).summarize().percentiles[100 * q];
|
||||
world.obsAnnotations = clipDataframe(
|
||||
@@ -183,7 +176,7 @@ function setClippedDataframes(world) {
|
||||
/*
|
||||
Subset the current world based upon the current selection, maintaining any existing
|
||||
clip. Returns new world. Parameters:
|
||||
* unvierse
|
||||
* universe
|
||||
* world - the current world
|
||||
* crossfilter - the selection state
|
||||
*/
|
||||
|
||||
@@ -5,6 +5,7 @@ import BitArray from "./bitArray";
|
||||
import {
|
||||
sortArray,
|
||||
lowerBound,
|
||||
binarySearch,
|
||||
lowerBoundIndirect,
|
||||
upperBoundIndirect
|
||||
} from "./sort";
|
||||
@@ -56,6 +57,14 @@ export default class ImmutableTypedCrossfilter {
|
||||
return this.data;
|
||||
}
|
||||
|
||||
setData(data) {
|
||||
return new ImmutableTypedCrossfilter(
|
||||
data,
|
||||
this.dimensions,
|
||||
this.selectionCache
|
||||
);
|
||||
}
|
||||
|
||||
dimensionNames() {
|
||||
/* return array of all dimensions (by name) */
|
||||
return Object.keys(this.dimensions);
|
||||
@@ -113,6 +122,23 @@ export default class ImmutableTypedCrossfilter {
|
||||
return new ImmutableTypedCrossfilter(data, dimensions, selectionCache);
|
||||
}
|
||||
|
||||
renameDimension(oldName, newName) {
|
||||
/*
|
||||
rename a dimension
|
||||
*/
|
||||
const { [oldName]: dim, ...dimensions } = this.dimensions;
|
||||
const { data, selectionCache } = this;
|
||||
dim.dim.rename(newName);
|
||||
return new ImmutableTypedCrossfilter(
|
||||
data,
|
||||
{
|
||||
...dimensions,
|
||||
[newName]: dim
|
||||
},
|
||||
selectionCache
|
||||
);
|
||||
}
|
||||
|
||||
select(name, spec) {
|
||||
/*
|
||||
select on named dimension, as indicated by `spec`. Spec is an object
|
||||
@@ -288,6 +314,10 @@ class _ImmutableBaseDimension {
|
||||
this.name = name;
|
||||
}
|
||||
|
||||
rename(name) {
|
||||
this.name = name;
|
||||
}
|
||||
|
||||
select(spec) {
|
||||
const { mode } = spec;
|
||||
if (mode === undefined) {
|
||||
@@ -436,7 +466,7 @@ class ImmutableEnumDimension extends ImmutableScalarDimension {
|
||||
const { values } = spec;
|
||||
return super.selectExact({
|
||||
mode: spec.mode,
|
||||
values: values.map(v => lowerBound(enumIndex, v, 0, enumIndex.length))
|
||||
values: values.map(v => binarySearch(enumIndex, v, 0, enumIndex.length))
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -413,3 +413,16 @@ export function upperBoundIndirect(valueArray, indexArray, value, first, last) {
|
||||
}
|
||||
return upperBoundNonFloatIndirect(valueArray, indexArray, value, first, last);
|
||||
}
|
||||
|
||||
// Search for `value` in the sorted array `arr`, in the range [first, last).
|
||||
// Return the first index where arr[index] == value, OR if value not present,
|
||||
// return `last`
|
||||
//
|
||||
// The same semantics/behavior as:
|
||||
// C++: binary_search()
|
||||
//
|
||||
export function binarySearch(valueArray, value, first, last) {
|
||||
const index = lowerBound(valueArray, value, first, last);
|
||||
if (index !== last && value === valueArray[index]) return index;
|
||||
return last;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user