Performance work (#334)

* range encode filter range lists

* speed up data load

* add comment on scanpy read params

* update to latest scanpy/anndata

* performance improvments in data loading

* fix typo

* work around scanpy bug

* remove debugging print statements
This commit is contained in:
Bruce Martin
2018-10-16 15:49:40 -07:00
committed by GitHub
parent 6b0aafd27c
commit 76ec29734a
5 changed files with 142 additions and 11 deletions
@@ -0,0 +1,60 @@
import { rangeEncodeIndices } from "../../src/util/actionHelpers";
describe("rangeEncodeIndices", () => {
test("small array edge cases", () => {
expect(rangeEncodeIndices([])).toMatchObject([]);
expect(rangeEncodeIndices([1])).toMatchObject([1]);
expect(rangeEncodeIndices([1, 99])).toMatchObject([1, 99]);
expect(rangeEncodeIndices([99, 1])).toMatchObject([1, 99]);
expect(rangeEncodeIndices([1, 100, 4])).toMatchObject([1, 4, 100]);
});
test("sorted flag", () => {
expect(rangeEncodeIndices([1, 9, 432], 10, true)).toMatchObject([
1,
9,
432
]);
expect(rangeEncodeIndices([1, 9, 432], 10, false)).toMatchObject([
1,
9,
432
]);
expect(rangeEncodeIndices([0, 1, 2, 3, 9, 10, 432], 2, true)).toMatchObject(
[[0, 3], [9, 10], 432]
);
expect(
rangeEncodeIndices([0, 1, 2, 3, 9, 10, 432], 2, false)
).toMatchObject([[0, 3], [9, 10], 432]);
});
test("begin or end edge cases", () => {
expect(
rangeEncodeIndices([3, 4, 5, 6, 7, 10, 11, 12, 99], 3, false)
).toMatchObject([[3, 7], [10, 12], 99]);
expect(
rangeEncodeIndices([3, 4, 5, 6, 7, 10, 11, 12], 3, false)
).toMatchObject([[3, 7], [10, 12]]);
expect(
rangeEncodeIndices([0, 3, 4, 5, 6, 7, 10, 11, 12], 3, false)
).toMatchObject([0, [3, 7], [10, 12]]);
expect(
rangeEncodeIndices([0, 3, 4, 5, 6, 7, 10, 11, 12, 99], 3, false)
).toMatchObject([0, [3, 7], [10, 12], 99]);
});
test("minRangeLength", () => {
expect(
rangeEncodeIndices([3, 4, 5, 6, 7, 10, 11, 12, 99], 4, false)
).toMatchObject([[3, 7], 10, 11, 12, 99]);
expect(
rangeEncodeIndices([3, 4, 5, 6, 7, 10, 11, 12], 4, false)
).toMatchObject([[3, 7], 10, 11, 12]);
expect(
rangeEncodeIndices([0, 3, 4, 5, 6, 7, 10, 11, 12], 4, false)
).toMatchObject([0, [3, 7], 10, 11, 12]);
expect(
rangeEncodeIndices([0, 3, 4, 5, 6, 7, 10, 11, 12, 99], 4, false)
).toMatchObject([0, [3, 7], 10, 11, 12, 99]);
});
});
+17 -3
View File
@@ -2,8 +2,18 @@
import _ from "lodash"; import _ from "lodash";
import * as globals from "../globals"; import * as globals from "../globals";
import { Universe, kvCache } from "../util/stateManager"; import { Universe, kvCache } from "../util/stateManager";
import { catchErrorsWrap, doJsonRequest } from "../util/actionHelpers"; import {
catchErrorsWrap,
doJsonRequest,
rangeEncodeIndices
} from "../util/actionHelpers";
/*
Bootstrap application with the initial data loading.
* /config - application configuration
* /schema - schema of dataframe
* /annotations/obs - all metadata annotation
*/
const doInitialDataLoad = () => const doInitialDataLoad = () =>
catchErrorsWrap(async dispatch => { catchErrorsWrap(async dispatch => {
dispatch({ type: "initial data load start" }); dispatch({ type: "initial data load start" });
@@ -235,8 +245,12 @@ const requestDifferentialExpression = (set1, set2, num_genes = 10) => async (
*/ */
const state = getState(); const state = getState();
const { universe } = state.controls; const { universe } = state.controls;
const set1ByIndex = _.map(set1, s => universe.obsNameToIndexMap[s]); const set1ByIndex = rangeEncodeIndices(
const set2ByIndex = _.map(set2, s => universe.obsNameToIndexMap[s]); _.map(set1, s => universe.obsNameToIndexMap[s])
);
const set2ByIndex = rangeEncodeIndices(
_.map(set2, s => universe.obsNameToIndexMap[s])
);
const diffExpFetch = await fetch( const diffExpFetch = await fetch(
`${globals.API.prefix}${globals.API.version}diffexp/obs`, `${globals.API.prefix}${globals.API.version}diffexp/obs`,
{ {
+57 -4
View File
@@ -1,3 +1,5 @@
import _ from "lodash";
/* /*
Catch unexpected errors and make sure we don't lose them! Catch unexpected errors and make sure we don't lose them!
*/ */
@@ -11,10 +13,7 @@ export function catchErrorsWrap(fn) {
} }
/* /*
Bootstrap application with the initial data loading. Wrapper to perform an async fetch and JSON decode response.
* /config - application configuration
* /schema - schema of dataframe
* /annotations/obs - all metadata annotation
*/ */
export const doJsonRequest = async url => { export const doJsonRequest = async url => {
const res = await fetch(url, { const res = await fetch(url, {
@@ -26,3 +25,57 @@ export const doJsonRequest = async url => {
}); });
return res.json(); return res.json();
}; };
/*
This function "packs" filter index lists into the more efficient
"range" form specified in the REST 0.2 spec.
Specifically, it turns an array of indices [0, 1, 2, 10, 11, 14, ...]
into a form that encodes runs of consecutive numbers as [min, max].
Array may not be sorted, but will only contain uniq values.
Parameters:
indices - input array of numbers (index)
minRangeLength - hint, min range length before it is encoded into range format.
sorted - boolean hint indicating array is presorted, ascending order
So [1, 2, 3, 4, 10, 11, 14] -> [ [1, 4], [10, 11], 14]
*/
export const rangeEncodeIndices = (
indices,
minRangeLength = 3,
sorted = false
) => {
if (indices.length === 0) {
return indices;
}
if (!sorted) {
indices = _.sortBy(indices);
}
const result = new Array(indices.length);
let resultTail = 0;
let i = 0;
while (i < indices.length) {
const begin = indices[i];
let current;
do {
current = indices[i];
i += 1;
} while (i < indices.length && indices[i] === current + 1);
if (current - begin + 1 >= minRangeLength) {
result[resultTail] = [begin, current];
resultTail += 1;
} else {
for (let j = begin; j <= current; j += 1, resultTail += 1) {
result[resultTail] = j;
}
}
}
result.length = resultTail;
return result;
};
+6 -2
View File
@@ -30,7 +30,7 @@ class ScanpyEngine(CXGDriver):
self.cell_count = self.data.shape[0] self.cell_count = self.data.shape[0]
self.gene_count = self.data.shape[1] self.gene_count = self.data.shape[1]
self._create_schema() self._create_schema()
self.layout(None) self.layout({})
def _create_schema(self): def _create_schema(self):
self.schema = { self.schema = {
@@ -78,7 +78,11 @@ class ScanpyEngine(CXGDriver):
@staticmethod @staticmethod
def _load_data(data): def _load_data(data):
return sc.read(os.path.join(data, "data.h5ad")) # See https://scanpy.readthedocs.io/en/latest/api/scanpy.api.read.html
# Based upon this advice, setting cache=True parameter
# Note: as of current scanpy/anndata release, setting backed='r' will
# result in an error.
return sc.read(os.path.join(data, "data.h5ad"), cache=True)
@staticmethod @staticmethod
def _top_sort(values, sort_order, top_n=None): def _top_sort(values, sort_order, top_n=None):
+2 -2
View File
@@ -1,4 +1,4 @@
anndata==0.6.10 anndata==0.6.11
Flask==0.12.4 Flask==0.12.4
Flask-Caching==1.4.0 Flask-Caching==1.4.0
Flask-Compress==1.4.0 Flask-Compress==1.4.0
@@ -7,5 +7,5 @@ Flask-RESTful==0.3.6
flask-restful-swagger-2==0.35 flask-restful-swagger-2==0.35
numpy==1.14.5 numpy==1.14.5
pandas==0.23.1 pandas==0.23.1
scanpy==1.3.1 scanpy==1.3.2
scipy==1.1.0 scipy==1.1.0