Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
81 changes: 78 additions & 3 deletions guava-tests/test/com/google/common/math/QuantilesTest.java
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,7 @@
import com.google.common.truth.Correspondence;
import com.google.common.truth.Correspondence.BinaryPredicate;
import java.util.ArrayList;
import java.util.Arrays;
import java.util.List;
import java.util.Random;
import junit.framework.TestCase;
Expand All @@ -61,11 +62,12 @@ public class QuantilesTest extends TestCase {
/*
* Since Quantiles provides a fluent-style API, each test covers a chain of methods resulting in
* the computation of one or more quantiles (or in an error) rather than individual methods. The
* tests are divided into three sections:
* tests are divided into five sections:
* 1. Tests on a hardcoded dataset for chains starting with median(), quartiles(), and scale(10);
* 2. Tests on hardcoded datasets include non-finite values for chains starting with scale(10);
* 3. Tests on a mechanically generated dataset for chains starting with percentiles();
* 4. Tests of illegal usages of the API.
* 4. Tests on mechanically generated datasets with many duplicate values;
* 5. Tests of illegal usages of the API.
*/

/*
Expand Down Expand Up @@ -600,7 +602,80 @@ public void testPercentiles_indexes_varargsAll_computeInPlace() {
assertThat(dataset).usingExactEquality().containsExactlyElementsIn(PSEUDORANDOM_DATASET);
}

// 4. Tests of illegal usages of the API:
// 4. Tests on datasets with many duplicate values:

// These datasets are large enough that a selection algorithm which degenerates on duplicate
// values (as ours did until https://github.com/google/guava/issues/3789) takes minutes rather
// than milliseconds to work through them.
private static final int DUPLICATE_HEAVY_DATASET_SIZE = 100_000;

public void testPercentiles_index_computeInPlace_allValuesEqual() {
for (int index = 0; index <= 100; index++) {
double[] dataset = new double[DUPLICATE_HEAVY_DATASET_SIZE];
Arrays.fill(dataset, 7.5);
assertWithMessage("quantile at index %s", index)
.that(percentiles().index(index).computeInPlace(dataset))
.isWithin(ALLOWED_ERROR)
.of(7.5);
}
}

public void testPercentiles_index_computeInPlace_fewDistinctValues() {
for (int index = 0; index <= 100; index++) {
double[] dataset = fewDistinctValuesDataset();
double[] sorted = dataset.clone();
Arrays.sort(sorted);
assertWithMessage("quantile at index %s", index)
.that(percentiles().index(index).computeInPlace(dataset))
.isWithin(ALLOWED_ERROR)
.of(expectedPercentileFromSorted(index, sorted));
}
}

public void testPercentiles_indexes_computeInPlace_fewDistinctValues() {
double[] dataset = fewDistinctValuesDataset();
double[] sorted = dataset.clone();
Arrays.sort(sorted);
List<Integer> indexes = new ArrayList<>();
ImmutableMap.Builder<Integer, Double> expectedBuilder = ImmutableMap.builder();
for (int index = 0; index <= 100; index++) {
indexes.add(index);
expectedBuilder.put(index, expectedPercentileFromSorted(index, sorted));
}
assertThat(percentiles().indexes(Ints.toArray(indexes)).computeInPlace(dataset))
.comparingValuesUsing(QUANTILE_CORRESPONDENCE)
.containsExactlyEntriesIn(expectedBuilder.buildOrThrow());

// Assert that the dataset still contains the same elements, although reordered. (We compare
// sorted copies directly rather than using containsExactlyElementsIn, which would build
// collections of 100,000 boxed doubles.)
double[] datasetSorted = dataset.clone();
Arrays.sort(datasetSorted);
assertWithMessage("dataset contains the same elements after computing in place")
.that(Arrays.equals(datasetSorted, sorted))
.isTrue();
}

/** Returns a pseudorandomly generated dataset drawn from only five distinct values. */
private static double[] fewDistinctValuesDataset() {
double[] dataset = new double[DUPLICATE_HEAVY_DATASET_SIZE];
Random random = new Random(4088805267152079440L);
for (int i = 0; i < dataset.length; i++) {
dataset[i] = random.nextInt(5);
}
return dataset;
}

private static double expectedPercentileFromSorted(int index, double[] sorted) {
double position = (double) index * (sorted.length - 1) / 100.0;
int positionFloor = (int) Math.floor(position);
int positionCeil = (int) Math.ceil(position);
double lowerValue = sorted[positionFloor];
double upperValue = sorted[positionCeil];
return lowerValue + (position - positionFloor) * (upperValue - lowerValue);
}

// 5. Tests of illegal usages of the API:

private static final ImmutableList<Double> EMPTY_DATASET = ImmutableList.of();

Expand Down
68 changes: 68 additions & 0 deletions guava/src/com/google/common/math/Quantiles.java
Original file line number Diff line number Diff line change
Expand Up @@ -552,7 +552,30 @@ private static void selectInPlace(int required, double[] array, int from, int to

// Let's play quickselect! We'll repeatedly partition the range [from, to] containing the
// required element, as long as it has more than one element.
//
// partition() puts values equal to the pivot on the low side, so a dataset with many
// duplicates can shrink the range by only one element per iteration, which is quadratic. To
// stay linear-ish on such datasets without slowing down the common case, we give the plain
// partition a budget of iterations which is comfortably more than well-behaved data needs, and
// fall back to a three-way partition once that budget runs out.
int budget = 2 * IntMath.log2(to - from + 1, RoundingMode.UP) + 4;
while (to > from) {
if (--budget < 0) {
// We're not making progress, so duplicates are likely. partitionAroundEqualValues() gathers
// every value equal to the pivot into one run, which we can then exclude in its entirety.
long equalValues = partitionAroundEqualValues(array, from, to);
int firstEqual = (int) (equalValues >>> Integer.SIZE);
int lastEqual = (int) equalValues;
if (required >= firstEqual && required <= lastEqual) {
return; // The required index holds a value equal to the pivot, so it's already in place.
}
if (required < firstEqual) {
to = firstEqual - 1;
} else {
from = lastEqual + 1;
}
continue;
}
int partitionPoint = partition(array, from, to);
if (partitionPoint >= required) {
to = partitionPoint - 1;
Expand Down Expand Up @@ -593,6 +616,51 @@ private static int partition(double[] array, int from, int to) {
return partitionPoint;
}

/**
* Performs a three-way partition operation on the slice of {@code array} with elements in the
* range [{@code from}, {@code to}], using the same pivot selection as {@link #partition}. Unlike
* that method, this one groups <i>all</i> the values equal to the pivot, not just one of them,
* into a single run: if it returns {@code first} and {@code last} then we know that the values
* with indexes in [{@code from}, {@code first}) are less than the pivot, the values with indexes
* in [{@code first}, {@code last}] are equal to it, and the values with indexes in ({@code last},
* {@code to}] are greater than it. Every value in that run is therefore already at an index it
* would occupy in the sorted dataset.
*
* <p>Returns those two indexes packed into a single {@code long}, with {@code first} in the high
* 32 bits and {@code last} in the low 32 bits, to avoid allocating.
*/
private static long partitionAroundEqualValues(double[] array, int from, int to) {
// Select a pivot, and move it to the start of the slice i.e. to index from.
movePivotToStartOfSlice(array, from, to);
double pivot = array[from];

// Sweep the slice, maintaining the invariant that the values with indexes in [from, first) are
// less than the pivot, those in [first, index) are equal to it, and those in (last, to] are
// greater than it. The values in [index, last] have not been examined yet.
int first = from;
int last = to;
int index = from;
while (index <= last) {
double value = array[index];
if (value < pivot) {
// While first is still tracking index, this swap would be a no-op, so skip it.
if (first != index) {
swap(array, first, index);
}
first++;
index++;
} else if (value > pivot) {
// The value swapped into index hasn't been examined yet, so don't advance index.
swap(array, index, last);
last--;
} else {
index++;
}
}

return ((long) first << Integer.SIZE) | (last & 0xffffffffL);
}

/**
* Selects the pivot to use, namely the median of the values at {@code from}, {@code to}, and
* halfway between the two (rounded down), from {@code array}, and ensure (by swapping elements if
Expand Down
Loading