diff --git a/paimon-core/src/main/java/org/apache/paimon/table/source/TopNDataSplitEvaluator.java b/paimon-core/src/main/java/org/apache/paimon/table/source/TopNDataSplitEvaluator.java index d6bf3d32153e..c67c05f094ab 100644 --- a/paimon-core/src/main/java/org/apache/paimon/table/source/TopNDataSplitEvaluator.java +++ b/paimon-core/src/main/java/org/apache/paimon/table/source/TopNDataSplitEvaluator.java @@ -90,6 +90,14 @@ private List getTopNSplits(SortValue order, int limit, List splits return results; } + /** + * Orders splits by their best row under the query's sort order and keeps the first {@code + * limit} ones. In the NULLS LAST branches a split whose sort column is provably all null + * ({@link RichSplit#allNull}, i.e. the null count equals the row count) is the worst candidate + * and sorts last. A null min/max that is not provably all-null means the bound is unknown — for + * example files written with {@code stats.mode=counts} — and {@link #ascCompare}/{@link + * #descCompare} order such a split first so it is read, conservatively. + */ private List pickTopNSplits( List splits, DataType fieldType, @@ -107,9 +115,12 @@ private List pickTopNSplits( result = ascCompare(fieldType, x.min, y.min); } } else { - result = ascCompare(fieldType, x.min, y.min); + result = Boolean.compare(x.allNull, y.allNull); if (result == 0) { - result = nullsLastCompare(x.nullCount, y.nullCount); + result = ascCompare(fieldType, x.min, y.min); + if (result == 0) { + result = nullsLastCompare(x.nullCount, y.nullCount); + } } } return result; @@ -124,9 +135,12 @@ private List pickTopNSplits( result = descCompare(fieldType, x.max, y.max); } } else { - result = descCompare(fieldType, x.max, y.max); + result = Boolean.compare(x.allNull, y.allNull); if (result == 0) { - result = nullsLastCompare(x.nullCount, y.nullCount); + result = descCompare(fieldType, x.max, y.max); + if (result == 0) { + result = nullsLastCompare(x.nullCount, y.nullCount); + } } } return result; @@ -141,7 +155,7 @@ private List pickTopNSplits( private int nullsFirstCompare(Long left, Long right) { if (left == null) { - return -1; + return right == null ? 0 : -1; } else if (right == null) { return 1; } else { @@ -151,7 +165,7 @@ private int nullsFirstCompare(Long left, Long right) { private int nullsLastCompare(Long left, Long right) { if (left == null) { - return -1; + return right == null ? 0 : -1; } else if (right == null) { return 1; } else { @@ -161,7 +175,7 @@ private int nullsLastCompare(Long left, Long right) { private int ascCompare(DataType type, Object left, Object right) { if (left == null) { - return -1; + return right == null ? 0 : -1; } else if (right == null) { return 1; } else { @@ -171,7 +185,7 @@ private int ascCompare(DataType type, Object left, Object right) { private int descCompare(DataType type, Object left, Object right) { if (left == null) { - return -1; + return right == null ? 0 : -1; } else if (right == null) { return 1; } else { @@ -191,12 +205,14 @@ private static class RichSplit { private final Object min; private final Object max; private final Long nullCount; + private final boolean allNull; private RichSplit(DataSplit split, Object min, Object max, Long nullCount) { this.split = split; this.min = min; this.max = max; this.nullCount = nullCount; + this.allNull = nullCount != null && nullCount.longValue() == split.rowCount(); } private DataSplit split() { diff --git a/paimon-core/src/test/java/org/apache/paimon/table/source/TableScanTest.java b/paimon-core/src/test/java/org/apache/paimon/table/source/TableScanTest.java index a60893aa6543..6d706d5b9ac5 100644 --- a/paimon-core/src/test/java/org/apache/paimon/table/source/TableScanTest.java +++ b/paimon-core/src/test/java/org/apache/paimon/table/source/TableScanTest.java @@ -869,6 +869,92 @@ public void testPushDownTopNKeepsWideDeletionVectorSplits() throws Exception { assertThat(result).containsExactly(wideSplit, tightLowSplit); } + @Test + public void testPushDownTopNNullsLastSortsAllNullSplitLast() throws Exception { + createAppendOnlyTable(); + + DataField field = table.schema().fields().get(1); + FieldRef ref = new FieldRef(1, field.name(), field.type()); + + // split whose sort column is entirely null; with NULLS LAST it is the worst TopN + // candidate and must not take a limit slot from a split with real values + DataSplit allNullSplit = newAllNullTestSplit("all-null", 5); + DataSplit realSplit = newTestSplit("real", 10, 19, null); + + TopN ascTopN = new TopN(ref, ASCENDING, NULLS_LAST, 1); + List ascResult = + new TopNDataSplitEvaluator(table.schema(), table.schemaManager()) + .evaluate( + ascTopN.orders().get(0), + ascTopN.limit(), + Arrays.asList(allNullSplit, realSplit)); + assertThat(ascResult).containsExactly(realSplit); + + DataSplit realMaxSplit = newTestSplit("real-max", 100, 109, null); + TopN descTopN = new TopN(ref, DESCENDING, NULLS_LAST, 1); + List descResult = + new TopNDataSplitEvaluator(table.schema(), table.schemaManager()) + .evaluate( + descTopN.orders().get(0), + descTopN.limit(), + Arrays.asList(allNullSplit, realMaxSplit)); + assertThat(descResult).containsExactly(realMaxSplit); + + // sanity: NULLS FIRST keeps treating the null-containing split as the best candidate + TopN nullsFirstTopN = new TopN(ref, DESCENDING, NULLS_FIRST, 1); + List nullsFirstResult = + new TopNDataSplitEvaluator(table.schema(), table.schemaManager()) + .evaluate( + nullsFirstTopN.orders().get(0), + nullsFirstTopN.limit(), + Arrays.asList(allNullSplit, realMaxSplit)); + assertThat(nullsFirstResult).containsExactly(allNullSplit); + + // two all-null splits tie on every key and both are kept: the comparator returns 0 for + // equal keys rather than a nonzero value. + DataSplit allNullSplit2 = newAllNullTestSplit("all-null-2", 5); + TopN nullsFirstTopN2 = new TopN(ref, ASCENDING, NULLS_FIRST, 2); + List tiedResult = + new TopNDataSplitEvaluator(table.schema(), table.schemaManager()) + .evaluate( + nullsFirstTopN2.orders().get(0), + nullsFirstTopN2.limit(), + Arrays.asList(allNullSplit, allNullSplit2)); + assertThat(tiedResult).containsExactlyInAnyOrder(allNullSplit, allNullSplit2); + } + + @Test + public void testPushDownTopNNullsLastKeepsStatsUnknownSplitFirst() throws Exception { + createAppendOnlyTable(); + + DataField field = table.schema().fields().get(1); + FieldRef ref = new FieldRef(1, field.name(), field.type()); + + // stats-mode=counts-like split: min/max unknown (null) but only 2 of 5 rows are null, + // so it is NOT provably all-null and must stay ahead of splits with known bounds + DataSplit unknownSplit = newTestSplitWithField1Stats("unknown", null, null, 2L, 5); + DataSplit realSplit = newTestSplit("real", 10, 19, null); + DataSplit realMaxSplit = newTestSplit("real-max", 100, 109, null); + + TopN ascTopN = new TopN(ref, ASCENDING, NULLS_LAST, 1); + List ascResult = + new TopNDataSplitEvaluator(table.schema(), table.schemaManager()) + .evaluate( + ascTopN.orders().get(0), + ascTopN.limit(), + Arrays.asList(unknownSplit, realSplit)); + assertThat(ascResult).containsExactly(unknownSplit); + + TopN descTopN = new TopN(ref, DESCENDING, NULLS_LAST, 1); + List descResult = + new TopNDataSplitEvaluator(table.schema(), table.schemaManager()) + .evaluate( + descTopN.orders().get(0), + descTopN.limit(), + Arrays.asList(unknownSplit, realMaxSplit)); + assertThat(descResult).containsExactly(unknownSplit); + } + @Test public void testPushDownTopNSchemaEvolution() throws Exception { createAppendOnlyTable(); @@ -995,6 +1081,56 @@ private DataSplit newTestSplit( return builder.build(); } + private DataSplit newAllNullTestSplit(String name, int rowCount) { + return newTestSplitWithField1Stats(name, null, null, (long) rowCount, rowCount); + } + + private DataSplit newTestSplitWithField1Stats( + String name, Integer minValue, Integer maxValue, Long nullCount, int rowCount) { + DataFileMeta file = + DataFileMeta.forAppend( + name, + 0, + (long) rowCount, + new SimpleStats( + newStatsRowOptionalField1(0, minValue, 0L), + newStatsRowOptionalField1(0, maxValue, 0L), + fromLongArray(new Long[] {0L, nullCount, 0L})), + 0, + 0, + table.schema().id(), + Collections.emptyList(), + null, + FileSource.APPEND, + null, + null, + null, + null); + + return DataSplit.builder() + .withSnapshot(1) + .withPartition(BinaryRow.EMPTY_ROW) + .withBucket(0) + .withBucketPath("dummy") + .rawConvertible(true) + .withDataFiles(Collections.singletonList(file)) + .build(); + } + + private BinaryRow newStatsRowOptionalField1(int pt, Integer a, long c) { + BinaryRow row = new BinaryRow(3); + BinaryRowWriter writer = new BinaryRowWriter(row); + writer.writeInt(0, pt); + if (a == null) { + writer.setNullAt(1); + } else { + writer.writeInt(1, a); + } + writer.writeLong(2, c); + writer.complete(); + return row; + } + private BinaryRow newStatsRow(int pt, int a, long b) { BinaryRow row = new BinaryRow(3); BinaryRowWriter writer = new BinaryRowWriter(row);