From 86af449e899cfb3d22333f3c4006634c65df3316 Mon Sep 17 00:00:00 2001 From: JingsongLi Date: Thu, 1 Oct 2026 23:10:30 +0800 Subject: [PATCH 1/4] [core] Support composite BTree global indexes for data evolution tables --- .../multimodal-table/global-index/btree.mdx | 38 ++ .../globalindex/CompositeKeySerializer.java | 103 ++++++ .../globalindex/GlobalIndexEvaluator.java | 18 + .../globalindex/GlobalIndexKeyExtractor.java | 2 +- .../paimon/globalindex/GlobalIndexReader.java | 6 + .../paimon/globalindex/KeySerializer.java | 6 + .../globalindex/SortedGlobalIndexer.java | 5 +- .../globalindex/btree/BTreeGlobalIndexer.java | 22 +- .../btree/BTreeGlobalIndexerFactory.java | 37 ++ .../globalindex/btree/BTreeIndexWriter.java | 15 +- .../btree/CompositeBTreePredicate.java | 88 +++++ .../btree/LazyFilteredBTreeReader.java | 15 + .../btree/CompositeBTreeIndexTest.java | 215 +++++++++++ .../globalindex/DataEvolutionBatchScan.java | 16 +- .../DataEvolutionGlobalIndexCoverage.java | 15 +- .../DataEvolutionGlobalIndexScanner.java | 77 +++- .../paimon/globalindex/GlobalIndexQuery.java | 154 ++++++-- .../sorted/SortedGlobalIndexScanner.java | 37 +- .../sorted/SortedGlobalIndexWriter.java | 39 +- .../manifest/IndexManifestFileHandler.java | 8 + .../globalindex/GlobalIndexQueryTest.java | 23 +- .../sorted/SortedGlobalIndexTestUtils.java | 44 ++- .../paimon/table/CompositeBTreeTableTest.java | 338 ++++++++++++++++++ .../globalindex/SortedIndexTopoBuilder.java | 101 +++++- .../procedure/CreateGlobalIndexProcedure.java | 2 +- .../paimon/flink/SortedGlobalIndexITCase.java | 36 ++ .../data_evolution_global_index_scanner.py | 3 + .../global_index_scalar_search_mode_test.py | 9 + .../sorted/SortedIndexTopoBuilder.java | 56 ++- .../CompositeBTreeIndexProcedureTest.scala | 93 +++++ 30 files changed, 1489 insertions(+), 132 deletions(-) create mode 100644 paimon-common/src/main/java/org/apache/paimon/globalindex/CompositeKeySerializer.java create mode 100644 paimon-common/src/main/java/org/apache/paimon/globalindex/btree/CompositeBTreePredicate.java create mode 100644 paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java create mode 100644 paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java create mode 100644 paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala diff --git a/docs/docs/multimodal-table/global-index/btree.mdx b/docs/docs/multimodal-table/global-index/btree.mdx index d73fdbfc9b0f..eb9955d97423 100644 --- a/docs/docs/multimodal-table/global-index/btree.mdx +++ b/docs/docs/multimodal-table/global-index/btree.mdx @@ -145,6 +145,44 @@ print(pa_table) +## Composite BTree Indexes + +Spark and Flink can build a composite index by listing columns in key order: + +```sql +CALL sys.create_global_index( + table => 'db.my_table', + index_column => 'category,item_number', + index_type => 'btree' +); + +SELECT * FROM my_table +WHERE category = 'category-a' + AND item_number = 107; +``` + +The index stores typed tuples and one row-ID posting list for each distinct tuple. +An AND containing equality conditions for every indexed column uses a single +composite point lookup, regardless of the SQL condition order. This avoids +materializing the individual columns' posting lists before intersecting them. +Additional predicates remain filters, and OR branches can each use their own +composite lookup. When several composite indexes match, the query prefers the +one with the most columns. + +Composite indexes can coexist with single-column indexes on the same columns. +The current composite query path requires equality conditions for all key +columns; prefix queries, ranges, and conditions on only some key columns use +available single-column indexes or ordinary table scans. + +Coverage is determined by the selected index. In `full` and `detail` modes, rows +outside the composite index's coverage are scanned even when single-column +indexes cover those rows. Incremental builds fill missing ranges; under the +`IGNORE` column-update policy, updates to any key column refresh the affected +composite index ranges. Dropping an index requires the same ordered column list +used to create it. PyPaimon skips composite index files and uses available +single-column indexes or ordinary table scans; composite index building is not +supported in PyPaimon. + ## BTree Options | Option | Default | Description | diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/CompositeKeySerializer.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/CompositeKeySerializer.java new file mode 100644 index 000000000000..3e8e602db14f --- /dev/null +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/CompositeKeySerializer.java @@ -0,0 +1,103 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.paimon.globalindex; + +import org.apache.paimon.data.GenericRow; +import org.apache.paimon.data.InternalRow; +import org.apache.paimon.memory.MemorySlice; +import org.apache.paimon.memory.MemorySliceInput; +import org.apache.paimon.memory.MemorySliceOutput; +import org.apache.paimon.types.RowType; + +import java.util.Comparator; + +/** Length-delimited tuple keys, compared by their typed components with nulls first. */ +public class CompositeKeySerializer implements KeySerializer { + + private final KeySerializer[] serializers; + private final InternalRow.FieldGetter[] getters; + private final Comparator[] comparators; + + @SuppressWarnings("unchecked") + public CompositeKeySerializer(RowType type) { + int count = type.getFieldCount(); + serializers = new KeySerializer[count]; + getters = new InternalRow.FieldGetter[count]; + comparators = new Comparator[count]; + for (int i = 0; i < count; i++) { + serializers[i] = KeySerializer.create(type.getTypeAt(i)); + getters[i] = InternalRow.createFieldGetter(type.getTypeAt(i), i); + comparators[i] = serializers[i].createComparator(); + } + } + + @Override + public byte[] serialize(Object key) { + InternalRow row = (InternalRow) key; + MemorySliceOutput output = new MemorySliceOutput(32); + for (int i = 0; i < serializers.length; i++) { + Object value = getters[i].getFieldOrNull(row); + if (value == null) { + output.writeInt(-1); + } else { + byte[] bytes = serializers[i].serialize(value); + output.writeInt(bytes.length); + output.writeBytes(bytes); + } + } + return output.toSlice().copyBytes(); + } + + @Override + public Object deserialize(MemorySlice data) { + MemorySliceInput input = data.toInput(); + GenericRow row = new GenericRow(serializers.length); + for (int i = 0; i < serializers.length; i++) { + int length = input.readInt(); + if (length >= 0) { + row.setField(i, serializers[i].deserialize(input.readSlice(length))); + } else if (length != -1) { + throw new IllegalArgumentException( + "Invalid composite key component length: " + length); + } + } + if (input.available() != 0) { + throw new IllegalArgumentException("Trailing bytes in composite key"); + } + return row; + } + + @Override + public Comparator createComparator() { + return (left, right) -> { + for (int i = 0; i < serializers.length; i++) { + Object a = getters[i].getFieldOrNull((InternalRow) left); + Object b = getters[i].getFieldOrNull((InternalRow) right); + int comparison = + a == null + ? (b == null ? 0 : -1) + : b == null ? 1 : comparators[i].compare(a, b); + if (comparison != 0) { + return comparison; + } + } + return 0; + }; + } +} diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexEvaluator.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexEvaluator.java index b772cc7edbe9..22224a83eecf 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexEvaluator.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexEvaluator.java @@ -37,6 +37,7 @@ import org.apache.paimon.predicate.TopN; import org.apache.paimon.types.RowType; import org.apache.paimon.utils.IOUtils; +import org.apache.paimon.utils.Range; import javax.annotation.Nullable; @@ -306,9 +307,21 @@ public static final class Evaluation { private final GlobalIndexResult result; private final Set contributingFieldIds; + @Nullable private final List coveredRanges; Evaluation(GlobalIndexResult result, Collection contributingFieldIds) { + this(result, contributingFieldIds, null); + } + + Evaluation( + GlobalIndexResult result, + Collection contributingFieldIds, + @Nullable List coveredRanges) { this.result = result; + this.coveredRanges = + coveredRanges == null + ? null + : Collections.unmodifiableList(new ArrayList<>(coveredRanges)); this.contributingFieldIds = Collections.unmodifiableSet(new HashSet<>(contributingFieldIds)); } @@ -317,6 +330,11 @@ public GlobalIndexResult result() { return result; } + @Nullable + public List coveredRanges() { + return coveredRanges; + } + public Set contributingFieldIds() { return contributingFieldIds; } diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexKeyExtractor.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexKeyExtractor.java index 9a0881fa1371..d2d036e20da2 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexKeyExtractor.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexKeyExtractor.java @@ -25,7 +25,7 @@ import java.io.IOException; import java.io.Serializable; -/** Extracts zero or more normalized index keys from one source-column value. */ +/** Extracts zero or more normalized index keys from one source value or projected tuple. */ public interface GlobalIndexKeyExtractor extends Serializable { /** Type of the normalized keys emitted by {@link #extract(Object, KeyConsumer)}. */ diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexReader.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexReader.java index f5d6ab9fad65..4506713b9176 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexReader.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexReader.java @@ -36,6 +36,12 @@ public interface GlobalIndexReader extends FunctionVisitor>>, Closeable { + /** Point lookup of a full composite key, with literals in index column order. */ + default CompletableFuture> visitCompositeEqual( + List literals) { + return CompletableFuture.completedFuture(Optional.empty()); + } + @Override default CompletableFuture> visitIsNaN(FieldRef fieldRef) { return CompletableFuture.completedFuture(Optional.empty()); diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/KeySerializer.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/KeySerializer.java index a58b573d9de0..a33283d1197d 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/KeySerializer.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/KeySerializer.java @@ -35,6 +35,7 @@ import org.apache.paimon.types.FloatType; import org.apache.paimon.types.IntType; import org.apache.paimon.types.LocalZonedTimestampType; +import org.apache.paimon.types.RowType; import org.apache.paimon.types.SmallIntType; import org.apache.paimon.types.TimeType; import org.apache.paimon.types.TimestampType; @@ -64,6 +65,11 @@ public KeySerializer defaultMethod(DataType dataType) { "DataType: " + dataType + " is not supported by global index now."); } + @Override + public KeySerializer visit(RowType rowType) { + return new CompositeKeySerializer(rowType); + } + @Override public KeySerializer visit(CharType charType) { return new StringSerializer(); diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/SortedGlobalIndexer.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/SortedGlobalIndexer.java index 1f3bc4e6034d..f706060a0955 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/SortedGlobalIndexer.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/SortedGlobalIndexer.java @@ -21,6 +21,9 @@ /** A global indexer whose normalized keys must be sorted before they are written. */ public interface SortedGlobalIndexer extends GlobalIndexer { - /** Defines how source-column values are normalized into the keys consumed by the writer. */ + /** + * Defines how source values or projected tuples are normalized into the keys consumed by the + * writer. + */ GlobalIndexKeyExtractor keyExtractor(); } diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexer.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexer.java index 774b22bf1fce..b46e04f86d20 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexer.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexer.java @@ -31,6 +31,8 @@ import org.apache.paimon.io.cache.CacheManager; import org.apache.paimon.options.Options; import org.apache.paimon.types.DataField; +import org.apache.paimon.types.DataType; +import org.apache.paimon.types.RowType; import org.apache.paimon.utils.BloomFilter; import org.apache.paimon.utils.LazyField; import org.apache.paimon.utils.Range; @@ -38,6 +40,8 @@ import javax.annotation.Nullable; import java.io.IOException; +import java.util.ArrayList; +import java.util.Collections; import java.util.List; import java.util.concurrent.ExecutorService; @@ -75,8 +79,22 @@ public class BTreeGlobalIndexer implements SortedGlobalIndexer { private final LazyField cacheManager; public BTreeGlobalIndexer(DataField dataField, Options options) { - this.keySerializer = KeySerializer.create(dataField.type()); - this.keyExtractor = GlobalIndexKeyExtractor.identity(dataField.type()); + this(dataField, Collections.emptyList(), options); + } + + public BTreeGlobalIndexer(DataField dataField, List extraFields, Options options) { + List fields = new ArrayList<>(); + fields.add(dataField); + fields.addAll(extraFields); + for (DataField field : fields) { + if (field.type() instanceof RowType) { + throw new UnsupportedOperationException( + "BTree index columns must have scalar types: " + field.name()); + } + } + DataType keyType = extraFields.isEmpty() ? dataField.type() : new RowType(fields); + this.keySerializer = KeySerializer.create(keyType); + this.keyExtractor = GlobalIndexKeyExtractor.identity(keyType); this.options = options; this.fallbackScanMaxSize = options.get(BTreeIndexOptions.BTREE_INDEX_FALLBACK_SCAN_MAX_SIZE).getBytes(); diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java index 98d5c36241bd..9b5b8572f4c5 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java @@ -18,16 +18,23 @@ package org.apache.paimon.globalindex.btree; +import org.apache.paimon.data.GenericRow; import org.apache.paimon.globalindex.GlobalIndexIOMeta; import org.apache.paimon.globalindex.GlobalIndexer; import org.apache.paimon.globalindex.GlobalIndexerFactory; import org.apache.paimon.globalindex.KeySerializer; import org.apache.paimon.globalindex.SortedFileMetaSelector; import org.apache.paimon.options.Options; +import org.apache.paimon.predicate.FieldRef; +import org.apache.paimon.predicate.LeafPredicate; import org.apache.paimon.predicate.Predicate; import org.apache.paimon.types.DataField; +import org.apache.paimon.types.RowType; +import java.util.ArrayList; +import java.util.Collections; import java.util.List; +import java.util.Optional; /** The {@link GlobalIndexerFactory} for btree index. */ public class BTreeGlobalIndexerFactory implements GlobalIndexerFactory { @@ -45,10 +52,40 @@ public List selectFiles( List extraFields, Predicate predicate, List files) { + if (extraFields != null && !extraFields.isEmpty()) { + List fields = new ArrayList<>(); + fields.add(indexField); + fields.addAll(extraFields); + Optional> matched = + CompositeBTreePredicate.match(fields, predicate); + if (!matched.isPresent()) { + return files; + } + if (CompositeBTreePredicate.isContradictory(fields, predicate)) { + return Collections.emptyList(); + } + if (files.stream().anyMatch(file -> file.metadata() == null)) { + return files; + } + KeySerializer serializer = KeySerializer.create(new RowType(fields)); + Object[] values = matched.get().stream().map(leaf -> leaf.literals().get(0)).toArray(); + return new SortedFileMetaSelector(files, serializer) + .visitEqual( + new FieldRef(0, indexField.name(), new RowType(fields)), + GenericRow.of(values)) + .orElse(files); + } return SortedFileMetaSelector.selectFiles( predicate, files, KeySerializer.create(indexField.type())); } + @Override + public GlobalIndexer create( + DataField indexField, List extraFields, Options options) { + return new BTreeGlobalIndexer( + indexField, extraFields == null ? Collections.emptyList() : extraFields, options); + } + @Override public GlobalIndexer create(DataField dataField, Options options) { return new BTreeGlobalIndexer(dataField, options); diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeIndexWriter.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeIndexWriter.java index fecf600f56ac..238e70df4166 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeIndexWriter.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeIndexWriter.java @@ -20,6 +20,7 @@ import org.apache.paimon.compression.BlockCompressionFactory; import org.apache.paimon.fs.PositionOutputStream; +import org.apache.paimon.globalindex.CompositeKeySerializer; import org.apache.paimon.globalindex.GlobalIndexSingleColumnWriter; import org.apache.paimon.globalindex.KeySerializer; import org.apache.paimon.globalindex.ResultEntry; @@ -148,19 +149,27 @@ public void write(@Nullable Object key, long rowId) { return; } - if (lastKey != null && comparator.compare(key, lastKey) != 0) { + boolean differentKey = lastKey == null || comparator.compare(key, lastKey) != 0; + if (lastKey != null && differentKey) { try { flush(); } catch (IOException e) { throw new RuntimeException("Error in writing btree index files.", e); } } - lastKey = key; + if (differentKey || !(keySerializer instanceof CompositeKeySerializer)) { + // Sorted engine iterators can reuse the row backing a tuple key. + lastKey = + keySerializer instanceof CompositeKeySerializer + ? keySerializer.deserialize( + MemorySlice.wrap(keySerializer.serialize(key))) + : key; + } currentRowIds.add(rowId); // update stats if (firstKey == null) { - firstKey = key; + firstKey = lastKey; } } diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/CompositeBTreePredicate.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/CompositeBTreePredicate.java new file mode 100644 index 000000000000..02b71aec8909 --- /dev/null +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/CompositeBTreePredicate.java @@ -0,0 +1,88 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.paimon.globalindex.btree; + +import org.apache.paimon.globalindex.KeySerializer; +import org.apache.paimon.predicate.Equal; +import org.apache.paimon.predicate.LeafPredicate; +import org.apache.paimon.predicate.Predicate; +import org.apache.paimon.predicate.PredicateBuilder; +import org.apache.paimon.types.DataField; + +import java.util.ArrayList; +import java.util.Comparator; +import java.util.List; +import java.util.Optional; + +/** Matches a full conjunction of equalities in physical composite-key order. */ +public final class CompositeBTreePredicate { + + private CompositeBTreePredicate() {} + + /** SQL equalities with NULL, or distinct values for the same key field, cannot match. */ + public static boolean isContradictory(List fields, Predicate predicate) { + List matched = match(fields, predicate).get(); + for (int i = 0; i < fields.size(); i++) { + Object value = matched.get(i).literals().get(0); + if (value == null) { + return true; + } + Comparator comparator = + KeySerializer.create(fields.get(i).type()).createComparator(); + for (Predicate child : PredicateBuilder.splitAnd(predicate)) { + if (child instanceof LeafPredicate) { + LeafPredicate leaf = (LeafPredicate) child; + if (leaf.function() instanceof Equal + && leaf.fieldRefOptional().equals(matched.get(i).fieldRefOptional())) { + Object other = leaf.literals().get(0); + if (other == null || comparator.compare(value, other) != 0) { + return true; + } + } + } + } + } + return false; + } + + public static Optional> match(List fields, Predicate predicate) { + List conjuncts = PredicateBuilder.splitAnd(predicate); + List matched = new ArrayList<>(); + for (DataField field : fields) { + LeafPredicate equality = null; + for (Predicate conjunct : conjuncts) { + if (conjunct instanceof LeafPredicate) { + LeafPredicate leaf = (LeafPredicate) conjunct; + if (leaf.function() instanceof Equal + && leaf.fieldRefOptional().isPresent() + && field.name().equals(leaf.fieldRefOptional().get().name()) + && field.type().equalsIgnoreNullable(leaf.type())) { + equality = leaf; + break; + } + } + } + if (equality == null) { + return Optional.empty(); + } + matched.add(equality); + } + return Optional.of(matched); + } +} diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/LazyFilteredBTreeReader.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/LazyFilteredBTreeReader.java index 03bbb8a68d6b..2f945f5fbb44 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/LazyFilteredBTreeReader.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/LazyFilteredBTreeReader.java @@ -18,6 +18,8 @@ package org.apache.paimon.globalindex.btree; +import org.apache.paimon.data.GenericRow; +import org.apache.paimon.globalindex.CompositeKeySerializer; import org.apache.paimon.globalindex.GlobalIndexIOMeta; import org.apache.paimon.globalindex.GlobalIndexResult; import org.apache.paimon.globalindex.KeySerializer; @@ -37,6 +39,7 @@ import java.io.IOException; import java.util.Comparator; import java.util.List; +import java.util.Objects; import java.util.Optional; import java.util.concurrent.CompletableFuture; import java.util.concurrent.ExecutorService; @@ -109,6 +112,18 @@ private Pair fullRangeBounds(List files) { return remaining == 0 ? Pair.of(min, max) : null; } + @Override + public CompletableFuture> visitCompositeEqual( + List literals) { + if (!(keySerializer instanceof CompositeKeySerializer)) { + return CompletableFuture.completedFuture(Optional.empty()); + } + if (literals.stream().anyMatch(Objects::isNull)) { + return CompletableFuture.completedFuture(Optional.of(GlobalIndexResult.createEmpty())); + } + return visitEqual((FieldRef) null, GenericRow.of(literals.toArray())); + } + @Override public CompletableFuture> visitEqual( FieldRef fieldRef, Object literal) { diff --git a/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java b/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java new file mode 100644 index 000000000000..55e903ac7488 --- /dev/null +++ b/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java @@ -0,0 +1,215 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.paimon.globalindex.btree; + +import org.apache.paimon.data.BinaryString; +import org.apache.paimon.data.GenericRow; +import org.apache.paimon.fs.Path; +import org.apache.paimon.fs.PositionOutputStream; +import org.apache.paimon.fs.local.LocalFileIO; +import org.apache.paimon.globalindex.GlobalIndexIOMeta; +import org.apache.paimon.globalindex.GlobalIndexReader; +import org.apache.paimon.globalindex.GlobalIndexSingleColumnWriter; +import org.apache.paimon.globalindex.GlobalIndexer; +import org.apache.paimon.globalindex.KeySerializer; +import org.apache.paimon.globalindex.ResultEntry; +import org.apache.paimon.globalindex.SortedGlobalIndexer; +import org.apache.paimon.globalindex.io.GlobalIndexFileWriter; +import org.apache.paimon.memory.MemorySlice; +import org.apache.paimon.options.Options; +import org.apache.paimon.types.DataField; +import org.apache.paimon.types.DataTypes; +import org.apache.paimon.types.RowType; +import org.apache.paimon.utils.Range; + +import org.junit.jupiter.api.Test; +import org.junit.jupiter.api.io.TempDir; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.ValueSource; + +import java.util.Arrays; +import java.util.Collections; +import java.util.Comparator; +import java.util.UUID; +import java.util.concurrent.ExecutorService; + +import static org.apache.paimon.shade.guava30.com.google.common.util.concurrent.MoreExecutors.newDirectExecutorService; +import static org.assertj.core.api.Assertions.assertThat; + +/** Tests for typed composite BTree keys. */ +class CompositeBTreeIndexTest { + + @TempDir java.nio.file.Path tempPath; + + @ParameterizedTest + @ValueSource(ints = {1, 2}) + void testMutableTupleKeysPostingListsAndLocalRanges(int version) throws Exception { + RowType type = + new RowType( + Arrays.asList( + new DataField(10, "category", DataTypes.STRING()), + new DataField(20, "item_number", DataTypes.INT()), + new DataField(30, "tag", DataTypes.STRING()))); + Options options = new Options(); + options.set(BTreeIndexOptions.BTREE_INDEX_FILE_VERSION, version); + options.set(BTreeIndexOptions.BTREE_INDEX_BLOOM_FILTER_ENABLED, true); + options.set(BTreeIndexOptions.BTREE_INDEX_COMPRESSION, "lz4"); + GlobalIndexer indexer = + GlobalIndexer.create( + "btree", type.getFields().get(0), type.getFields().subList(1, 3), options); + LocalFileIO io = LocalFileIO.create(); + Path directory = new Path(tempPath.toUri()); + GlobalIndexFileWriter files = + new GlobalIndexFileWriter() { + @Override + public String newFileName(String prefix) { + return prefix + UUID.randomUUID(); + } + + @Override + public PositionOutputStream newOutputStream(String name) + throws java.io.IOException { + return io.newOutputStream(new Path(directory, name), false); + } + }; + GenericRow reused = row("category-a", -1, ""); + GlobalIndexSingleColumnWriter writer = + (GlobalIndexSingleColumnWriter) indexer.createWriter(files); + writer.write(reused, 0); + reused.setField(1, 7); + writer.write(reused, 1); + reused.setField(2, BinaryString.fromString("tag")); + writer.write(reused, 2); + writer.write(reused, 3); + reused.setField(0, BinaryString.fromString("category-b")); + writer.write(reused, 4); + ResultEntry result = writer.finish().get(0); + Path path = new Path(directory, result.fileName()); + GlobalIndexIOMeta meta = + new GlobalIndexIOMeta(path, io.getFileSize(path), result.rowCount(), result.meta()); + ExecutorService executor = newDirectExecutorService(); + try (GlobalIndexReader reader = + indexer.createReader( + file -> io.newInputStream(file.filePath()), + Collections.singletonList(meta), + 5, + null, + executor)) { + assertThat( + reader.visitCompositeEqual( + Arrays.asList( + BinaryString.fromString("category-a"), + -1, + BinaryString.fromString(""))) + .get() + .get() + .results() + .toRangeList()) + .containsExactly(new Range(0, 0)); + assertThat( + reader.visitCompositeEqual( + Arrays.asList( + BinaryString.fromString("category-a"), + 7, + BinaryString.fromString("tag"))) + .get() + .get() + .results() + .toRangeList()) + .containsExactly(new Range(2, 3)); + assertThat( + reader.visitCompositeEqual( + Arrays.asList( + BinaryString.fromString("category-a"), + 8, + BinaryString.fromString("tag"))) + .get() + .get() + .results() + .isEmpty()) + .isTrue(); + } + try (GlobalIndexReader reader = + indexer.createReader( + file -> io.newInputStream(file.filePath()), + Collections.singletonList(meta), + 5, + Collections.singletonList(new Range(3, 3)), + executor)) { + assertThat( + reader.visitCompositeEqual( + Arrays.asList( + BinaryString.fromString("category-a"), + 7, + BinaryString.fromString("tag"))) + .get() + .get() + .results() + .toRangeList()) + .containsExactly(new Range(3, 3)); + } finally { + executor.shutdownNow(); + } + } + + @Test + void testCompositeKeysPreserveTypesBoundariesAndNulls() { + RowType type = + new RowType( + Arrays.asList( + new DataField(10, "category", DataTypes.STRING()), + new DataField(20, "item_number", DataTypes.INT()), + new DataField(30, "tag", DataTypes.STRING()))); + SortedGlobalIndexer indexer = + (SortedGlobalIndexer) + GlobalIndexer.create( + "btree", + type.getFields().get(0), + type.getFields().subList(1, 3), + new Options()); + KeySerializer serializer = KeySerializer.create(indexer.keyExtractor().keyType()); + Comparator comparator = serializer.createComparator(); + GenericRow first = row("category-a", -1, "a\u0000b"); + GenericRow second = row("category-a", 107, ""); + GenericRow nullable = row(null, 107, null); + for (GenericRow key : Arrays.asList(first, second, nullable)) { + Object restored = serializer.deserialize(MemorySlice.wrap(serializer.serialize(key))); + assertThat(restored).isEqualTo(key); + assertThat(comparator.compare(restored, key)).isZero(); + } + assertThat(comparator.compare(first, second)).isNegative(); + assertThat(comparator.compare(nullable, first)).isNegative(); + assertThat(serializer.serialize(row("a", 1, "bc"))) + .isNotEqualTo(serializer.serialize(row("ab", 1, "c"))); + assertThat( + GlobalIndexer.create( + "btree", + type.getFields().get(0), + Collections.emptyList(), + new Options())) + .isInstanceOf(BTreeGlobalIndexer.class); + } + + private GenericRow row(String category, int itemNumber, String tag) { + return GenericRow.of( + category == null ? null : BinaryString.fromString(category), + itemNumber, + tag == null ? null : BinaryString.fromString(tag)); + } +} diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java index 9ffcb69dd0c4..aa0d6559e820 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java @@ -399,8 +399,8 @@ private Plan planWithIndexQuery() { partitionFilter, indexFiles, table.coreOptions().scalarIndexSearchMode()) - .unindexedRanges( - indexQuery.contributingFieldIds(table.rowType()), + .unindexedRangesFromCoverage( + indexQuery.coveredRanges(), table.coreOptions().scalarIndexSearchMode() == CoreOptions.GlobalIndexSearchMode.DETAIL ? GlobalIndexBuilderUtils.calcRowRanges( @@ -472,11 +472,7 @@ private Plan planEagerIndex( return dataPlan; } GlobalIndexResult candidates = - result.get() - .result() - .or( - scanner.unindexedRowsForContributingFields( - result.get().contributingFieldIds())); + result.get().result().or(scanner.unindexedRowsForEvaluation(result.get())); RowRangeIndex rowRangeIndex = RowRangeIndex.create(candidates.results().toRangeList()); ScoreGetter scores = candidates instanceof ScoredGlobalIndexResult @@ -553,11 +549,7 @@ private Optional evalGlobalIndex() { if (result.isPresent()) { long coverageStart = System.nanoTime(); GlobalIndexResult finalResult = - result.get() - .result() - .or( - scanner.unindexedRowsForContributingFields( - result.get().contributingFieldIds())); + result.get().result().or(scanner.unindexedRowsForEvaluation(result.get())); long coverageDuration = System.nanoTime() - coverageStart; long totalDuration = System.nanoTime() - totalStart; LOG.info( diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexCoverage.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexCoverage.java index 6a945c054474..eb86c3397bd5 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexCoverage.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexCoverage.java @@ -67,6 +67,13 @@ public DataEvolutionGlobalIndexCoverage( this.coverageByField = new HashMap<>(); for (IndexFileMeta indexFile : indexFiles) { GlobalIndexMeta meta = checkNotNull(indexFile.globalIndexMeta()); + // Tuple indexes cannot answer an individual column's scalar predicate. + // Their coverage is supplied explicitly by the selected composite query. + if ("btree".equals(indexFile.indexType()) + && meta.extraFieldIds() != null + && meta.extraFieldIds().length > 0) { + continue; + } Range range = new Range(meta.rowRangeStart(), meta.rowRangeEnd()); addCoverage(meta.indexFieldId(), range); if (meta.extraFieldIds() != null) { @@ -87,6 +94,11 @@ public List unindexedRanges(int fieldId) { public List unindexedRanges( Collection fieldIds, @Nullable List plannedDataRanges) { + return unindexedRangesFromCoverage(indexedRanges(fieldIds), plannedDataRanges); + } + + public List unindexedRangesFromCoverage( + List indexedRanges, @Nullable List plannedDataRanges) { if (searchMode == GlobalIndexSearchMode.FAST) { return Collections.emptyList(); } @@ -101,8 +113,7 @@ public List unindexedRanges( dataRanges = Collections.singletonList(new Range(0, snapshot.nextRowId() - 1)); } - List predicateIndexedRanges = - Range.sortAndMergeOverlap(indexedRanges(fieldIds), true); + List predicateIndexedRanges = Range.sortAndMergeOverlap(indexedRanges, true); List unindexedRanges = new ArrayList<>(); for (Range dataRange : Range.sortAndMergeOverlap(dataRanges, true)) { unindexedRanges.addAll(dataRange.exclude(predicateIndexedRanges)); diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java index 5b000ad7546d..10434d22900f 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java @@ -50,6 +50,7 @@ import java.util.Collections; import java.util.HashMap; import java.util.HashSet; +import java.util.LinkedHashMap; import java.util.List; import java.util.Map; import java.util.Optional; @@ -62,7 +63,6 @@ import static org.apache.paimon.CoreOptions.GLOBAL_INDEX_THREAD_NUM; import static org.apache.paimon.predicate.PredicateVisitor.collectFieldIds; import static org.apache.paimon.table.source.snapshot.TimeTravelUtil.tryTravelOrLatest; -import static org.apache.paimon.utils.Preconditions.checkArgument; import static org.apache.paimon.utils.Preconditions.checkNotNull; /** Scanner for shard-based global indexes on data-evolution tables. */ @@ -80,6 +80,8 @@ public class DataEvolutionGlobalIndexScanner implements Closeable { private final IndexPathFactory indexPathFactory; private final DataEvolutionGlobalIndexCoverage coverage; private final FileStoreTable table; + private final List indexFiles; + private final FileIO fileIO; private DataEvolutionGlobalIndexScanner( FileStoreTable table, @@ -113,6 +115,8 @@ private DataEvolutionGlobalIndexScanner( Collection coverageIndexFiles, Collection indexFiles) { this.table = table; + this.indexFiles = new ArrayList<>(indexFiles); + this.fileIO = fileIO; this.options = options; this.rowType = rowType; this.executor = @@ -147,23 +151,15 @@ private DataEvolutionGlobalIndexScanner( /** Groups metadata for both planning-time readers and reader-side query plans. */ static Map> groupIndexFiles( Collection indexFiles) { - Map primaryGroups = new HashMap<>(); + Map, IndexMetaFileGroup> primaryGroups = new LinkedHashMap<>(); for (IndexFileMeta indexFile : indexFiles) { GlobalIndexMeta meta = checkNotNull(indexFile.globalIndexMeta()); int indexFieldId = meta.indexFieldId(); List fieldIds = meta.getIndexedFieldIds(); - IndexMetaFileGroup group = primaryGroups.get(indexFieldId); + IndexMetaFileGroup group = primaryGroups.get(fieldIds); if (group == null) { group = new IndexMetaFileGroup(indexFieldId, fieldIds); - primaryGroups.put(indexFieldId, group); - } else { - checkArgument( - group.fieldIds.equals(fieldIds), - "Primary field %s owns multiple indexes with different columns %s and %s; " - + "a primary column can own at most one index.", - indexFieldId, - group.fieldIds, - fieldIds); + primaryGroups.put(fieldIds, group); } group.addFile(indexFile.indexType(), meta.rowRange(), indexFile); } @@ -325,6 +321,8 @@ private static Filter topNIndexFileFilter( GlobalIndexMeta globalIndex = indexFile.globalIndexMeta(); return globalIndex != null && globalIndex.indexFieldId() == fieldId + && (globalIndex.extraFieldIds() == null + || globalIndex.extraFieldIds().length == 0) && BTreeGlobalIndexerFactory.IDENTIFIER.equals(indexFile.indexType()); }; } @@ -363,6 +361,14 @@ static Filter indexFileFilter( return indexFileFilter; } + private static boolean isCompositeBTree(IndexFileMeta file) { + GlobalIndexMeta meta = file.globalIndexMeta(); + return "btree".equals(file.indexType()) + && meta != null + && meta.extraFieldIds() != null + && meta.extraFieldIds().length > 0; + } + private static List globalIndexFiles(Collection indexFiles) { return indexFiles.stream() .filter(indexFile -> indexFile.globalIndexMeta() != null) @@ -370,10 +376,33 @@ private static List globalIndexFiles(Collection in } public Optional scan(Predicate predicate) { - return globalIndexEvaluator.evaluate(predicate); + return scanWithCoverage(predicate).map(GlobalIndexEvaluator.Evaluation::result); } public Optional scanWithCoverage(Predicate predicate) { + GlobalIndexQuery query = + predicate == null + || indexFiles.stream() + .noneMatch( + DataEvolutionGlobalIndexScanner::isCompositeBTree) + ? null + : GlobalIndexQuery.create(rowType, predicate, indexFiles, indexPathFactory); + if (query != null && query.hasCompositeQuery()) { + try { + return Optional.of( + new GlobalIndexEvaluator.Evaluation( + query.evaluate( + fileIO, + options, + indexFiles.stream() + .map(file -> file.globalIndexMeta().rowRange()) + .collect(Collectors.toList())), + query.contributingFieldIds(rowType), + query.coveredRanges())); + } catch (IOException e) { + throw new RuntimeException("Failed to evaluate composite BTree query", e); + } + } return globalIndexEvaluator.evaluateWithContributingFields(predicate); } @@ -393,12 +422,29 @@ public Optional scan(TopN topN) { public GlobalIndexResult unindexedRows(Predicate predicate) { RoaringNavigableMap64 rows = new RoaringNavigableMap64(); - for (Range range : coverage.unindexedRanges(rowType, predicate)) { + GlobalIndexQuery query = + predicate == null + ? null + : GlobalIndexQuery.create(rowType, predicate, indexFiles, indexPathFactory); + List unindexed = + query != null && query.hasCompositeQuery() + ? coverage.unindexedRangesFromCoverage(query.coveredRanges(), null) + : coverage.unindexedRanges(rowType, predicate); + for (Range range : unindexed) { rows.addRange(range); } return GlobalIndexResult.create(rows); } + public GlobalIndexResult unindexedRowsForEvaluation( + GlobalIndexEvaluator.Evaluation evaluation) { + if (evaluation.coveredRanges() == null) { + return unindexedRowsForContributingFields(evaluation.contributingFieldIds()); + } + return GlobalIndexResult.fromRanges( + coverage.unindexedRangesFromCoverage(evaluation.coveredRanges(), null)); + } + public GlobalIndexResult unindexedRowsForContributingFields( Collection contributingFieldIds) { RoaringNavigableMap64 rows = new RoaringNavigableMap64(); @@ -425,6 +471,9 @@ private Collection createReaders( Set readers = new HashSet<>(); for (Map.Entry>> entry : group.metas.entrySet()) { String indexType = entry.getKey(); + if ("btree".equals(indexType) && !extraFields.isEmpty()) { + continue; + } Map> metas = entry.getValue(); GlobalIndexerFactory globalIndexerFactory = GlobalIndexerFactoryUtils.load(indexType); GlobalIndexer globalIndexer = diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java index 7745f8ba2155..006c1aebd484 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java @@ -21,6 +21,7 @@ import org.apache.paimon.fs.FileIO; import org.apache.paimon.fs.Path; import org.apache.paimon.globalindex.DataEvolutionGlobalIndexScanner.IndexMetaFileGroup; +import org.apache.paimon.globalindex.btree.CompositeBTreePredicate; import org.apache.paimon.index.IndexFileMeta; import org.apache.paimon.index.IndexPathFactory; import org.apache.paimon.io.DataInputView; @@ -28,12 +29,14 @@ import org.apache.paimon.options.Options; import org.apache.paimon.predicate.And; import org.apache.paimon.predicate.CompoundPredicate; +import org.apache.paimon.predicate.Equal; import org.apache.paimon.predicate.FieldRef; import org.apache.paimon.predicate.GreaterOrEqual; import org.apache.paimon.predicate.LeafPredicate; import org.apache.paimon.predicate.LessOrEqual; import org.apache.paimon.predicate.Or; import org.apache.paimon.predicate.Predicate; +import org.apache.paimon.predicate.PredicateBuilder; import org.apache.paimon.types.DataField; import org.apache.paimon.types.RowType; import org.apache.paimon.utils.InstantiationUtil; @@ -58,6 +61,7 @@ import java.util.concurrent.ExecutionException; import java.util.concurrent.ExecutorService; import java.util.function.Function; +import java.util.stream.Collectors; import static org.apache.paimon.CoreOptions.GLOBAL_INDEX_THREAD_NUM; import static org.apache.paimon.predicate.PredicateVisitor.collectFieldIds; @@ -73,7 +77,7 @@ */ class GlobalIndexQuery { - /** A leaf or paired range query; null for an AND/OR node. */ + /** A leaf, paired range or composite lookup; null for an AND/OR node. */ @Nullable private final Predicate predicate; /** For compound nodes, true means OR and false means AND; ignored for leaves. */ @@ -131,6 +135,59 @@ private static GlobalIndexQuery createForPredicate( boolean union = compound.function() instanceof Or; List children = new ArrayList<>(); List predicates = GlobalIndexEvaluator.normalizedChildren(compound); + if (!union) { + // Prefer the longest full composite key before any single-column posting is read. + while (!predicates.isEmpty()) { + IndexGroup selected = null; + List matched = null; + Predicate conjunction = PredicateBuilder.and(predicates); + for (List fieldGroups : groups.values()) { + for (IndexGroup group : fieldGroups) { + if (!group.isCompositeBTree() + || (selected != null + && group.extraFields.size() + <= selected.extraFields.size())) { + continue; + } + Optional> match = + CompositeBTreePredicate.match(group.fields(), conjunction); + if (match.isPresent()) { + selected = group; + matched = match.get(); + } + } + } + if (selected == null) { + break; + } + List covered = new ArrayList<>(); + for (Predicate child : predicates) { + if (child instanceof LeafPredicate) { + LeafPredicate leaf = (LeafPredicate) child; + if (leaf.function() instanceof Equal + && matched.stream() + .anyMatch( + equality -> + equality.fieldRefOptional() + .equals(leaf.fieldRefOptional()))) { + covered.add(child); + } + } + } + Predicate composite = PredicateBuilder.and(covered); + List selectedGroups = new ArrayList<>(); + for (IndexGroup group : groups.get(selected.field.id())) { + if (group.type.equals(selected.type) + && group.fields().equals(selected.fields())) { + selectedGroups.add(group.selectFiles(composite)); + } + } + children.add( + new GlobalIndexQuery( + composite, false, Collections.emptyList(), selectedGroups)); + predicates.removeAll(covered); + } + } for (int i = 0; i < predicates.size(); i++) { Predicate child = predicates.get(i); GlobalIndexQuery query = null; @@ -188,33 +245,47 @@ private static GlobalIndexQuery createForIndexedField( } List selectedGroups = new ArrayList<>(); for (IndexGroup group : fieldGroups) { - List selectedFiles = - GlobalIndexerFactoryUtils.selectFiles( - group.type, group.field, group.extraFields, predicate, group.files); - if (!selectedFiles.isEmpty()) { - selectedGroups.add( - selectedFiles == group.files - ? group - : new IndexGroup( - group.type, - group.field, - group.extraFields, - group.range, - selectedFiles)); + // Non-leading columns cannot serve a scalar lookup in a tuple BTree. + if (!group.isCompositeBTree()) { + selectedGroups.add(group.selectFiles(predicate)); } } + if (selectedGroups.isEmpty()) { + return null; + } return new GlobalIndexQuery(predicate, false, Collections.emptyList(), selectedGroups); } boolean isEmpty() { if (predicate != null) { - return groups.isEmpty(); + return groups.stream().allMatch(group -> group.files.isEmpty()); } return union ? children.stream().allMatch(GlobalIndexQuery::isEmpty) : children.stream().anyMatch(GlobalIndexQuery::isEmpty); } + boolean hasCompositeQuery() { + return groups.stream().anyMatch(IndexGroup::isCompositeBTree) + || children.stream().anyMatch(GlobalIndexQuery::hasCompositeQuery); + } + + /** Coverage of the selected query paths, including files pruned safely by key metadata. */ + List coveredRanges() { + if (predicate != null) { + return Range.sortAndMergeOverlap( + groups.stream().map(group -> group.range).collect(Collectors.toList()), true); + } + List coverage = null; + for (GlobalIndexQuery child : children) { + coverage = + coverage == null + ? child.coveredRanges() + : Range.and(coverage, child.coveredRanges()); + } + return coverage == null ? Collections.emptyList() : coverage; + } + /** Residual predicates discarded during planning must not expand unindexed coverage. */ Set contributingFieldIds(RowType rowType) { Set fields = new HashSet<>(); @@ -271,10 +342,13 @@ private GlobalIndexResult evaluateWithExecutor( } return result == null ? GlobalIndexResult.createEmpty() : result; } - Function>> query = - predicateQuery(); GlobalIndexResult result = GlobalIndexResult.createEmpty(); for (IndexGroup group : groups) { + if (group.files.isEmpty()) { + continue; + } + Function>> query = + predicateQuery(group); GlobalIndexResult splitRows = localSplitRows(ranges, group.range); if (splitRows.results().isEmpty()) { continue; @@ -306,7 +380,25 @@ private GlobalIndexResult evaluateWithExecutor( } private Function>> - predicateQuery() { + predicateQuery(IndexGroup group) { + if (group.isCompositeBTree()) { + if (CompositeBTreePredicate.isContradictory(group.fields(), predicate)) { + return reader -> + CompletableFuture.completedFuture( + Optional.of(GlobalIndexResult.createEmpty())); + } + List equalities = + CompositeBTreePredicate.match(group.fields(), predicate) + .orElseThrow( + () -> + new IllegalArgumentException( + "Incomplete composite BTree predicate")); + List literals = + equalities.stream() + .map(leaf -> leaf.literals().get(0)) + .collect(Collectors.toList()); + return reader -> reader.visitCompositeEqual(literals); + } if (predicate instanceof LeafPredicate) { LeafPredicate leaf = (LeafPredicate) predicate; return reader -> @@ -446,8 +538,9 @@ public int hashCode() { } /** - * Files read together by one index reader, grouped by primary column, index type and row range. - * A group may overlap multiple data splits; primary and extra column predicates can share it. + * Files read together by one index reader, grouped by ordered index columns, index type and row + * range. A group may overlap multiple data splits; primary and extra column predicates can + * share it. */ private static class IndexGroup { /** Factory identifier used to create the index reader on the worker. */ @@ -506,6 +599,27 @@ private static List fromMetadata( return result; } + private boolean isCompositeBTree() { + return "btree".equals(type) && !extraFields.isEmpty(); + } + + private List fields() { + List fields = new ArrayList<>(); + fields.add(field); + fields.addAll(extraFields); + return fields; + } + + private IndexGroup selectFiles(Predicate predicate) { + List selected = + GlobalIndexerFactoryUtils.selectFiles( + type, field, extraFields, predicate, files); + // Retain the row range even when metadata proves there are no matching files. + return selected == files + ? this + : new IndexGroup(type, field, extraFields, range, selected); + } + @Override public boolean equals(Object obj) { if (!(obj instanceof IndexGroup)) { diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexScanner.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexScanner.java index 7f44a334515b..3871fb789af0 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexScanner.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexScanner.java @@ -58,7 +58,7 @@ public class SortedGlobalIndexScanner implements Serializable { private final RowType rowType; private final Options options; - private DataField indexField; + private List indexFields; @Nullable private Snapshot snapshot; @@ -76,12 +76,20 @@ public SortedGlobalIndexScanner(Table table, String indexType, Options options) } public SortedGlobalIndexScanner withIndexField(String indexField) { - checkArgument( - rowType.containsField(indexField), - "Column '%s' does not exist in table '%s'.", - indexField, - table.fullName()); - this.indexField = rowType.getField(indexField); + return withIndexFields(Collections.singletonList(indexField)); + } + + public SortedGlobalIndexScanner withIndexFields(List fieldNames) { + checkArgument(!fieldNames.isEmpty(), "At least one index column is required."); + this.indexFields = new ArrayList<>(); + for (String name : fieldNames) { + checkArgument( + rowType.containsField(name), + "Column '%s' does not exist in table '%s'.", + name, + table.fullName()); + indexFields.add(rowType.getField(name)); + } return this; } @@ -132,25 +140,16 @@ public Optional> incrementalScan() { } snapshotReader = withManifestEntryFilter(snapshotReader.withSnapshot(snapshot)); - Preconditions.checkArgument(indexField != null, "indexField must be set before scan."); + Preconditions.checkArgument(indexFields != null, "indexFields must be set before scan."); List currentIndexes = - currentIndexEntries( - table, - snapshot, - indexType, - Collections.singletonList(indexField), - partitionPredicate); + currentIndexEntries(table, snapshot, indexType, indexFields, partitionPredicate); List rangesToBuild = new ArrayList<>(unindexedRowRanges(snapshot, currentIndexes)); List deletedIndexEntries = Collections.emptyList(); if (detectDataFileChange()) { // Scans data manifests through reusable binary views without materializing entries. deletedIndexEntries = DataEvolutionGlobalIndexRefreshPlanner.findIndexesToRefresh( - table, - snapshot, - partitionPredicate, - currentIndexes, - Collections.singletonList(indexField)); + table, snapshot, partitionPredicate, currentIndexes, indexFields); for (IndexManifestEntry entry : deletedIndexEntries) { rangesToBuild.add(entry.indexFile().globalIndexMeta().rowRange()); } diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexWriter.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexWriter.java index b0bd408d877a..0d6b77c8cf41 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexWriter.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexWriter.java @@ -63,7 +63,7 @@ public class SortedGlobalIndexWriter implements Serializable { private final Options options; private final long recordsPerRange; - private DataField indexField; + private List indexFields; private GlobalIndexKeyExtractor keyExtractor; public SortedGlobalIndexWriter(Table table, String indexType) { @@ -80,13 +80,26 @@ public SortedGlobalIndexWriter(Table table, String indexType, Options options) { } public SortedGlobalIndexWriter withIndexField(String indexField) { - checkArgument( - rowType.containsField(indexField), - "Column '%s' does not exist in table '%s'.", - indexField, - table.fullName()); - this.indexField = rowType.getField(indexField); - GlobalIndexer indexer = GlobalIndexer.create(indexType, this.indexField, options); + return withIndexFields(Collections.singletonList(indexField)); + } + + public SortedGlobalIndexWriter withIndexFields(List fieldNames) { + checkArgument(!fieldNames.isEmpty(), "At least one index column is required."); + this.indexFields = new ArrayList<>(); + for (String name : fieldNames) { + checkArgument( + rowType.containsField(name), + "Column '%s' does not exist in table '%s'.", + name, + table.fullName()); + indexFields.add(rowType.getField(name)); + } + GlobalIndexer indexer = + GlobalIndexer.create( + indexType, + indexFields.get(0), + indexFields.subList(1, indexFields.size()), + options); checkArgument( indexer instanceof SortedGlobalIndexer, "Index algorithm %s does not expose sorted index keys.", @@ -131,7 +144,13 @@ public SortedSingleColumnIndexWriter createTaskWriter(Range rowRange) throws IOE } public GlobalIndexSingleColumnWriter createWriter() throws IOException { - GlobalIndexWriter indexWriter = createIndexWriter(table, indexType, indexField, options); + GlobalIndexWriter indexWriter = + createIndexWriter( + table, + indexType, + indexFields.get(0), + indexFields.subList(1, indexFields.size()), + options); if (!(indexWriter instanceof GlobalIndexSingleColumnWriter)) { throw new RuntimeException( "Unexpected implementation, the index writer of " @@ -155,7 +174,7 @@ public CommitMessage flushIndex( table.store().pathFactory().globalIndexFileFactory(), table.coreOptions(), rowRange, - Collections.singletonList(indexField), + indexFields, indexType, resultEntries, sourceMeta); diff --git a/paimon-core/src/main/java/org/apache/paimon/manifest/IndexManifestFileHandler.java b/paimon-core/src/main/java/org/apache/paimon/manifest/IndexManifestFileHandler.java index 16cfd163ed1c..9f4b9f485945 100644 --- a/paimon-core/src/main/java/org/apache/paimon/manifest/IndexManifestFileHandler.java +++ b/paimon-core/src/main/java/org/apache/paimon/manifest/IndexManifestFileHandler.java @@ -254,6 +254,14 @@ private void validateRetainedIndexFiles( && !DataEvolutionIndexSourceMeta.isDataEvolutionMeta( addedMeta.sourceMeta())) || retainedMeta.indexFieldId() != addedMeta.indexFieldId() + || (!Arrays.equals( + retainedMeta.extraFieldIds(), addedMeta.extraFieldIds()) + && (("btree".equals(retained.indexFile().indexType()) + && retainedMeta.extraFieldIds() != null + && retainedMeta.extraFieldIds().length > 0) + || ("btree".equals(added.indexFile().indexType()) + && addedMeta.extraFieldIds() != null + && addedMeta.extraFieldIds().length > 0))) || (Arrays.equals( retainedMeta.extraFieldIds(), addedMeta.extraFieldIds()) && !Range.intersect( diff --git a/paimon-core/src/test/java/org/apache/paimon/globalindex/GlobalIndexQueryTest.java b/paimon-core/src/test/java/org/apache/paimon/globalindex/GlobalIndexQueryTest.java index d2821933f7bb..90e899c7d9e2 100644 --- a/paimon-core/src/test/java/org/apache/paimon/globalindex/GlobalIndexQueryTest.java +++ b/paimon-core/src/test/java/org/apache/paimon/globalindex/GlobalIndexQueryTest.java @@ -571,21 +571,14 @@ void testGroupsPreserveIndexTypesRangesAndExtraFields() { } @Test - void testGroupsRejectInconsistentIndexedFields() { - assertThatThrownBy( - () -> - DataEvolutionGlobalIndexScanner.groupIndexFiles( - Arrays.asList( - indexFile( - "es-index", - "first", - 0, - 99, - 10, - new int[] {20}), - indexFile("es-index", "tail", 100, 199, 10, null)))) - .isInstanceOf(IllegalArgumentException.class) - .hasMessageContaining("different columns"); + void testGroupsAllowSingleAndCompositeIndexesWithTheSamePrimaryField() { + Map> groups = + DataEvolutionGlobalIndexScanner.groupIndexFiles( + Arrays.asList( + indexFile("btree", "composite", 0, 99, 10, new int[] {20}), + indexFile("btree", "single", 100, 199, 10, null))); + assertThat(groups.get(10)).hasSize(2); + assertThat(groups.get(20)).hasSize(1); } @Test diff --git a/paimon-core/src/test/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexTestUtils.java b/paimon-core/src/test/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexTestUtils.java index d25f68cda6df..a77b43560873 100644 --- a/paimon-core/src/test/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexTestUtils.java +++ b/paimon-core/src/test/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexTestUtils.java @@ -27,7 +27,6 @@ import org.apache.paimon.table.SpecialFields; import org.apache.paimon.table.sink.CommitMessage; import org.apache.paimon.table.source.DataSplit; -import org.apache.paimon.types.DataField; import org.apache.paimon.types.RowType; import org.apache.paimon.utils.CloseableIterator; import org.apache.paimon.utils.InternalRowUtils; @@ -35,7 +34,6 @@ import org.apache.paimon.utils.Range; import java.util.ArrayList; -import java.util.Arrays; import java.util.Collections; import java.util.Comparator; import java.util.List; @@ -54,13 +52,30 @@ public static List buildIndex( DataSplit dataSplit, long scanSnapshotId) throws Exception { + return buildIndex( + table, + indexType, + Collections.singletonList(indexFieldName), + dataSplit, + scanSnapshotId); + } + + public static List buildIndex( + FileStoreTable table, + String indexType, + List indexFieldNames, + DataSplit dataSplit, + long scanSnapshotId) + throws Exception { SortedGlobalIndexWriter writer = - new SortedGlobalIndexWriter(table, indexType).withIndexField(indexFieldName); - DataField indexField = table.rowType().getField(indexFieldName); - RowType readRowType = - SpecialFields.rowTypeWithRowId(table.rowType()) - .project(Arrays.asList(indexFieldName, SpecialFields.ROW_ID.name())); - InternalRow.FieldGetter fieldGetter = InternalRow.createFieldGetter(indexField.type(), 0); + new SortedGlobalIndexWriter(table, indexType).withIndexFields(indexFieldNames); + List readFields = new ArrayList<>(indexFieldNames); + readFields.add(SpecialFields.ROW_ID.name()); + RowType readRowType = SpecialFields.rowTypeWithRowId(table.rowType()).project(readFields); + InternalRow.FieldGetter[] getters = new InternalRow.FieldGetter[indexFieldNames.size()]; + for (int i = 0; i < getters.length; i++) { + getters[i] = InternalRow.createFieldGetter(readRowType.getTypeAt(i), i); + } GlobalIndexKeyExtractor keyExtractor = writer.keyExtractor(); List> rows = new ArrayList<>(); try (RecordReader reader = @@ -71,8 +86,17 @@ public static List buildIndex( CloseableIterator iterator = reader.toCloseableIterator()) { while (iterator.hasNext()) { InternalRow row = iterator.next(); - Object value = fieldGetter.getFieldOrNull(row); - long rowId = row.getLong(1); + Object value; + if (getters.length == 1) { + value = getters[0].getFieldOrNull(row); + } else { + GenericRow tuple = new GenericRow(getters.length); + for (int i = 0; i < getters.length; i++) { + tuple.setField(i, getters[i].getFieldOrNull(row)); + } + value = tuple; + } + long rowId = row.getLong(getters.length); keyExtractor.extract( value, key -> diff --git a/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java b/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java new file mode 100644 index 000000000000..da68f32d3848 --- /dev/null +++ b/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java @@ -0,0 +1,338 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.paimon.table; + +import org.apache.paimon.CoreOptions; +import org.apache.paimon.data.BinaryString; +import org.apache.paimon.data.GenericRow; +import org.apache.paimon.globalindex.DataEvolutionGlobalIndexScanner; +import org.apache.paimon.globalindex.ScanResult; +import org.apache.paimon.globalindex.sorted.SortedGlobalIndexScanner; +import org.apache.paimon.globalindex.sorted.SortedGlobalIndexTestUtils; +import org.apache.paimon.index.IndexFileMeta; +import org.apache.paimon.io.CompactIncrement; +import org.apache.paimon.io.DataIncrement; +import org.apache.paimon.manifest.IndexManifestEntry; +import org.apache.paimon.predicate.Predicate; +import org.apache.paimon.predicate.PredicateBuilder; +import org.apache.paimon.table.sink.BatchTableCommit; +import org.apache.paimon.table.sink.BatchTableWrite; +import org.apache.paimon.table.sink.BatchWriteBuilder; +import org.apache.paimon.table.sink.CommitMessage; +import org.apache.paimon.table.sink.CommitMessageImpl; +import org.apache.paimon.table.source.DataSplit; +import org.apache.paimon.table.source.ReadBuilder; +import org.apache.paimon.table.source.Split; +import org.apache.paimon.table.source.TableScan; +import org.apache.paimon.utils.InstantiationUtil; +import org.apache.paimon.utils.Range; + +import org.junit.jupiter.api.Test; + +import java.util.ArrayList; +import java.util.Arrays; +import java.util.Collections; +import java.util.HashMap; +import java.util.List; +import java.util.Map; +import java.util.Optional; +import java.util.stream.Collectors; + +import static org.assertj.core.api.Assertions.assertThat; + +/** Composite lookups must avoid single-column postings and retain uncovered data. */ +class CompositeBTreeTableTest extends DataEvolutionTestBase { + + @Test + void testCoexistsAndAvoidsSingleColumnPostings() throws Exception { + createTableDefault(); + FileStoreTable table = + table().copy(Collections.singletonMap("sorted-index.records-per-file", "13")); + append(table, 0, 100); + build(table, "f1"); + build(table, "f0"); + build(table, "f1", "f0"); + deleteSingleColumnFiles(table); + PredicateBuilder predicates = new PredicateBuilder(table.rowType()); + Predicate query = query(predicates, "category-a", 7); + for (boolean inReader : Arrays.asList(false, true)) { + FileStoreTable configured = configured(table, "full", inReader); + assertThat(read(configured, query)) + .containsExactlyInAnyOrder("p7", "p27", "p47", "p67", "p87"); + // Both OR branches are composite queries; predicate order differs from index order. + Predicate union = PredicateBuilder.or(query, query(predicates, "category-b", 8)); + assertThat(read(configured, union)).hasSize(10).contains("p7", "p18", "p98"); + assertThat(read(configured, query(predicates, "missing", 7))).isEmpty(); + assertThat(read(configured, PredicateBuilder.and(query, predicates.equal(0, 8)))) + .isEmpty(); + } + try (DataEvolutionGlobalIndexScanner scanner = + DataEvolutionGlobalIndexScanner.create( + table, + table.store().newIndexFileHandler().scanEntries().stream() + .map(IndexManifestEntry::indexFile) + .collect(Collectors.toList())) + .get()) { + assertThat(scanner.scan(query).get().results().toRangeList()) + .containsExactly( + new Range(7, 7), + new Range(27, 27), + new Range(47, 47), + new Range(67, 67), + new Range(87, 87)); + } + } + + @Test + void testPartialCompositeCoverageDoesNotBorrowSingleColumnCoverage() throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 20); + build(table, "f1", "f0"); + append(table, 20, 40); + build(table, "f1"); + build(table, "f0"); + deleteSingleColumnFiles(table); + Predicate query = query(new PredicateBuilder(table.rowType()), "category-a", 7); + for (boolean inReader : Arrays.asList(false, true)) { + assertThat(read(configured(table, "fast", inReader), query)).containsExactly("p7"); + for (String mode : Arrays.asList("full", "detail")) { + assertThat(read(configured(table, mode, inReader), query)) + .containsExactlyInAnyOrder("p7", "p27"); + } + } + } + + @Test + void testPartialSingleCoverageDoesNotBorrowCompositeCoverage() throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 20); + build(table, "f1"); + append(table, 20, 40); + build(table, "f1", "f0"); + Predicate scalar = + new PredicateBuilder(table.rowType()) + .equal(1, BinaryString.fromString("category-a")); + for (boolean inReader : Arrays.asList(false, true)) { + assertThat(read(configured(table, "fast", inReader), scalar)).hasSize(10); + for (String mode : Arrays.asList("full", "detail")) { + assertThat(read(configured(table, mode, inReader), scalar)) + .hasSize(20) + .contains("p27"); + } + } + } + + @Test + void testIncrementalBuildAndRefreshOfEitherComponent() throws Exception { + createTableDefault(); + FileStoreTable table = + table().copy( + Collections.singletonMap( + CoreOptions.GLOBAL_INDEX_COLUMN_UPDATE_ACTION.key(), + "IGNORE")); + append(table, 0, 20); + build(table, "f1", "f0"); + append(table, 20, 40); + build(table, "f1", "f0"); + assertThat( + new SortedGlobalIndexScanner(table, "btree") + .withIndexFields(Arrays.asList("f1", "f0")) + .incrementalScan()) + .isEmpty(); + update(table, "f0", 8); + build(table, "f1", "f0"); + PredicateBuilder predicates = new PredicateBuilder(table.rowType()); + assertThat(read(configured(table, "fast", false), query(predicates, "category-a", 7))) + .containsExactly("p27"); + assertThat(read(configured(table, "fast", true), query(predicates, "category-a", 8))) + .containsExactlyInAnyOrder("p7", "p8", "p28"); + update(table, "f1", BinaryString.fromString("changed")); + build(table, "f1", "f0"); + assertThat(read(configured(table, "fast", true), query(predicates, "changed", 8))) + .containsExactly("p7"); + assertThat(read(configured(table, "fast", false), query(predicates, "category-a", 8))) + .containsExactlyInAnyOrder("p8", "p28"); + } + + @Test + void testThreeFieldsNullComponentsAndUnsupportedPartialPredicate() throws Exception { + createTableDefault(); + FileStoreTable table = + table().copy(Collections.singletonMap("btree-index.bloom-filter.enabled", "true")); + BatchWriteBuilder writes = table.newBatchWriteBuilder(); + try (BatchTableWrite write = writes.newWrite(); + BatchTableCommit commit = writes.newCommit()) { + write.write( + GenericRow.of( + -1, + BinaryString.fromString("category-a"), + BinaryString.fromString("a\u0000b"))); + write.write( + GenericRow.of( + -1, + BinaryString.fromString("category-a"), + BinaryString.fromString(""))); + write.write(GenericRow.of(-1, null, BinaryString.fromString("null-category"))); + write.write( + GenericRow.of( + null, + BinaryString.fromString("category-a"), + BinaryString.fromString("null-number"))); + commit.commit(write.prepareCommit()); + } + build(table, "f1", "f0", "f2"); + PredicateBuilder predicates = new PredicateBuilder(table.rowType()); + Predicate full = + PredicateBuilder.and( + query(predicates, "category-a", -1), + predicates.equal(2, BinaryString.fromString("a\u0000b"))); + for (boolean inReader : Arrays.asList(false, true)) { + assertThat(read(configured(table, "full", inReader), full)).containsExactly("a\u0000b"); + // A partial tuple must use the ordinary data scan, never serialize a scalar as a tuple. + assertThat(read(configured(table, "full", inReader), predicates.equal(0, -1))) + .hasSize(3); + } + } + + private FileStoreTable table() throws Exception { + return (FileStoreTable) catalog.getTable(identifier()); + } + + private Predicate query(PredicateBuilder builder, String category, int itemNumber) { + return PredicateBuilder.and( + builder.equal(0, itemNumber), builder.equal(1, BinaryString.fromString(category))); + } + + private FileStoreTable configured(FileStoreTable table, String mode, boolean inReader) { + Map options = new HashMap<>(); + options.put(CoreOptions.SCALAR_INDEX_SEARCH_MODE.key(), mode); + options.put( + CoreOptions.GLOBAL_INDEX_QUERY_IN_READER_ENABLED.key(), String.valueOf(inReader)); + return table.copy(options); + } + + private List read(FileStoreTable table, Predicate predicate) throws Exception { + ReadBuilder builder = table.newReadBuilder().withFilter(predicate); + TableScan.Plan plan = builder.newScan().plan(); + // Exercise split checkpoint serialization as well as direct execution. + List splits = new ArrayList<>(); + for (Split split : plan.splits()) { + splits.add( + InstantiationUtil.deserializeObject( + InstantiationUtil.serializeObject(split), getClass().getClassLoader())); + } + List values = new ArrayList<>(); + builder.newRead() + .executeFilter() + .createReader(splits) + .forEachRemaining(row -> values.add(row.getString(2).toString())); + return values; + } + + private void append(FileStoreTable table, int from, int to) throws Exception { + BatchWriteBuilder writes = table.newBatchWriteBuilder(); + try (BatchTableWrite write = writes.newWrite(); + BatchTableCommit commit = writes.newCommit()) { + for (int i = from; i < to; i++) { + write.write( + GenericRow.of( + i % 10, + BinaryString.fromString( + (i / 10) % 2 == 0 ? "category-a" : "category-b"), + BinaryString.fromString("p" + i))); + } + commit.commit(write.prepareCommit()); + } + } + + private void update(FileStoreTable table, String field, Object value) throws Exception { + BatchWriteBuilder writes = table.newBatchWriteBuilder(); + try (BatchTableWrite write = + writes.newWrite() + .withWriteType( + table.rowType().project(Collections.singletonList(field))); + BatchTableCommit commit = writes.newCommit()) { + for (int i = 0; i < 20; i++) { + Object original = + field.equals("f0") + ? i % 10 + : BinaryString.fromString( + (i / 10) % 2 == 0 ? "category-a" : "category-b"); + write.write(GenericRow.of(i == 7 ? value : original)); + } + List messages = write.prepareCommit(); + setFirstRowId(messages, 0L); + commit.commit(messages); + } + } + + private void build(FileStoreTable table, String... fields) throws Exception { + List names = Arrays.asList(fields); + Optional> scan = + new SortedGlobalIndexScanner(table, "btree") + .withIndexFields(names) + .incrementalScan(); + if (!scan.isPresent()) { + return; + } + ScanResult result = scan.get(); + List messages = new ArrayList<>(); + for (DataSplit split : result.entries()) { + messages.addAll( + SortedGlobalIndexTestUtils.buildIndex( + table, "btree", names, split, result.scanSnapshotId())); + } + for (IndexManifestEntry entry : result.deletedIndexEntries()) { + messages.add( + new CommitMessageImpl( + entry.partition(), + entry.bucket(), + null, + DataIncrement.deleteIndexIncrement( + Collections.singletonList(entry.indexFile())), + CompactIncrement.emptyIncrement())); + } + try (BatchTableCommit commit = table.newBatchWriteBuilder().newCommit()) { + commit.commit(messages); + } + } + + private void deleteSingleColumnFiles(FileStoreTable table) throws Exception { + int deleted = 0; + for (IndexManifestEntry entry : table.store().newIndexFileHandler().scanEntries()) { + IndexFileMeta file = entry.indexFile(); + if (file.globalIndexMeta().extraFieldIds() == null + || file.globalIndexMeta().extraFieldIds().length == 0) { + assertThat( + table.fileIO() + .delete( + table.store() + .pathFactory() + .globalIndexFileFactory() + .toPath(file), + false)) + .isTrue(); + deleted++; + } + } + assertThat(deleted).isPositive(); + } +} diff --git a/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/globalindex/SortedIndexTopoBuilder.java b/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/globalindex/SortedIndexTopoBuilder.java index 67a73da98695..d0df9c171efe 100644 --- a/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/globalindex/SortedIndexTopoBuilder.java +++ b/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/globalindex/SortedIndexTopoBuilder.java @@ -88,6 +88,7 @@ import java.util.Map; import java.util.Optional; import java.util.function.Supplier; +import java.util.stream.Collectors; import static org.apache.paimon.globalindex.GlobalIndexBuilderUtils.groupSplitsByRange; import static org.apache.paimon.globalindex.GlobalIndexBuilderUtils.shardSplitsByRowRange; @@ -150,10 +151,36 @@ public static Optional> buildIndexStream( PartitionPredicate partitionPredicate, Options userOptions) throws Exception { + List> definitions = new ArrayList<>(); + for (String name : indexColumns) { + definitions.add(Collections.singletonList(name)); + } + return buildIndexDefinitions( + env, + indexScannerSupplier, + table, + definitions, + indexType, + partitionPredicate, + userOptions); + } + + private static Optional> buildIndexDefinitions( + StreamExecutionEnvironment env, + Supplier indexScannerSupplier, + FileStoreTable table, + List> definitions, + String indexType, + PartitionPredicate partitionPredicate, + Options userOptions) + throws Exception { List> allStreams = new ArrayList<>(); - for (String indexColumn : indexColumns) { + for (List indexColumns : definitions) { + String indexColumn = indexColumns.get(0); SortedGlobalIndexScanner indexScanner = - indexScannerSupplier.get().withIndexField(indexColumn); + indexColumns.size() == 1 + ? indexScannerSupplier.get().withIndexField(indexColumn) + : indexScannerSupplier.get().withIndexFields(indexColumns); if (partitionPredicate != null) { indexScanner = indexScanner.withPartitionPredicate(partitionPredicate); } @@ -176,14 +203,19 @@ public static Optional> buildIndexStream( // 2. Select necessary columns (index field + ROW_ID) List selectedColumns = new ArrayList<>(); - selectedColumns.add(indexColumn); + selectedColumns.addAll(indexColumns); RowType dataReadType = SpecialFields.rowTypeWithRowId(table.rowType().project(selectedColumns)); String buildTaskIdField = buildTaskIdFieldName(dataReadType); GlobalIndexer indexer = GlobalIndexer.create( - indexType, table.rowType().getField(indexColumn), userOptions); + indexType, + table.rowType().getField(indexColumn), + selectedColumns.subList(1, selectedColumns.size()).stream() + .map(table.rowType()::getField) + .collect(Collectors.toList()), + userOptions); if (!(indexer instanceof SortedGlobalIndexer)) { throw new IllegalArgumentException( "Index algorithm " + indexType + " does not expose sorted index keys."); @@ -196,7 +228,10 @@ public static Optional> buildIndexStream( int indexFieldPos = sortReadType.getFieldIndex(indexColumn); int rowIdPos = sortReadType.getFieldIndex(SpecialFields.ROW_ID.name()); DataType indexFieldType = sortReadType.getTypeAt(indexFieldPos); - DataType sourceFieldType = table.rowType().getField(indexColumn).type(); + DataType sourceFieldType = + indexColumns.size() == 1 + ? table.rowType().getField(indexColumn).type() + : table.rowType().project(indexColumns); // 3. Calculate maximum parallelism bound long recordsPerRange = @@ -245,7 +280,7 @@ public static Optional> buildIndexStream( splitTasks, readBuilder, new SortedGlobalIndexWriter(table, indexType, userOptions) - .withIndexField(indexColumn), + .withIndexFields(indexColumns), scanResult.scanSnapshotId(), partitionFieldSize, taskIdPos, @@ -319,6 +354,29 @@ public static void buildIndexAndExecute( } } + public static void buildIndexAndExecute( + StreamExecutionEnvironment env, + FileStoreTable table, + List indexColumns, + String indexType, + PartitionPredicate partitionPredicate, + Options userOptions) + throws Exception { + Optional> written = + buildIndexDefinitions( + env, + () -> new SortedGlobalIndexScanner(table, indexType, userOptions), + table, + Collections.singletonList(indexColumns), + indexType, + partitionPredicate, + userOptions); + if (written.isPresent()) { + commit(table, written.get(), CoreOptions.createCommitUser(userOptions)); + env.execute("Create " + indexType + " global index for table: " + table.name()); + } + } + protected static DataStream executeForBuildTasks( StreamExecutionEnvironment env, List buildTasks, @@ -357,7 +415,7 @@ protected static DataStream executeForBuildTasks( sortRows( env, rowDataStream, - keyExtractor.isIdentity(), + keyExtractor.isIdentity() && !(indexFieldType instanceof RowType), taskIdPos, indexFieldPos, coreOptions, @@ -516,6 +574,7 @@ private static class ReadDataOperator private transient TableRead tableRead; private transient InternalRow.FieldGetter sourceFieldGetter; + private transient InternalRow.FieldGetter[] compositeGetters; public ReadDataOperator( ReadBuilder readBuilder, @@ -530,7 +589,15 @@ public ReadDataOperator( public void open() throws Exception { super.open(); this.tableRead = readBuilder.newRead(); - this.sourceFieldGetter = InternalRow.createFieldGetter(sourceFieldType, 0); + if (sourceFieldType instanceof RowType) { + RowType tupleType = (RowType) sourceFieldType; + compositeGetters = new InternalRow.FieldGetter[tupleType.getFieldCount()]; + for (int i = 0; i < compositeGetters.length; i++) { + compositeGetters[i] = InternalRow.createFieldGetter(tupleType.getTypeAt(i), i); + } + } else { + this.sourceFieldGetter = InternalRow.createFieldGetter(sourceFieldType, 0); + } } @Override @@ -542,10 +609,24 @@ public void processElement(StreamRecord element) throws Excepti try { InternalRow row; while ((row = batch.next()) != null) { - long rowId = row.getLong(1); + long rowId = + row.getLong( + compositeGetters == null ? 1 : compositeGetters.length); + Object sourceValue; + if (compositeGetters == null) { + sourceValue = sourceFieldGetter.getFieldOrNull(row); + } else { + GenericRow tuple = new GenericRow(compositeGetters.length); + for (int i = 0; i < compositeGetters.length; i++) { + tuple.setField(i, compositeGetters[i].getFieldOrNull(row)); + } + sourceValue = + org.apache.paimon.utils.InternalRowUtils.copy( + tuple, sourceFieldType); + } boolean[] emitted = new boolean[1]; keyExtractor.extract( - sourceFieldGetter.getFieldOrNull(row), + sourceValue, key -> { emitted[0] = true; output.collect( diff --git a/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/procedure/CreateGlobalIndexProcedure.java b/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/procedure/CreateGlobalIndexProcedure.java index f93d19a0dcf8..f40fc215af81 100644 --- a/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/procedure/CreateGlobalIndexProcedure.java +++ b/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/procedure/CreateGlobalIndexProcedure.java @@ -138,7 +138,7 @@ public String[] call( SortedIndexTopoBuilder.buildIndexAndExecute( procedureContext.getExecutionEnvironment(), table, - indexColumns.get(0), + indexColumns, indexType, partitionPredicate, userOptions); diff --git a/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java b/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java index 70be19ff6dc9..80f37a826fa6 100644 --- a/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java +++ b/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java @@ -80,6 +80,42 @@ public void testBTreeIndex() throws Catalog.TableNotExistException { assertThat(sql("SELECT * FROM T WHERE id = 100")).containsOnly(Row.of(100, "name_100")); } + @Test + public void testCompositeBTreeIndex() throws Exception { + tEnv.getConfig().set(TableConfigOptions.TABLE_DML_SYNC, true); + sql( + "CREATE TABLE T_COMPOSITE (id INT, category STRING, item_number INT) WITH (" + + "'row-tracking.enabled' = 'true', 'data-evolution.enabled' = 'true', " + + "'sorted-index.records-per-file' = '13', 'btree-index.bloom-filter.enabled' = 'true')"); + String values = + IntStream.range(0, 40) + .mapToObj( + i -> + String.format( + "(%d, '%s', %d)", + i, + (i / 10) % 2 == 0 ? "category-a" : "category-b", + i % 10)) + .collect(Collectors.joining(",")); + sql("INSERT INTO T_COMPOSITE VALUES " + values); + sql( + "CALL sys.create_global_index(`table` => 'default.T_COMPOSITE', index_column => 'category,item_number', index_type => 'btree')"); + assertThat( + sql( + "SELECT id FROM T_COMPOSITE WHERE item_number = 7 AND category = 'category-a'")) + .containsExactlyInAnyOrder(Row.of(7), Row.of(27)); + assertThat(paimonTable("T_COMPOSITE").store().newIndexFileHandler().scanEntries()) + .allSatisfy( + entry -> + assertThat(entry.indexFile().globalIndexMeta().getIndexedFieldIds()) + .hasSize(2)); + sql("ALTER TABLE T_COMPOSITE SET ('global-index.query-in-reader.enabled' = 'true')"); + assertThat( + sql( + "SELECT id FROM T_COMPOSITE WHERE category = 'category-a' AND item_number = 7")) + .containsExactlyInAnyOrder(Row.of(7), Row.of(27)); + } + @Test public void testBitmapIndex() throws Catalog.TableNotExistException { sql( diff --git a/paimon-python/pypaimon/globalindex/data_evolution_global_index_scanner.py b/paimon-python/pypaimon/globalindex/data_evolution_global_index_scanner.py index 1dbbfb0154b8..84d58abdaa19 100644 --- a/paimon-python/pypaimon/globalindex/data_evolution_global_index_scanner.py +++ b/paimon-python/pypaimon/globalindex/data_evolution_global_index_scanner.py @@ -374,6 +374,9 @@ def is_supported_scalar_index(index_file): return ( index_file.global_index_meta is not None and index_file.index_type in _SUPPORTED_SCALAR_INDEX_TYPES + # Composite keys are only decoded by the Java reader. Skip these files + # so a scalar lookup cannot interpret tuple bytes as a column value. + and not getattr(index_file.global_index_meta, "extra_field_ids", None) ) diff --git a/paimon-python/pypaimon/tests/global_index_scalar_search_mode_test.py b/paimon-python/pypaimon/tests/global_index_scalar_search_mode_test.py index 84d3e9468380..e4a58c1670d7 100644 --- a/paimon-python/pypaimon/tests/global_index_scalar_search_mode_test.py +++ b/paimon-python/pypaimon/tests/global_index_scalar_search_mode_test.py @@ -25,6 +25,7 @@ ) from pypaimon.globalindex.data_evolution_global_index_scanner import ( DataEvolutionGlobalIndexScanner, + _supported_scalar_index_files, ) from pypaimon.utils.range import Range @@ -58,6 +59,14 @@ def _scanner(coverage): class ScalarGlobalIndexSearchModeTest(unittest.TestCase): + def test_composite_btree_is_skipped_before_scalar_decoding(self): + single = SimpleNamespace( + index_type="btree", global_index_meta=SimpleNamespace(extra_field_ids=None)) + composite = SimpleNamespace( + index_type="btree", global_index_meta=SimpleNamespace(extra_field_ids=[2])) + self.assertEqual([single], _supported_scalar_index_files([single, composite])) + self.assertIsNone(DataEvolutionGlobalIndexScanner.create(None, [composite])) + def test_default_values(self): options = CoreOptions(Options.from_none()) self.assertIsNone(options.global_index_search_mode()) diff --git a/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/globalindex/sorted/SortedIndexTopoBuilder.java b/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/globalindex/sorted/SortedIndexTopoBuilder.java index 7431d062f884..4c7331a142c0 100644 --- a/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/globalindex/sorted/SortedIndexTopoBuilder.java +++ b/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/globalindex/sorted/SortedIndexTopoBuilder.java @@ -45,6 +45,7 @@ import org.apache.paimon.table.source.DataSplit; import org.apache.paimon.table.source.Split; import org.apache.paimon.types.DataField; +import org.apache.paimon.types.DataType; import org.apache.paimon.types.DataTypes; import org.apache.paimon.types.RowType; import org.apache.paimon.utils.InstantiationUtil; @@ -72,6 +73,7 @@ import java.util.Map; import java.util.NoSuchElementException; import java.util.Optional; +import java.util.stream.Collectors; import static org.apache.paimon.globalindex.GlobalIndexBuilderUtils.groupSplitsByRange; import static org.apache.paimon.globalindex.GlobalIndexBuilderUtils.shardSplitsByRowRange; @@ -101,9 +103,37 @@ public List buildIndex( DataField indexField, Options options) throws IOException { + return buildIndex( + spark, + relation, + partitionPredicate, + table, + indexType, + readType, + indexField, + Collections.emptyList(), + options); + } + + @Override + public List buildIndex( + SparkSession spark, + DataSourceV2Relation relation, + PartitionPredicate partitionPredicate, + FileStoreTable table, + String indexType, + RowType readType, + DataField indexField, + List extraFields, + Options options) + throws IOException { + List indexFields = new ArrayList<>(); + indexFields.add(indexField); + indexFields.addAll(extraFields); + List indexNames = + indexFields.stream().map(DataField::name).collect(Collectors.toList()); SortedGlobalIndexScanner indexScanner = - new SortedGlobalIndexScanner(table, indexType, options) - .withIndexField(indexField.name()); + new SortedGlobalIndexScanner(table, indexType, options).withIndexFields(indexNames); if (partitionPredicate != null) { indexScanner = indexScanner.withPartitionPredicate(partitionPredicate); } @@ -131,7 +161,7 @@ public List buildIndex( int maxParallelism = options.get(SortedIndexOptions.SORTED_INDEX_BUILD_MAX_PARALLELISM); List allMessages = new ArrayList<>(); - GlobalIndexer indexer = GlobalIndexer.create(indexType, indexField, options); + GlobalIndexer indexer = GlobalIndexer.create(indexType, indexField, extraFields, options); if (!(indexer instanceof SortedGlobalIndexer)) { throw new IllegalArgumentException( "Index algorithm " + indexType + " does not expose sorted index keys."); @@ -143,8 +173,7 @@ public List buildIndex( final int partitionKeyNum = table.partitionKeys().size(); BinaryRowSerializer binaryRowSerializer = new BinaryRowSerializer(partitionKeyNum); SortedGlobalIndexWriter indexWriter = - new SortedGlobalIndexWriter(table, indexType, options) - .withIndexField(indexField.name()); + new SortedGlobalIndexWriter(table, indexType, options).withIndexFields(indexNames); final byte[] serializedWriter = InstantiationUtil.serializeObject(indexWriter); if (keyExtractor.isIdentity()) { List buildTasks = new ArrayList<>(); @@ -176,7 +205,14 @@ public List buildIndex( taskInputs.add( selected.select( functions.col(taskIdField), - functions.col(indexField.name()), + extraFields.isEmpty() + ? functions.col(indexField.name()) + : functions + .struct( + indexNames.stream() + .map(functions::col) + .toArray(Column[]::new)) + .alias(indexField.name()), functions.col(SpecialFields.ROW_ID.name()))); } } @@ -419,10 +455,7 @@ private static String buildTaskIdFieldName(RowType readType) { } private static RowType normalizedReadType( - RowType readType, - String taskIdField, - DataField sourceField, - org.apache.paimon.types.DataType keyType) { + RowType readType, String taskIdField, DataField sourceField, DataType keyType) { return RowType.of( new DataField(BUILD_TASK_ID_FIELD_ID, taskIdField, DataTypes.BIGINT().notNull()), new DataField(sourceField.id(), sourceField.name(), keyType), @@ -453,8 +486,7 @@ private static class SortedTaskInput { private InternalRow next; - private SortedTaskInput( - Iterator input, org.apache.paimon.types.DataType keyType) { + private SortedTaskInput(Iterator input, DataType keyType) { this.input = input; this.keyGetter = InternalRow.createFieldGetter(keyType, 1); advance(); diff --git a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala new file mode 100644 index 000000000000..7f5137e1cd9c --- /dev/null +++ b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala @@ -0,0 +1,93 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one + * or more contributor license agreements. See the NOTICE file + * distributed with this work for additional information + * regarding copyright ownership. The ASF licenses this file + * to you under the Apache License, Version 2.0 (the + * "License"); you may not use this file except in compliance + * with the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.paimon.spark.procedure + +import org.apache.paimon.spark.PaimonSparkTestBase + +import org.apache.spark.sql.Row + +import scala.collection.JavaConverters._ + +class CompositeBTreeIndexProcedureTest extends PaimonSparkTestBase { + + test("composite btree creation, incremental build and component refresh") { + withTable("T", "S") { + sql("""CREATE TABLE T (id INT, category STRING, item_number INT) + |TBLPROPERTIES ('bucket' = '-1', 'row-tracking.enabled' = 'true', + |'data-evolution.enabled' = 'true', 'global-index.column-update-action' = 'IGNORE', + |'sorted-index.records-per-file' = '13', 'btree-index.bloom-filter.enabled' = 'true') + |""".stripMargin) + def insert(from: Int, to: Int): Unit = { + val values = (from until to) + .map { + i => + val categoryValue = if ((i / 10) % 2 == 0) "category-a" else "category-b" + s"($i, '$categoryValue', ${i % 10})" + } + .mkString(",") + sql(s"INSERT INTO T VALUES $values") + } + def build(columns: String): Unit = { + sql( + s"CALL sys.create_global_index(table => 'test.T', index_column => '$columns', index_type => 'btree')") + } + insert(0, 40) + build("category") + build("item_number") + build("category,item_number") + val indexes = loadTable("T").store().newIndexFileHandler().scanEntries().asScala + assert(indexes.exists(_.indexFile().globalIndexMeta().getIndexedFieldIds().size() == 2)) + for (inReader <- Seq(false, true)) { + sql( + s"ALTER TABLE T SET TBLPROPERTIES ('global-index.query-in-reader.enabled' = '$inReader')") + checkAnswer( + sql("SELECT id FROM T WHERE item_number = 7 AND category = 'category-a'"), + Seq(Row(7), Row(27))) + } + insert(40, 60) + build("category,item_number") + checkAnswer( + sql("SELECT id FROM T WHERE item_number = 7 AND category = 'category-a'"), + Seq(Row(7), Row(27), Row(47))) + sql("CREATE TABLE S (id INT, item_number INT)") + sql("INSERT INTO S VALUES (7, 107)") + sql( + "MERGE INTO T USING S ON T.id = S.id WHEN MATCHED THEN UPDATE SET T.item_number = S.item_number") + build("category,item_number") + checkAnswer( + sql("SELECT id FROM T WHERE category = 'category-a' AND item_number = 107"), + Seq(Row(7))) + checkAnswer( + sql("SELECT id FROM T WHERE category = 'category-a' AND item_number = 7"), + Seq(Row(27), Row(47))) + val snapshot = loadTable("T").snapshotManager().latestSnapshot().id() + build("category,item_number") + assert(loadTable("T").snapshotManager().latestSnapshot().id() == snapshot) + sql( + "CALL sys.drop_global_index(table => 'test.T', index_column => 'category,item_number', index_type => 'btree')") + assert( + loadTable("T") + .store() + .newIndexFileHandler() + .scanEntries() + .asScala + .forall(_.indexFile().globalIndexMeta().getIndexedFieldIds().size() == 1)) + } + } +} From 250f1d7cb04fd7091ecf39a30183b837ebbb0ba2 Mon Sep 17 00:00:00 2001 From: JingsongLi Date: Thu, 1 Oct 2026 23:53:22 +0800 Subject: [PATCH 2/4] [core] Fix composite BTree index review findings --- .../multimodal-table/global-index/btree.mdx | 8 + .../globalindex/DataEvolutionBatchScan.java | 4 +- .../DataEvolutionGlobalIndexCoverage.java | 3 + .../DataEvolutionGlobalIndexScanner.java | 54 +++- .../paimon/globalindex/GlobalIndexQuery.java | 84 +++++- .../AbstractDataEvolutionVectorRead.java | 4 +- .../source/DataEvolutionFullTextRead.java | 2 +- .../paimon/table/CompositeBTreeTableTest.java | 259 +++++++++++++++++- .../source/FullTextSearchBuilderTest.java | 62 +++++ .../VectorSearchRowFilterExactnessTest.java | 48 ++++ .../globalindex/SortedIndexTopoBuilder.java | 2 +- .../paimon/flink/SortedGlobalIndexITCase.java | 12 +- .../pypaimon/manifest/index_manifest_file.py | 11 +- .../tests/index_manifest_write_test.py | 35 +++ .../sorted/SortedIndexTopoBuilder.java | 1 + .../CompositeBTreeIndexProcedureTest.scala | 30 ++ 16 files changed, 586 insertions(+), 33 deletions(-) diff --git a/docs/docs/multimodal-table/global-index/btree.mdx b/docs/docs/multimodal-table/global-index/btree.mdx index eb9955d97423..d5bd41ff5c92 100644 --- a/docs/docs/multimodal-table/global-index/btree.mdx +++ b/docs/docs/multimodal-table/global-index/btree.mdx @@ -169,6 +169,14 @@ Additional predicates remain filters, and OR branches can each use their own composite lookup. When several composite indexes match, the query prefers the one with the most columns. +Queries combining a composite lookup with another indexed scalar condition are +evaluated during planning, so unsupported scalar predicates and scan budgets +retain the usual fallback behavior across data splits. Pure composite lookups +can still use reader-side index evaluation. + +Vector and full-text search pre-filters currently use single-column indexes or +the usual data fallback; they do not evaluate composite BTree indexes. + Composite indexes can coexist with single-column indexes on the same columns. The current composite query path requires equality conditions for all key columns; prefix queries, ranges, and conditions on only some key columns use diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java index aa0d6559e820..00b3260e5f0d 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java @@ -389,7 +389,9 @@ private Plan planWithIndexQuery() { indexFilter, indexFiles, table.store().pathFactory().globalIndexFileFactory()); - if (indexQuery == null) { + // Scalar reader support can depend on global file coverage and scan budgets. Resolve it + // before split pruning so unsupported residuals cannot discard composite matches. + if (indexQuery == null || (indexQuery.hasCompositeQuery() && indexQuery.hasScalarQuery())) { return planEagerIndex(dataPlan, snapshot, partitionFilter, indexFiles, indexFilter); } List unindexed = diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexCoverage.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexCoverage.java index eb86c3397bd5..b959b8de04b2 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexCoverage.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexCoverage.java @@ -67,6 +67,9 @@ public DataEvolutionGlobalIndexCoverage( this.coverageByField = new HashMap<>(); for (IndexFileMeta indexFile : indexFiles) { GlobalIndexMeta meta = checkNotNull(indexFile.globalIndexMeta()); + if (!DataEvolutionGlobalIndexScanner.isIndexInSchema(table.rowType(), indexFile)) { + continue; + } // Tuple indexes cannot answer an individual column's scalar predicate. // Their coverage is supplied explicitly by the selected composite query. if ("btree".equals(indexFile.indexType()) diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java index 10434d22900f..740e3ac14d54 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java @@ -220,12 +220,30 @@ public static Optional create( return create(table, null, null, indexFiles); } + /** Search pre-filters use per-column coverage, so they require scalar BTree definitions. */ + public static Optional createForScalarFilters( + FileStoreTable table, + @Nullable Snapshot pinnedSnapshot, + @Nullable PartitionPredicate partitionFilter, + Collection indexFiles) { + return create( + table, + pinnedSnapshot, + partitionFilter, + indexFiles.stream() + .filter(file -> !isCompositeBTree(file)) + .collect(Collectors.toList())); + } + public static Optional create( FileStoreTable table, @Nullable Snapshot pinnedSnapshot, @Nullable PartitionPredicate partitionFilter, Collection indexFiles) { - List globalIndexFiles = globalIndexFiles(indexFiles); + List globalIndexFiles = + globalIndexFiles(indexFiles).stream() + .filter(file -> isIndexInSchema(table.rowType(), file)) + .collect(Collectors.toList()); if (globalIndexFiles.isEmpty()) { return Optional.empty(); } @@ -341,7 +359,8 @@ static Filter indexFileFilter( return false; } GlobalIndexMeta globalIndex = entry.indexFile().globalIndexMeta(); - if (globalIndex == null) { + if (globalIndex == null + || !isIndexInSchema(table.rowType(), entry.indexFile())) { return false; } // Collect indexes whose primary column is filtered, and also multi-column @@ -369,6 +388,12 @@ private static boolean isCompositeBTree(IndexFileMeta file) { && meta.extraFieldIds().length > 0; } + /** Dropped indexed columns leave manifest entries that cannot serve the current schema. */ + static boolean isIndexInSchema(RowType rowType, IndexFileMeta file) { + return file.globalIndexMeta().getIndexedFieldIds().stream() + .allMatch(rowType::containsField); + } + private static List globalIndexFiles(Collection indexFiles) { return indexFiles.stream() .filter(indexFile -> indexFile.globalIndexMeta() != null) @@ -389,16 +414,12 @@ public Optional scanWithCoverage(Predicate pred : GlobalIndexQuery.create(rowType, predicate, indexFiles, indexPathFactory); if (query != null && query.hasCompositeQuery()) { try { - return Optional.of( - new GlobalIndexEvaluator.Evaluation( - query.evaluate( - fileIO, - options, - indexFiles.stream() - .map(file -> file.globalIndexMeta().rowRange()) - .collect(Collectors.toList())), - query.contributingFieldIds(rowType), - query.coveredRanges())); + return query.evaluateWithCoverage( + fileIO, + options, + indexFiles.stream() + .map(file -> file.globalIndexMeta().rowRange()) + .collect(Collectors.toList())); } catch (IOException e) { throw new RuntimeException("Failed to evaluate composite BTree query", e); } @@ -426,6 +447,15 @@ public GlobalIndexResult unindexedRows(Predicate predicate) { predicate == null ? null : GlobalIndexQuery.create(rowType, predicate, indexFiles, indexPathFactory); + if (query != null && query.hasCompositeQuery() && query.hasScalarQuery()) { + // Scalar shards can decline a predicate at runtime, so the legacy API must use + // evaluated coverage as well. Internal callers retain the evaluation directly. + Optional evaluation = scanWithCoverage(predicate); + return evaluation.isPresent() + ? unindexedRowsForEvaluation(evaluation.get()) + : GlobalIndexResult.fromRanges( + coverage.unindexedRangesFromCoverage(Collections.emptyList(), null)); + } List unindexed = query != null && query.hasCompositeQuery() ? coverage.unindexedRangesFromCoverage(query.coveredRanges(), null) diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java index 006c1aebd484..b530824ec70a 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java @@ -21,6 +21,7 @@ import org.apache.paimon.fs.FileIO; import org.apache.paimon.fs.Path; import org.apache.paimon.globalindex.DataEvolutionGlobalIndexScanner.IndexMetaFileGroup; +import org.apache.paimon.globalindex.GlobalIndexEvaluator.Evaluation; import org.apache.paimon.globalindex.btree.CompositeBTreePredicate; import org.apache.paimon.index.IndexFileMeta; import org.apache.paimon.index.IndexPathFactory; @@ -107,7 +108,13 @@ static GlobalIndexQuery create( List files, IndexPathFactory pathFactory) { Map> groupsByField = - DataEvolutionGlobalIndexScanner.groupIndexFiles(files); + DataEvolutionGlobalIndexScanner.groupIndexFiles( + files.stream() + .filter( + file -> + DataEvolutionGlobalIndexScanner.isIndexInSchema( + rowType, file)) + .collect(Collectors.toList())); Map> groups = new LinkedHashMap<>(); groupsByField.forEach( (fieldId, fieldGroups) -> { @@ -186,6 +193,19 @@ private static GlobalIndexQuery createForPredicate( new GlobalIndexQuery( composite, false, Collections.emptyList(), selectedGroups)); predicates.removeAll(covered); + // The tuple equality already narrows these columns to one value. Leave their + // remaining conditions as data filters instead of reading scalar postings. + List keyEqualities = matched; + predicates.removeIf( + child -> { + if (!(child instanceof LeafPredicate)) { + return false; + } + Optional field = ((LeafPredicate) child).fieldRefOptional(); + return keyEqualities.stream() + .anyMatch( + equality -> equality.fieldRefOptional().equals(field)); + }); } } for (int i = 0; i < predicates.size(); i++) { @@ -270,6 +290,11 @@ boolean hasCompositeQuery() { || children.stream().anyMatch(GlobalIndexQuery::hasCompositeQuery); } + boolean hasScalarQuery() { + return groups.stream().anyMatch(group -> !group.isCompositeBTree()) + || children.stream().anyMatch(GlobalIndexQuery::hasScalarQuery); + } + /** Coverage of the selected query paths, including files pruned safely by key metadata. */ List coveredRanges() { if (predicate != null) { @@ -326,31 +351,65 @@ GlobalIndexResult evaluate(FileIO fileIO, Options options, List ranges) throws IOException { ExecutorService executor = GlobalIndexReadThreadPool.getExecutorService(options.get(GLOBAL_INDEX_THREAD_NUM)); - return evaluateWithExecutor(fileIO, options, ranges, executor); + return evaluateWithExecutor(fileIO, options, ranges, executor, false).get().result(); + } + + /** Preserve unsupported-reader fallback and coverage of the paths that actually contributed. */ + Optional evaluateWithCoverage(FileIO fileIO, Options options, List ranges) + throws IOException { + ExecutorService executor = + GlobalIndexReadThreadPool.getExecutorService(options.get(GLOBAL_INDEX_THREAD_NUM)); + return evaluateWithExecutor(fileIO, options, ranges, executor, true); } - private GlobalIndexResult evaluateWithExecutor( - FileIO fileIO, Options options, List ranges, ExecutorService executor) + private Optional evaluateWithExecutor( + FileIO fileIO, + Options options, + List ranges, + ExecutorService executor, + boolean allowUnsupported) throws IOException { if (predicate == null) { GlobalIndexResult result = null; + Set fields = new HashSet<>(); + List coverage = null; for (GlobalIndexQuery child : children) { - GlobalIndexResult matches = - child.evaluateWithExecutor(fileIO, options, ranges, executor); + Optional evaluation = + child.evaluateWithExecutor( + fileIO, options, ranges, executor, allowUnsupported); + if (!evaluation.isPresent()) { + if (union) { + return Optional.empty(); + } + continue; + } + GlobalIndexResult matches = evaluation.get().result(); result = result == null ? matches : union ? result.or(matches) : result.and(matches); + fields.addAll(evaluation.get().contributingFieldIds()); + List childCoverage = evaluation.get().coveredRanges(); + coverage = coverage == null ? childCoverage : Range.and(coverage, childCoverage); } - return result == null ? GlobalIndexResult.createEmpty() : result; + return result == null + ? Optional.empty() + : Optional.of(new Evaluation(result, fields, coverage)); } GlobalIndexResult result = GlobalIndexResult.createEmpty(); + Set fields = new HashSet<>(); + List coverage = new ArrayList<>(); + boolean supported = groups.isEmpty(); for (IndexGroup group : groups) { if (group.files.isEmpty()) { + supported = true; + fields.addAll(collectFieldIds(new RowType(group.fields()), predicate)); + coverage.add(group.range); continue; } Function>> query = predicateQuery(group); GlobalIndexResult splitRows = localSplitRows(ranges, group.range); if (splitRows.results().isEmpty()) { + supported = true; continue; } GlobalIndexer indexer = @@ -365,8 +424,14 @@ private GlobalIndexResult evaluateWithExecutor( executor)) { Optional matches = query.apply(reader).get(); if (!matches.isPresent()) { + if (allowUnsupported) { + continue; + } throw new IOException("Index reader does not support predicate: " + predicate); } + supported = true; + fields.addAll(collectFieldIds(new RowType(group.fields()), predicate)); + coverage.add(group.range); // Clip in index-local coordinates before offset() iterates the retained rows. result = result.or(splitRows.and(matches.get()).offset(group.range.from)); } catch (InterruptedException e) { @@ -376,7 +441,10 @@ private GlobalIndexResult evaluateWithExecutor( throw new IOException("Failed to evaluate index query split", e.getCause()); } } - return result; + return supported + ? Optional.of( + new Evaluation(result, fields, Range.sortAndMergeOverlap(coverage, true))) + : Optional.empty(); } private Function>> diff --git a/paimon-core/src/main/java/org/apache/paimon/table/source/AbstractDataEvolutionVectorRead.java b/paimon-core/src/main/java/org/apache/paimon/table/source/AbstractDataEvolutionVectorRead.java index 9acef5d9ac45..2e07eb6206a8 100644 --- a/paimon-core/src/main/java/org/apache/paimon/table/source/AbstractDataEvolutionVectorRead.java +++ b/paimon-core/src/main/java/org/apache/paimon/table/source/AbstractDataEvolutionVectorRead.java @@ -241,7 +241,7 @@ private RoaringNavigableMap64 scalarMatchedRows(List spl } Optional optionalScanner = - DataEvolutionGlobalIndexScanner.create( + DataEvolutionGlobalIndexScanner.createForScalarFilters( table, planSnapshot, partitionFilter, scalarIndexFiles); if (!optionalScanner.isPresent()) { return null; @@ -285,7 +285,7 @@ protected RoaringNavigableMap64 rawPreFilter(List splits) scalarIndexFiles.addAll(split.scalarIndexFiles()); } Optional optionalScanner = - DataEvolutionGlobalIndexScanner.create( + DataEvolutionGlobalIndexScanner.createForScalarFilters( table, planSnapshot, partitionFilter, scalarIndexFiles); if (!optionalScanner.isPresent()) { return null; diff --git a/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextRead.java b/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextRead.java index 33cf8f6c3547..8568589bcb05 100644 --- a/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextRead.java +++ b/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextRead.java @@ -259,7 +259,7 @@ private RoaringNavigableMap64 matchedRows( private Optional evaluateWithIndexes( Set scalarIndexFiles, @Nullable Snapshot planSnapshot) { Optional optionalScanner = - DataEvolutionGlobalIndexScanner.create( + DataEvolutionGlobalIndexScanner.createForScalarFilters( table, planSnapshot, partitionFilter, scalarIndexFiles); if (!optionalScanner.isPresent()) { return Optional.empty(); diff --git a/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java b/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java index da68f32d3848..659260099bcc 100644 --- a/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java +++ b/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java @@ -21,16 +21,19 @@ import org.apache.paimon.CoreOptions; import org.apache.paimon.data.BinaryString; import org.apache.paimon.data.GenericRow; +import org.apache.paimon.globalindex.DataEvolutionGlobalIndexCoverage; import org.apache.paimon.globalindex.DataEvolutionGlobalIndexScanner; import org.apache.paimon.globalindex.ScanResult; import org.apache.paimon.globalindex.sorted.SortedGlobalIndexScanner; import org.apache.paimon.globalindex.sorted.SortedGlobalIndexTestUtils; +import org.apache.paimon.index.GlobalIndexMeta; import org.apache.paimon.index.IndexFileMeta; import org.apache.paimon.io.CompactIncrement; import org.apache.paimon.io.DataIncrement; import org.apache.paimon.manifest.IndexManifestEntry; import org.apache.paimon.predicate.Predicate; import org.apache.paimon.predicate.PredicateBuilder; +import org.apache.paimon.schema.SchemaChange; import org.apache.paimon.table.sink.BatchTableCommit; import org.apache.paimon.table.sink.BatchTableWrite; import org.apache.paimon.table.sink.BatchWriteBuilder; @@ -44,6 +47,9 @@ import org.apache.paimon.utils.Range; import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.CsvSource; +import org.junit.jupiter.params.provider.ValueSource; import java.util.ArrayList; import java.util.Arrays; @@ -81,6 +87,24 @@ void testCoexistsAndAvoidsSingleColumnPostings() throws Exception { assertThat(read(configured, query(predicates, "missing", 7))).isEmpty(); assertThat(read(configured, PredicateBuilder.and(query, predicates.equal(0, 8)))) .isEmpty(); + assertThat( + read( + configured, + PredicateBuilder.and( + query, + predicates.isNotNull(1), + predicates.startsWith( + 1, BinaryString.fromString("category-")), + predicates.greaterThan(0, 5)))) + .containsExactlyInAnyOrder("p7", "p27", "p47", "p67", "p87"); + assertThat( + read( + configured, + PredicateBuilder.and( + query, + predicates.startsWith( + 1, BinaryString.fromString("missing"))))) + .isEmpty(); } try (DataEvolutionGlobalIndexScanner scanner = DataEvolutionGlobalIndexScanner.create( @@ -99,6 +123,235 @@ void testCoexistsAndAvoidsSingleColumnPostings() throws Exception { } } + @Test + void testUnsupportedScalarPredicatesRetainCompositeFallback() throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 20); + build(table, "f1", "f0"); + append(table, 20, 40); + build(table, "f2"); + PredicateBuilder predicates = new PredicateBuilder(table.rowType()); + Predicate joint = query(predicates, "category-a", 7); + Predicate unsupported = predicates.notLike(2, BinaryString.fromString("x%")); + Predicate conjunction = PredicateBuilder.and(joint, unsupported); + Predicate union = + PredicateBuilder.or(joint, predicates.notLike(2, BinaryString.fromString("p1%"))); + FileStoreTable withoutIndex = + table.copy( + Collections.singletonMap(CoreOptions.GLOBAL_INDEX_ENABLED.key(), "false")); + for (boolean inReader : Arrays.asList(false, true)) { + assertThat(read(configured(table, "fast", inReader), conjunction)) + .containsExactly("p7"); + for (String mode : Arrays.asList("full", "detail")) { + FileStoreTable configured = configured(table, mode, inReader); + assertThat(read(configured, conjunction)).containsExactlyInAnyOrder("p7", "p27"); + assertThat(read(configured, union)) + .containsExactlyInAnyOrderElementsOf(read(withoutIndex, union)); + FileStoreTable budgeted = + configured.copy( + Collections.singletonMap( + "btree-index.fallback-scan-max-size", "0 b")); + assertThat( + read( + budgeted, + PredicateBuilder.and( + joint, + predicates.contains( + 2, BinaryString.fromString("7"))))) + .containsExactlyInAnyOrder("p7", "p27"); + } + } + } + + @Test + void testPartialResidualIndexCoverageAcrossSplits() throws Exception { + createTableDefault(); + FileStoreTable table = + table().copy(Collections.singletonMap("source.split.target-size", "1 b")); + append(table, 0, 20); + build(table, "f2"); + append(table, 20, 40); + build(table, "f1", "f0"); + PredicateBuilder predicates = new PredicateBuilder(table.rowType()); + Predicate joint = query(predicates, "category-a", 7); + Predicate unsupported = + PredicateBuilder.and(joint, predicates.notLike(2, BinaryString.fromString("x%"))); + Predicate supported = + PredicateBuilder.and(joint, predicates.startsWith(2, BinaryString.fromString("p"))); + for (boolean inReader : Arrays.asList(false, true)) { + FileStoreTable configured = configured(table, "fast", inReader); + assertThat(read(configured, unsupported)).containsExactlyInAnyOrder("p7", "p27"); + assertThat(read(configured, supported)).containsExactly("p7"); + FileStoreTable budgeted = + configured.copy( + Collections.singletonMap("btree-index.fallback-scan-max-size", "0 b")); + assertThat( + read( + budgeted, + PredicateBuilder.and( + joint, + predicates.contains(2, BinaryString.fromString("p"))))) + .containsExactlyInAnyOrder("p7", "p27"); + for (String mode : Arrays.asList("full", "detail")) { + assertThat(read(configured(table, mode, inReader), supported)) + .containsExactlyInAnyOrder("p7", "p27"); + } + } + } + + @ParameterizedTest + @ValueSource(strings = {"full", "detail"}) + void testPublicScannerFallbackUsesActualSupportedCoverage(String mode) throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 20); + build(table, "f2"); + append(table, 20, 40); + build(table, "f2"); + build(table, "f1", "f0"); + List files = new ArrayList<>(); + for (IndexManifestEntry entry : table.store().newIndexFileHandler().scanEntries()) { + IndexFileMeta file = entry.indexFile(); + if (file.globalIndexMeta().indexFieldId() == 2 + && file.globalIndexMeta().rowRangeStart() == 20) { + // Model a residual shard over the scan budget without creating a large fixture. + file = + new IndexFileMeta( + file.indexType(), + file.fileName(), + 1_000_000_000L, + file.rowCount(), + file.globalIndexMeta(), + null); + } + files.add(file); + } + FileStoreTable configured = + configured(table, mode, false) + .copy( + Collections.singletonMap( + "btree-index.fallback-scan-max-size", "16 kb")); + PredicateBuilder predicates = new PredicateBuilder(configured.rowType()); + Predicate query = + PredicateBuilder.and( + query(predicates, "category-a", 7), + predicates.contains(2, BinaryString.fromString("p"))); + try (DataEvolutionGlobalIndexScanner scanner = + DataEvolutionGlobalIndexScanner.create(configured, files).get()) { + assertThat(scanner.scan(query).get().results().toRangeList()) + .containsExactly(new Range(7, 7)); + assertThat(scanner.unindexedRows(query).results().toRangeList()) + .containsExactly(new Range(20, 39)); + } + } + + @ParameterizedTest + @ValueSource(strings = {"full", "detail"}) + void testDroppedDefinitionDoesNotCoverSurvivingExtraColumn(String mode) throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 40); + IndexFileMeta stale = + new IndexFileMeta( + "test-scalar", + "stale", + 1, + 40, + new GlobalIndexMeta(0, 39, 0, new int[] {1}, null), + null); + IndexFileMeta valid = + new IndexFileMeta( + "test-scalar", + "valid", + 1, + 20, + new GlobalIndexMeta(0, 19, 1, null, null), + null); + catalog.alterTable(identifier(), SchemaChange.dropColumn("f0"), false); + FileStoreTable evolved = configured(table(), mode, false); + DataEvolutionGlobalIndexCoverage coverage = + new DataEvolutionGlobalIndexCoverage( + evolved, + evolved.snapshotManager().latestSnapshot(), + null, + Arrays.asList(stale, valid), + evolved.coreOptions().scalarIndexSearchMode()); + assertThat(coverage.unindexedRanges(1)).containsExactly(new Range(20, 39)); + } + + @ParameterizedTest + @CsvSource({"f0,false", "f0,true", "f1,false", "f1,true"}) + void testDroppedCompositeComponentFallsBack(String dropped, boolean scalar) throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 40); + build(table, "f1", "f0"); + String surviving = dropped.equals("f0") ? "f1" : "f0"; + if (scalar) { + build(table, surviving); + } + catalog.alterTable(identifier(), SchemaChange.dropColumn(dropped), false); + FileStoreTable evolved = table(); + PredicateBuilder predicates = new PredicateBuilder(evolved.rowType()); + Predicate query = + predicates.equal( + evolved.rowType().getFieldIndex(surviving), + surviving.equals("f0") ? 7 : BinaryString.fromString("category-a")); + FileStoreTable withoutIndex = + evolved.copy( + Collections.singletonMap(CoreOptions.GLOBAL_INDEX_ENABLED.key(), "false")); + List expected = read(withoutIndex, query); + assertThat(expected).hasSize(surviving.equals("f0") ? 4 : 20); + for (boolean inReader : Arrays.asList(false, true)) { + for (String mode : Arrays.asList("fast", "full", "detail")) { + assertThat(read(configured(evolved, mode, inReader), query)) + .containsExactlyInAnyOrderElementsOf(expected); + } + } + Optional scanner = + DataEvolutionGlobalIndexScanner.create( + evolved, + evolved.store().newIndexFileHandler().scanEntries().stream() + .map(IndexManifestEntry::indexFile) + .collect(Collectors.toList())); + assertThat(scanner.isPresent()).isEqualTo(scalar); + if (scanner.isPresent()) { + scanner.get().close(); + } + } + + @Test + void testDroppedUnrelatedScalarDoesNotBreakCompositeLookup() throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 40); + build(table, "f0"); + build(table, "f1", "f2"); + catalog.alterTable(identifier(), SchemaChange.dropColumn("f0"), false); + FileStoreTable evolved = table(); + PredicateBuilder predicates = new PredicateBuilder(evolved.rowType()); + Predicate query = + PredicateBuilder.and( + predicates.equal(0, BinaryString.fromString("category-a")), + predicates.equal(1, BinaryString.fromString("p7"))); + try (DataEvolutionGlobalIndexScanner scanner = + DataEvolutionGlobalIndexScanner.create( + evolved, + evolved.store().newIndexFileHandler().scanEntries().stream() + .map(IndexManifestEntry::indexFile) + .collect(Collectors.toList())) + .get()) { + assertThat(scanner.scan(query).get().results().toRangeList()) + .containsExactly(new Range(7, 7)); + } + for (boolean inReader : Arrays.asList(false, true)) { + for (String mode : Arrays.asList("fast", "full", "detail")) { + assertThat(read(configured(evolved, mode, inReader), query)).containsExactly("p7"); + } + } + } + @Test void testPartialCompositeCoverageDoesNotBorrowSingleColumnCoverage() throws Exception { createTableDefault(); @@ -243,7 +496,11 @@ private List read(FileStoreTable table, Predicate predicate) throws Exce builder.newRead() .executeFilter() .createReader(splits) - .forEachRemaining(row -> values.add(row.getString(2).toString())); + .forEachRemaining( + row -> + values.add( + row.getString(table.rowType().getFieldIndex("f2")) + .toString())); return values; } diff --git a/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java b/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java index 281f3ce32c31..89c9fc47c651 100644 --- a/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java +++ b/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java @@ -134,6 +134,68 @@ public void testFullTextSearchEndToEnd() throws Exception { assertThat(ids).containsAnyOf(0, 1, 3); } + @ParameterizedTest + @CsvSource({"full,true", "detail,true", "full,false", "detail,false"}) + public void testPartialCompositeKeepsScalarFilterCoverage(String mode, boolean scalar) + throws Exception { + createTableDefault(); + FileStoreTable table = + getTableDefault() + .copy( + Collections.singletonMap( + CoreOptions.SCALAR_INDEX_SEARCH_MODE.key(), mode)); + String[] documents = {"alpha keyword", "beta keyword", "gamma keyword"}; + writeDocuments(table, documents); + buildAndCommitIndex(table, documents); + if (scalar) { + buildAndCommitIdBTreeIndex(table, documents.length); + buildAndCommitBTreeIndex(table, documents); + } + List fields = + Arrays.asList( + table.rowType().getField(TEXT_FIELD_NAME), table.rowType().getField("id")); + GlobalIndexSingleColumnWriter writer = + (GlobalIndexSingleColumnWriter) + GlobalIndexBuilderUtils.createIndexWriter( + table, + "btree", + fields.get(0), + fields.subList(1, fields.size()), + table.coreOptions().toConfiguration()); + writer.write(GenericRow.of(BinaryString.fromString(documents[0]), 0), 0); + List files = + GlobalIndexBuilderUtils.toIndexFileMetas( + table.fileIO(), + table.store().pathFactory().globalIndexFileFactory(), + table.coreOptions(), + new Range(0, 0), + fields, + "btree", + writer.finish(), + null); + try (BatchTableCommit commit = table.newBatchWriteBuilder().newCommit()) { + commit.commit( + Collections.singletonList( + new CommitMessageImpl( + BinaryRow.EMPTY_ROW, + 0, + null, + DataIncrement.indexIncrement(files), + CompactIncrement.emptyIncrement()))); + } + PredicateBuilder predicates = new PredicateBuilder(table.rowType()); + GlobalIndexResult result = + table.newFullTextSearchBuilder() + .withQuery(TEXT_FIELD_NAME, matchQuery("keyword")) + .withLimit(1) + .withFilter( + PredicateBuilder.and( + predicates.equal(0, 1), + predicates.equal(1, BinaryString.fromString(documents[1])))) + .executeLocal(); + assertThat(result.results()).containsExactly(1L); + } + @Test public void testFullTextSearchExcludesDeletedIndexedRows() throws Exception { Identifier identifier = identifier("full_text_deleted_indexed_rows"); diff --git a/paimon-core/src/test/java/org/apache/paimon/table/source/VectorSearchRowFilterExactnessTest.java b/paimon-core/src/test/java/org/apache/paimon/table/source/VectorSearchRowFilterExactnessTest.java index e203229c30d9..036a1b6d488b 100644 --- a/paimon-core/src/test/java/org/apache/paimon/table/source/VectorSearchRowFilterExactnessTest.java +++ b/paimon-core/src/test/java/org/apache/paimon/table/source/VectorSearchRowFilterExactnessTest.java @@ -50,6 +50,7 @@ import org.junit.jupiter.api.Test; import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.CsvSource; import org.junit.jupiter.params.provider.ValueSource; import java.util.Arrays; @@ -129,6 +130,53 @@ public void testContainsOnBTreeColumnRanksOnlyMatchingRows() throws Exception { assertThat(result.results()).containsExactly(1L); } + @ParameterizedTest + @CsvSource({"full,true", "detail,true", "full,false", "detail,false"}) + public void testPartialCompositeKeepsScalarFilterCoverage(String mode, boolean indexed) + throws Exception { + FileStoreTable table = + createTable( + "vector_composite_" + mode + indexed, + schemaBuilder(true) + .option(CoreOptions.SCALAR_INDEX_SEARCH_MODE.key(), mode) + .option(CoreOptions.VECTOR_INDEX_SEARCH_MODE.key(), "full")); + String[] names = {"alpha", "beta", "gamma"}; + float[][] vectors = {{1.0f, 0.0f}, {0.6f, 0.8f}, {0.0f, 1.0f}}; + write(table, names, vectors); + buildAndCommitVectorIndex(table, vectors, new Range(0, indexed ? 2 : 0)); + buildAndCommitNameBTreeIndex(table, names); + buildAndCommitIdBTreeIndex(table, names.length); + List fields = + Arrays.asList(table.rowType().getField("name"), table.rowType().getField("id")); + GlobalIndexSingleColumnWriter writer = + (GlobalIndexSingleColumnWriter) + GlobalIndexBuilderUtils.createIndexWriter( + table, + "btree", + fields.get(0), + fields.subList(1, fields.size()), + table.coreOptions().toConfiguration()); + writer.write(GenericRow.of(BinaryString.fromString(names[0]), 0), 0); + writer.write(GenericRow.of(BinaryString.fromString(names[1]), 1), 1); + commitIndex( + table, + GlobalIndexBuilderUtils.toIndexFileMetas( + table.fileIO(), + table.store().pathFactory().globalIndexFileFactory(), + table.coreOptions(), + new Range(0, 1), + fields, + "btree", + writer.finish(), + null)); + PredicateBuilder predicates = new PredicateBuilder(table.rowType()); + Predicate query = + PredicateBuilder.and( + predicates.equal(0, 2), + predicates.equal(1, BinaryString.fromString("gamma"))); + assertThat(search(table, query, 1)).containsExactly(2L); + } + @Test public void testPartiallyIndexedConjunctionRanksOnlyMatchingRows() throws Exception { createTableDefault(); diff --git a/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/globalindex/SortedIndexTopoBuilder.java b/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/globalindex/SortedIndexTopoBuilder.java index d0df9c171efe..1e2981f4668e 100644 --- a/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/globalindex/SortedIndexTopoBuilder.java +++ b/paimon-flink/paimon-flink-common/src/main/java/org/apache/paimon/flink/globalindex/SortedIndexTopoBuilder.java @@ -251,7 +251,7 @@ private static Optional> buildIndexDefinitions( BinaryRow partition = partitionEntry.getKey(); byte[] partitionBytes = binaryRowSerializer.serializeToBytes(partition); Map> ranges = - keyExtractor.isIdentity() + keyExtractor.isIdentity() && !(indexFieldType instanceof RowType) ? partitionEntry.getValue() : shardSplitsByRowRange(partitionEntry.getValue(), recordsPerRange); for (Map.Entry> entry : ranges.entrySet()) { diff --git a/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java b/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java index 80f37a826fa6..9eb127cddb22 100644 --- a/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java +++ b/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java @@ -86,7 +86,8 @@ public void testCompositeBTreeIndex() throws Exception { sql( "CREATE TABLE T_COMPOSITE (id INT, category STRING, item_number INT) WITH (" + "'row-tracking.enabled' = 'true', 'data-evolution.enabled' = 'true', " - + "'sorted-index.records-per-file' = '13', 'btree-index.bloom-filter.enabled' = 'true')"); + + "'sorted-index.records-per-file' = '13', 'sorted-index.build.max-parallelism' = '4', " + + "'btree-index.bloom-filter.enabled' = 'true')"); String values = IntStream.range(0, 40) .mapToObj( @@ -106,9 +107,12 @@ public void testCompositeBTreeIndex() throws Exception { .containsExactlyInAnyOrder(Row.of(7), Row.of(27)); assertThat(paimonTable("T_COMPOSITE").store().newIndexFileHandler().scanEntries()) .allSatisfy( - entry -> - assertThat(entry.indexFile().globalIndexMeta().getIndexedFieldIds()) - .hasSize(2)); + entry -> { + assertThat(entry.indexFile().globalIndexMeta().getIndexedFieldIds()) + .hasSize(2); + assertThat(entry.indexFile().globalIndexMeta().rowRange().count()) + .isLessThanOrEqualTo(13); + }); sql("ALTER TABLE T_COMPOSITE SET ('global-index.query-in-reader.enabled' = 'true')"); assertThat( sql( diff --git a/paimon-python/pypaimon/manifest/index_manifest_file.py b/paimon-python/pypaimon/manifest/index_manifest_file.py index 3c1c9448bac7..1adb1edeb672 100644 --- a/paimon-python/pypaimon/manifest/index_manifest_file.py +++ b/paimon-python/pypaimon/manifest/index_manifest_file.py @@ -332,7 +332,8 @@ def _validate_retained_global_index_files( or added_meta is None or added_file.index_type != retained_file.index_type or retained_meta.index_field_id != added_meta.index_field_id - or _can_keep_existing_global_index(retained_meta, added_meta) + or _can_keep_existing_global_index( + retained_meta, added_meta, added_file.index_type) ): continue @@ -386,9 +387,13 @@ def _is_global_index(index_type: str) -> bool: return index_type not in (IndexManifestFile.DELETION_VECTORS_INDEX, _HASH_INDEX) -def _can_keep_existing_global_index(retained_meta, added_meta) -> bool: +def _can_keep_existing_global_index(retained_meta, added_meta, index_type: str) -> bool: + retained_fields = _extra_field_ids(retained_meta) + added_fields = _extra_field_ids(added_meta) + if index_type == 'btree' and retained_fields != added_fields: + return True return ( - _extra_field_ids(retained_meta) == _extra_field_ids(added_meta) + retained_fields == added_fields and not _row_ranges_intersect(retained_meta, added_meta) ) diff --git a/paimon-python/pypaimon/tests/index_manifest_write_test.py b/paimon-python/pypaimon/tests/index_manifest_write_test.py index e6e428f6b593..316dedddf688 100644 --- a/paimon-python/pypaimon/tests/index_manifest_write_test.py +++ b/paimon-python/pypaimon/tests/index_manifest_write_test.py @@ -126,6 +126,41 @@ def test_combine_all_deleted_returns_none(self): previous = imf.write([self._entry('idx-a', 1)]) self.assertIsNone(imf.combine_deletes(previous, [self._entry('idx-a', 1)])) + def test_composite_btree_coexists_with_distinct_definitions(self): + for retained_fields, added_fields in [ + (None, [2]), ([2], None), ([2], [3]), ([2, 3], [3, 2])]: + with self.subTest(retained_fields=retained_fields, added_fields=added_fields): + imf = IndexManifestFile(self._table()) + retained = self._entry('retained', 1) + retained.index_file.index_type = 'btree' + retained.index_file.global_index_meta.extra_field_ids = retained_fields + added = self._entry('added', 1) + added.index_file.index_type = 'btree' + added.index_file.global_index_meta.extra_field_ids = added_fields + previous = imf.write([retained]) + updated = imf.combine_changes(previous, [added], []) + self.assertEqual({'retained', 'added'}, + {entry.index_file.file_name for entry in imf.read(updated)}) + + def test_same_composite_definition_still_rejects_overlap(self): + imf = IndexManifestFile(self._table()) + retained = self._entry('retained', 1) + added = self._entry('added', 1) + retained.index_file.index_type = added.index_file.index_type = 'btree' + previous = imf.write([retained]) + with self.assertRaisesRegex(RuntimeError, 'overlapping row range'): + imf.combine_changes(previous, [added], []) + + def test_non_btree_distinct_definitions_still_reject_overlap(self): + imf = IndexManifestFile(self._table()) + retained = self._entry('retained', 1) + added = self._entry('added', 1) + retained.index_file.index_type = added.index_file.index_type = 'bitmap' + added.index_file.global_index_meta.extra_field_ids = [3] + previous = imf.write([retained]) + with self.assertRaisesRegex(RuntimeError, 'overlapping row range'): + imf.combine_changes(previous, [added], []) + if __name__ == '__main__': unittest.main() diff --git a/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/globalindex/sorted/SortedIndexTopoBuilder.java b/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/globalindex/sorted/SortedIndexTopoBuilder.java index 4c7331a142c0..460f2870f103 100644 --- a/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/globalindex/sorted/SortedIndexTopoBuilder.java +++ b/paimon-spark/paimon-spark-common/src/main/java/org/apache/paimon/spark/globalindex/sorted/SortedIndexTopoBuilder.java @@ -127,6 +127,7 @@ public List buildIndex( List extraFields, Options options) throws IOException { + extraFields = extraFields == null ? Collections.emptyList() : extraFields; List indexFields = new ArrayList<>(); indexFields.add(indexField); indexFields.addAll(extraFields); diff --git a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala index 7f5137e1cd9c..33d40458dad3 100644 --- a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala +++ b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala @@ -18,14 +18,44 @@ package org.apache.paimon.spark.procedure +import org.apache.paimon.options.Options import org.apache.paimon.spark.PaimonSparkTestBase +import org.apache.paimon.spark.globalindex.sorted.SortedIndexTopoBuilder +import org.apache.paimon.types.DataField import org.apache.spark.sql.Row +import java.util.Collections + import scala.collection.JavaConverters._ class CompositeBTreeIndexProcedureTest extends PaimonSparkTestBase { + test("nullable extra fields preserve single column empty builds") { + withTable("T") { + sql("""CREATE TABLE T (id INT) + |TBLPROPERTIES ('bucket' = '-1', 'row-tracking.enabled' = 'true', + |'data-evolution.enabled' = 'true') + |""".stripMargin) + val table = loadTable("T") + for (extras <- Seq(null, Collections.emptyList[DataField]())) { + assert( + new SortedIndexTopoBuilder() + .buildIndex( + null, + null, + null, + table, + "btree", + table.rowType(), + table.rowType().getField("id"), + extras, + new Options()) + .isEmpty) + } + } + } + test("composite btree creation, incremental build and component refresh") { withTable("T", "S") { sql("""CREATE TABLE T (id INT, category STRING, item_number INT) From 4e34273c094fe546f882826d70b295adecaad899 Mon Sep 17 00:00:00 2001 From: JingsongLi Date: Fri, 2 Oct 2026 09:44:40 +0800 Subject: [PATCH 3/4] [core] Fix composite BTree CI regressions --- .../paimon/globalindex/KeySerializer.java | 6 -- .../globalindex/btree/BTreeGlobalIndexer.java | 6 +- .../btree/BTreeGlobalIndexerFactory.java | 3 +- .../btree/CompositeBTreeIndexTest.java | 14 +++- .../sorted/SortedGlobalIndexTestUtils.java | 6 +- .../data_evolution_global_index_scanner.py | 3 +- .../global_index_scalar_search_mode_test.py | 5 +- .../tests/vector_search_filter_test.py | 66 +++++++++---------- 8 files changed, 62 insertions(+), 47 deletions(-) diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/KeySerializer.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/KeySerializer.java index a33283d1197d..a58b573d9de0 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/KeySerializer.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/KeySerializer.java @@ -35,7 +35,6 @@ import org.apache.paimon.types.FloatType; import org.apache.paimon.types.IntType; import org.apache.paimon.types.LocalZonedTimestampType; -import org.apache.paimon.types.RowType; import org.apache.paimon.types.SmallIntType; import org.apache.paimon.types.TimeType; import org.apache.paimon.types.TimestampType; @@ -65,11 +64,6 @@ public KeySerializer defaultMethod(DataType dataType) { "DataType: " + dataType + " is not supported by global index now."); } - @Override - public KeySerializer visit(RowType rowType) { - return new CompositeKeySerializer(rowType); - } - @Override public KeySerializer visit(CharType charType) { return new StringSerializer(); diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexer.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexer.java index b46e04f86d20..e4c632d02b89 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexer.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexer.java @@ -20,6 +20,7 @@ import org.apache.paimon.compression.BlockCompressionFactory; import org.apache.paimon.compression.CompressOptions; +import org.apache.paimon.globalindex.CompositeKeySerializer; import org.apache.paimon.globalindex.GlobalIndexIOMeta; import org.apache.paimon.globalindex.GlobalIndexKeyExtractor; import org.apache.paimon.globalindex.GlobalIndexReader; @@ -93,7 +94,10 @@ public BTreeGlobalIndexer(DataField dataField, List extraFields, Opti } } DataType keyType = extraFields.isEmpty() ? dataField.type() : new RowType(fields); - this.keySerializer = KeySerializer.create(keyType); + this.keySerializer = + extraFields.isEmpty() + ? KeySerializer.create(keyType) + : new CompositeKeySerializer((RowType) keyType); this.keyExtractor = GlobalIndexKeyExtractor.identity(keyType); this.options = options; this.fallbackScanMaxSize = diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java index 9b5b8572f4c5..e661650e848d 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java @@ -19,6 +19,7 @@ package org.apache.paimon.globalindex.btree; import org.apache.paimon.data.GenericRow; +import org.apache.paimon.globalindex.CompositeKeySerializer; import org.apache.paimon.globalindex.GlobalIndexIOMeta; import org.apache.paimon.globalindex.GlobalIndexer; import org.apache.paimon.globalindex.GlobalIndexerFactory; @@ -67,7 +68,7 @@ public List selectFiles( if (files.stream().anyMatch(file -> file.metadata() == null)) { return files; } - KeySerializer serializer = KeySerializer.create(new RowType(fields)); + KeySerializer serializer = new CompositeKeySerializer(new RowType(fields)); Object[] values = matched.get().stream().map(leaf -> leaf.literals().get(0)).toArray(); return new SortedFileMetaSelector(files, serializer) .visitEqual( diff --git a/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java b/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java index 55e903ac7488..a54d3b980126 100644 --- a/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java +++ b/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java @@ -23,6 +23,7 @@ import org.apache.paimon.fs.Path; import org.apache.paimon.fs.PositionOutputStream; import org.apache.paimon.fs.local.LocalFileIO; +import org.apache.paimon.globalindex.CompositeKeySerializer; import org.apache.paimon.globalindex.GlobalIndexIOMeta; import org.apache.paimon.globalindex.GlobalIndexReader; import org.apache.paimon.globalindex.GlobalIndexSingleColumnWriter; @@ -51,6 +52,7 @@ import static org.apache.paimon.shade.guava30.com.google.common.util.concurrent.MoreExecutors.newDirectExecutorService; import static org.assertj.core.api.Assertions.assertThat; +import static org.assertj.core.api.Assertions.assertThatThrownBy; /** Tests for typed composite BTree keys. */ class CompositeBTreeIndexTest { @@ -183,7 +185,8 @@ void testCompositeKeysPreserveTypesBoundariesAndNulls() { type.getFields().get(0), type.getFields().subList(1, 3), new Options()); - KeySerializer serializer = KeySerializer.create(indexer.keyExtractor().keyType()); + KeySerializer serializer = + new CompositeKeySerializer((RowType) indexer.keyExtractor().keyType()); Comparator comparator = serializer.createComparator(); GenericRow first = row("category-a", -1, "a\u0000b"); GenericRow second = row("category-a", 107, ""); @@ -206,6 +209,15 @@ void testCompositeKeysPreserveTypesBoundariesAndNulls() { .isInstanceOf(BTreeGlobalIndexer.class); } + @Test + void testCompositeSupportDoesNotEnablePhysicalRowIndexes() { + DataField field = new DataField(40, "nested", RowType.of(DataTypes.INT())); + for (String indexType : Arrays.asList("btree", "bitmap")) { + assertThatThrownBy(() -> GlobalIndexer.create(indexType, field, new Options())) + .isInstanceOf(UnsupportedOperationException.class); + } + } + private GenericRow row(String category, int itemNumber, String tag) { return GenericRow.of( category == null ? null : BinaryString.fromString(category), diff --git a/paimon-core/src/test/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexTestUtils.java b/paimon-core/src/test/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexTestUtils.java index a77b43560873..58a98ade99b5 100644 --- a/paimon-core/src/test/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexTestUtils.java +++ b/paimon-core/src/test/java/org/apache/paimon/globalindex/sorted/SortedGlobalIndexTestUtils.java @@ -20,6 +20,7 @@ import org.apache.paimon.data.GenericRow; import org.apache.paimon.data.InternalRow; +import org.apache.paimon.globalindex.CompositeKeySerializer; import org.apache.paimon.globalindex.GlobalIndexKeyExtractor; import org.apache.paimon.globalindex.KeySerializer; import org.apache.paimon.reader.RecordReader; @@ -108,7 +109,10 @@ public static List buildIndex( } Comparator comparator = - KeySerializer.create(keyExtractor.keyType()).createComparator(); + (keyExtractor.keyType() instanceof RowType + ? new CompositeKeySerializer((RowType) keyExtractor.keyType()) + : KeySerializer.create(keyExtractor.keyType())) + .createComparator(); rows.sort( (left, right) -> { if (left.getKey() == null) { diff --git a/paimon-python/pypaimon/globalindex/data_evolution_global_index_scanner.py b/paimon-python/pypaimon/globalindex/data_evolution_global_index_scanner.py index 84d58abdaa19..d770bc278ef9 100644 --- a/paimon-python/pypaimon/globalindex/data_evolution_global_index_scanner.py +++ b/paimon-python/pypaimon/globalindex/data_evolution_global_index_scanner.py @@ -376,7 +376,8 @@ def is_supported_scalar_index(index_file): and index_file.index_type in _SUPPORTED_SCALAR_INDEX_TYPES # Composite keys are only decoded by the Java reader. Skip these files # so a scalar lookup cannot interpret tuple bytes as a column value. - and not getattr(index_file.global_index_meta, "extra_field_ids", None) + and not (index_file.index_type == "btree" + and getattr(index_file.global_index_meta, "extra_field_ids", None)) ) diff --git a/paimon-python/pypaimon/tests/global_index_scalar_search_mode_test.py b/paimon-python/pypaimon/tests/global_index_scalar_search_mode_test.py index e4a58c1670d7..51b22e6e163d 100644 --- a/paimon-python/pypaimon/tests/global_index_scalar_search_mode_test.py +++ b/paimon-python/pypaimon/tests/global_index_scalar_search_mode_test.py @@ -64,7 +64,10 @@ def test_composite_btree_is_skipped_before_scalar_decoding(self): index_type="btree", global_index_meta=SimpleNamespace(extra_field_ids=None)) composite = SimpleNamespace( index_type="btree", global_index_meta=SimpleNamespace(extra_field_ids=[2])) - self.assertEqual([single], _supported_scalar_index_files([single, composite])) + bitmap = SimpleNamespace( + index_type="bitmap", global_index_meta=SimpleNamespace(extra_field_ids=[2])) + self.assertEqual([single, bitmap], + _supported_scalar_index_files([single, composite, bitmap])) self.assertIsNone(DataEvolutionGlobalIndexScanner.create(None, [composite])) def test_default_values(self): diff --git a/paimon-python/pypaimon/tests/vector_search_filter_test.py b/paimon-python/pypaimon/tests/vector_search_filter_test.py index 837e792ea7fa..0c7a9adeced3 100644 --- a/paimon-python/pypaimon/tests/vector_search_filter_test.py +++ b/paimon-python/pypaimon/tests/vector_search_filter_test.py @@ -1767,7 +1767,7 @@ def get_latest_snapshot(self_inner): self.assertEqual("ivf-flat", raw[0].index_type) def test_scan_attaches_scalar_index_when_filter_hits_extra_field(self): - id_name_index = _entry(None, field_id=2, index_type="btree", + id_name_index = _entry(None, field_id=2, index_type="bitmap", file_name="name-id.index", row_range_start=0, row_range_end=9) @@ -1802,7 +1802,7 @@ def test_scan_attaches_scalar_index_when_filter_hits_extra_field(self): class VectorSearchMultiShardScalarTest(unittest.TestCase): - """Scalar pre-filter across multiple btree shards of the same field. + """Scalar pre-filter across multiple index shards of the same field. Exercises the real DataEvolutionGlobalIndexScanner reader-construction path (with OffsetGlobalIndexReader + UnionGlobalIndexReader wrapping) so that: @@ -1893,10 +1893,10 @@ def test_primary_and_extra_field_indexes_share_coverage(self): ) fields = [_field(0, "a"), _field(1, "b"), _field(2, "c")] - primary = _entry(None, field_id=2, index_type="btree", + primary = _entry(None, field_id=2, index_type="bitmap", file_name="c-primary.index", row_range_start=0, row_range_end=4).index_file - extra = _entry(None, field_id=0, index_type="btree", + extra = _entry(None, field_id=0, index_type="bitmap", file_name="a-c.index", row_range_start=5, row_range_end=9).index_file extra.global_index_meta.extra_field_ids = [2] @@ -2024,11 +2024,11 @@ def test_extra_field_groups_are_padded_before_and(self): b_field = _field(1, "b") c_field = _field(2, "c") - short = _entry(None, field_id=0, index_type="btree", + short = _entry(None, field_id=0, index_type="bitmap", file_name="a-c.index", row_range_start=0, row_range_end=4).index_file short.global_index_meta.extra_field_ids = [2] - long = _entry(None, field_id=1, index_type="btree", + long = _entry(None, field_id=1, index_type="bitmap", file_name="b-c.index", row_range_start=0, row_range_end=9).index_file long.global_index_meta.extra_field_ids = [2] @@ -2087,15 +2087,15 @@ def test_extra_field_groups_pad_missing_coverage_before_and(self): b_field = _field(1, "b") c_field = _field(2, "c") - a_early = _entry(None, field_id=0, index_type="btree", + a_early = _entry(None, field_id=0, index_type="bitmap", file_name="a-c-early.index", row_range_start=2, row_range_end=3).index_file a_early.global_index_meta.extra_field_ids = [2] - a_late = _entry(None, field_id=0, index_type="btree", + a_late = _entry(None, field_id=0, index_type="bitmap", file_name="a-c-late.index", row_range_start=7, row_range_end=9).index_file a_late.global_index_meta.extra_field_ids = [2] - b_full = _entry(None, field_id=1, index_type="btree", + b_full = _entry(None, field_id=1, index_type="bitmap", file_name="b-c-full.index", row_range_start=0, row_range_end=9).index_file b_full.global_index_meta.extra_field_ids = [2] @@ -2155,11 +2155,11 @@ def test_extra_field_padding_does_not_convert_none_to_hits(self): b_field = _field(1, "b", "STRING") c_field = _field(2, "c", "STRING") - short = _entry(None, field_id=0, index_type="btree", + short = _entry(None, field_id=0, index_type="bitmap", file_name="a-c.index", row_range_start=0, row_range_end=4).index_file short.global_index_meta.extra_field_ids = [2] - long = _entry(None, field_id=1, index_type="btree", + long = _entry(None, field_id=1, index_type="bitmap", file_name="b-c.index", row_range_start=0, row_range_end=9).index_file long.global_index_meta.extra_field_ids = [2] @@ -2345,7 +2345,7 @@ def test_scanner_create_selects_extra_field_indexes(self): name_field = _field(0, "name", "STRING") id_field = _field(1, "id") emb_field = _field(2, "embedding", "FLOAT") - multi = _entry(None, field_id=0, index_type="btree", + multi = _entry(None, field_id=0, index_type="bitmap", file_name="name-id.index", row_range_start=0, row_range_end=9).index_file multi.global_index_meta.extra_field_ids = [1] @@ -2358,35 +2358,31 @@ def test_scanner_create_selects_extra_field_indexes(self): ) _patch_snapshot(self, table._entries) - class _StubBTreeReader: - def __init__(self_inner, key_serializer, file_io, index_path, - io_meta): - pass + from pypaimon.globalindex.global_index_reader import GlobalIndexReader - def visit_equal(self_inner, literal): - return GlobalIndexResult.create_empty() + class _StubReader(GlobalIndexReader): + def visit_equal(self_inner, field_ref, literal): + return _completed_future(GlobalIndexResult.create_empty()) def close(self_inner): pass with mock.patch( - "pypaimon.globalindex.btree.lazy_filtered_btree_reader.BTreeIndexReader", - _StubBTreeReader): - with mock.patch( - "pypaimon.globalindex.sorted_file_global_index_reader.SortedIndexFileMeta.deserialize", - return_value=BTreeIndexMeta(first_key=b'', last_key=b'zzzz', has_nulls=False)): - scanner = DataEvolutionGlobalIndexScanner.create( - table, - predicate=Predicate(method="equal", index=1, field="id", - literals=[3]), - ) - try: - self.assertIsNotNone(scanner) - readers = scanner._evaluator._readers_function(id_field) - self.assertTrue(readers) - finally: - if scanner is not None: - scanner.close() + "pypaimon.globalindex.data_evolution_global_index_scanner." + "_create_inner_readers", + return_value=[_StubReader()]): + scanner = DataEvolutionGlobalIndexScanner.create( + table, + predicate=Predicate(method="equal", index=1, field="id", + literals=[3]), + ) + try: + self.assertIsNotNone(scanner) + readers = scanner._evaluator._readers_function(id_field) + self.assertTrue(readers) + finally: + if scanner is not None: + scanner.close() class VectorSearchPartitionedFilterTest(unittest.TestCase): From c738a413214c3474f7c17334ee0b5bd19d5de123 Mon Sep 17 00:00:00 2001 From: JingsongLi Date: Fri, 2 Oct 2026 11:00:19 +0800 Subject: [PATCH 4/4] [core] Support composite BTree prefix and range queries --- .../multimodal-table/global-index/btree.mdx | 62 ++- .../globalindex/CompositeKeySerializer.java | 7 + .../paimon/globalindex/GlobalIndexReader.java | 6 + .../SortedFileGlobalIndexReader.java | 2 +- .../btree/BTreeGlobalIndexerFactory.java | 25 +- .../globalindex/btree/BTreeIndexReader.java | 38 ++ .../btree/CompositeBTreePredicate.java | 462 ++++++++++++++++-- .../btree/LazyFilteredBTreeReader.java | 24 + .../org/apache/paimon/sst/BlockIterator.java | 8 +- .../org/apache/paimon/sst/SstFileReader.java | 9 +- .../btree/CompositeBTreeIndexTest.java | 353 +++++++++++++ .../globalindex/DataEvolutionBatchScan.java | 12 +- .../DataEvolutionGlobalIndexScanner.java | 15 +- .../paimon/globalindex/GlobalIndexQuery.java | 260 ++++++---- .../globalindex/GlobalIndexQueryTest.java | 66 +++ .../paimon/table/CompositeBTreeTableTest.java | 175 ++++++- .../paimon/flink/SortedGlobalIndexITCase.java | 48 +- .../CompositeBTreeIndexProcedureTest.scala | 22 +- 18 files changed, 1412 insertions(+), 182 deletions(-) diff --git a/docs/docs/multimodal-table/global-index/btree.mdx b/docs/docs/multimodal-table/global-index/btree.mdx index d5bd41ff5c92..7e278c7b4e60 100644 --- a/docs/docs/multimodal-table/global-index/btree.mdx +++ b/docs/docs/multimodal-table/global-index/btree.mdx @@ -161,26 +161,54 @@ WHERE category = 'category-a' AND item_number = 107; ``` -The index stores typed tuples and one row-ID posting list for each distinct tuple. -An AND containing equality conditions for every indexed column uses a single -composite point lookup, regardless of the SQL condition order. This avoids -materializing the individual columns' posting lists before intersecting them. -Additional predicates remain filters, and OR branches can each use their own -composite lookup. When several composite indexes match, the query prefers the -one with the most columns. - -Queries combining a composite lookup with another indexed scalar condition are -evaluated during planning, so unsupported scalar predicates and scan budgets -retain the usual fallback behavior across data splits. Pure composite lookups -can still use reader-side index evaluation. +The index stores typed tuples in lexicographic column order and one row-ID posting +list for each distinct tuple. Conditions are matched by column name, regardless of +the SQL condition order. For an index on `(category, item_number, tag)`: + +| Conditions | Index access | +|---|---| +| Equality on all key columns | Point lookup; avoids expanding individual columns' posting lists. | +| `category = 'category-a'` | Scan the matching leading prefix. | +| `category = 'category-a' AND item_number > 107` | Equality prefix followed by a range. `>=`, `<`, `<=`, and `BETWEEN` are also supported. | +| `category IN ('category-a', 'category-b') AND item_number BETWEEN 10 AND 20` | One range for each equality/IN prefix. | +| `category IS NULL AND item_number = 107` | NULL is a fixed prefix value. `IS NOT NULL` can bound a range. | +| `category = 'category-a' AND item_number > 107 AND tag = 'x'` | Scan the prefix/range and test `tag` on index keys before decoding row IDs. | +| `category = 'category-a' AND tag = 'x'` | Scan the category prefix; test `tag` on index keys. The gap prevents `tag` from narrowing the scan interval. | +| Conditions on only `item_number` or `tag` | Use available single-column indexes or ordinary scans. No skip scan is performed. | + +Leading `=`, `IN`, and `IS NULL` conditions extend the lookup prefix. The first +range column ends interval construction. Supported conditions on later index +columns are still evaluated on keys before expanding postings; they do not +narrow the scanned interval. Other conditions remain data filters. OR branches +can each choose their own index path. String predicates such as `LIKE` can filter +keys within a selected prefix/range; they do not construct composite prefix +ranges by themselves. + +IN expansion is limited to 256 intervals per composite index lookup. The limit applies +to the product of distinct, matching values in consecutive key columns. +Exceeding the limit falls back to available single-column indexes or ordinary +scans. Prefix and range scans also obey `btree-index.fallback-scan-max-size`, +using the total file bytes remaining after min/max pruning, per index row-range +group. Complete point lookups do not use this scan budget. Budget-ineligible +composite paths are excluded before selecting an index, preserving available +single-column alternatives. + +Selection uses metadata heuristics: more bounded leading columns, more indexed +filter columns, broader row-ID coverage, complete point probes, fewer intervals, +then fewer selected file bytes and fewer total key columns. A single-column +BTree is preferred when only the composite leading column is filtered and the +single-column index covers all of the composite candidate's row ranges. This is +not a statistics-based cost optimizer. + +Composite prefix/range queries and queries combining composite and scalar +conditions, including scalar alternatives after a composite path is rejected, +are evaluated during planning, so runtime support and scan budgets +are resolved before data-split pruning. Pure composite point queries, including +bounded IN combinations and NULL prefixes, can use reader-side evaluation. Vector and full-text search pre-filters currently use single-column indexes or the usual data fallback; they do not evaluate composite BTree indexes. - Composite indexes can coexist with single-column indexes on the same columns. -The current composite query path requires equality conditions for all key -columns; prefix queries, ranges, and conditions on only some key columns use -available single-column indexes or ordinary table scans. Coverage is determined by the selected index. In `full` and `detail` modes, rows outside the composite index's coverage are scanned even when single-column @@ -201,7 +229,7 @@ supported in PyPaimon. | `btree-index.bloom-filter.enabled` | `false` | Whether to write a Bloom filter to accelerate BTree equality and `IN` lookups. | | `btree-index.cache-size` | `128 mb` | Cache size used by BTree index readers. | | `btree-index.high-priority-pool-ratio` | `0.1` | Fraction of `btree-index.cache-size` reserved for high-priority data such as index blocks; the rest caches data blocks. Must be in `[0, 1)`. | -| `btree-index.fallback-scan-max-size` | `256 mb` | Maximum total size of candidate BTree global index files to allow fallback index scans. Set to `0 b` to disable fallback scans. | +| `btree-index.fallback-scan-max-size` | `256 mb` | Maximum total size of candidate BTree global index files to allow fallback index scans, including composite prefix/range scans. Set to `0 b` to disable these scans; complete point probes remain supported. | | `btree-index.compression` | `none` | Compression algorithm used by BTree index blocks. | | `btree-index.compression-level` | `1` | Compression level used by codecs that support levels, such as `zstd`. | diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/CompositeKeySerializer.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/CompositeKeySerializer.java index 3e8e602db14f..2121d9736ba5 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/CompositeKeySerializer.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/CompositeKeySerializer.java @@ -30,12 +30,15 @@ /** Length-delimited tuple keys, compared by their typed components with nulls first. */ public class CompositeKeySerializer implements KeySerializer { + private final RowType rowType; + private final KeySerializer[] serializers; private final InternalRow.FieldGetter[] getters; private final Comparator[] comparators; @SuppressWarnings("unchecked") public CompositeKeySerializer(RowType type) { + this.rowType = type; int count = type.getFieldCount(); serializers = new KeySerializer[count]; getters = new InternalRow.FieldGetter[count]; @@ -47,6 +50,10 @@ public CompositeKeySerializer(RowType type) { } } + public RowType rowType() { + return rowType; + } + @Override public byte[] serialize(Object key) { InternalRow row = (InternalRow) key; diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexReader.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexReader.java index 4506713b9176..46b6b650d278 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexReader.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/GlobalIndexReader.java @@ -23,6 +23,7 @@ import org.apache.paimon.predicate.FullTextSearch; import org.apache.paimon.predicate.FunctionVisitor; import org.apache.paimon.predicate.LeafPredicate; +import org.apache.paimon.predicate.Predicate; import org.apache.paimon.predicate.TopN; import org.apache.paimon.predicate.VectorSearch; @@ -36,6 +37,11 @@ public interface GlobalIndexReader extends FunctionVisitor>>, Closeable { + /** Query an ordered composite key, including prefix ranges and key-only filters. */ + default CompletableFuture> visitComposite(Predicate predicate) { + return CompletableFuture.completedFuture(Optional.empty()); + } + /** Point lookup of a full composite key, with literals in index column order. */ default CompletableFuture> visitCompositeEqual( List literals) { diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/SortedFileGlobalIndexReader.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/SortedFileGlobalIndexReader.java index 3c4d80c4417c..71731aa07b93 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/SortedFileGlobalIndexReader.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/SortedFileGlobalIndexReader.java @@ -327,7 +327,7 @@ private CompletableFuture> visitParallel( return visitSelectedFiles(selector.get(), visitor); } - private CompletableFuture> visitSelectedFiles( + protected CompletableFuture> visitSelectedFiles( Optional> selectedOpt, Function> visitor) { if (!selectedOpt.isPresent()) { diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java index e661650e848d..8b0e5de2998b 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeGlobalIndexerFactory.java @@ -18,24 +18,18 @@ package org.apache.paimon.globalindex.btree; -import org.apache.paimon.data.GenericRow; -import org.apache.paimon.globalindex.CompositeKeySerializer; import org.apache.paimon.globalindex.GlobalIndexIOMeta; import org.apache.paimon.globalindex.GlobalIndexer; import org.apache.paimon.globalindex.GlobalIndexerFactory; import org.apache.paimon.globalindex.KeySerializer; import org.apache.paimon.globalindex.SortedFileMetaSelector; import org.apache.paimon.options.Options; -import org.apache.paimon.predicate.FieldRef; -import org.apache.paimon.predicate.LeafPredicate; import org.apache.paimon.predicate.Predicate; import org.apache.paimon.types.DataField; -import org.apache.paimon.types.RowType; import java.util.ArrayList; import java.util.Collections; import java.util.List; -import java.util.Optional; /** The {@link GlobalIndexerFactory} for btree index. */ public class BTreeGlobalIndexerFactory implements GlobalIndexerFactory { @@ -57,23 +51,8 @@ public List selectFiles( List fields = new ArrayList<>(); fields.add(indexField); fields.addAll(extraFields); - Optional> matched = - CompositeBTreePredicate.match(fields, predicate); - if (!matched.isPresent()) { - return files; - } - if (CompositeBTreePredicate.isContradictory(fields, predicate)) { - return Collections.emptyList(); - } - if (files.stream().anyMatch(file -> file.metadata() == null)) { - return files; - } - KeySerializer serializer = new CompositeKeySerializer(new RowType(fields)); - Object[] values = matched.get().stream().map(leaf -> leaf.literals().get(0)).toArray(); - return new SortedFileMetaSelector(files, serializer) - .visitEqual( - new FieldRef(0, indexField.name(), new RowType(fields)), - GenericRow.of(values)) + return CompositeBTreePredicate.plan(fields, predicate) + .map(plan -> plan.selectFiles(files)) .orElse(files); } return SortedFileMetaSelector.selectFiles( diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeIndexReader.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeIndexReader.java index 174df0704214..aafd04569084 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeIndexReader.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/BTreeIndexReader.java @@ -18,6 +18,7 @@ package org.apache.paimon.globalindex.btree; +import org.apache.paimon.data.InternalRow; import org.apache.paimon.fs.Path; import org.apache.paimon.fs.SeekableInputStream; import org.apache.paimon.globalindex.GlobalIndexIOMeta; @@ -339,6 +340,43 @@ public Optional visitEqual(Object literal) { return createResult(() -> pointQuery(literal)); } + public Optional visitComposite(CompositeBTreePredicate.Plan plan) { + return createResult( + () -> { + RoaringNavigableMap64 result = new RoaringNavigableMap64(); + for (CompositeBTreePredicate.Interval interval : plan.intervals()) { + if (plan.isPointLookup()) { + result.or(pointQuery(interval.pointKey())); + continue; + } + SstFileReader.SstFileIterator iterator = reader.createIterator(); + iterator.seekTo( + key -> + interval.lower() + .compareKey( + (InternalRow) + keySerializer.deserialize(key))); + BlockIterator batch; + boolean finished = false; + while (!finished && (batch = iterator.readBatch()) != null) { + while (batch.hasNext()) { + Map.Entry entry = batch.next(); + InternalRow key = + (InternalRow) keySerializer.deserialize(entry.getKey()); + if (interval.upper().compareKey(key) > 0) { + finished = true; + break; + } + if (plan.test(key)) { + addRowIdsTo(entry.getValue(), result); + } + } + } + } + return result; + }); + } + public Optional visitGreaterThan(Object literal) { return createResult(() -> rangeQuery(literal, maxKey, false, true)); } diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/CompositeBTreePredicate.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/CompositeBTreePredicate.java index 02b71aec8909..80e27295176c 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/CompositeBTreePredicate.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/CompositeBTreePredicate.java @@ -18,71 +18,459 @@ package org.apache.paimon.globalindex.btree; +import org.apache.paimon.data.GenericRow; +import org.apache.paimon.data.InternalRow; +import org.apache.paimon.globalindex.CompositeKeySerializer; +import org.apache.paimon.globalindex.GlobalIndexIOMeta; import org.apache.paimon.globalindex.KeySerializer; +import org.apache.paimon.globalindex.SortedIndexFileMeta; +import org.apache.paimon.memory.MemorySlice; +import org.apache.paimon.predicate.Between; +import org.apache.paimon.predicate.CompoundPredicate; +import org.apache.paimon.predicate.Contains; +import org.apache.paimon.predicate.EndsWith; import org.apache.paimon.predicate.Equal; +import org.apache.paimon.predicate.GreaterOrEqual; +import org.apache.paimon.predicate.GreaterThan; +import org.apache.paimon.predicate.In; +import org.apache.paimon.predicate.IsNotNull; +import org.apache.paimon.predicate.IsNull; +import org.apache.paimon.predicate.LeafFunction; import org.apache.paimon.predicate.LeafPredicate; +import org.apache.paimon.predicate.LessOrEqual; +import org.apache.paimon.predicate.LessThan; +import org.apache.paimon.predicate.Like; +import org.apache.paimon.predicate.NotBetween; +import org.apache.paimon.predicate.NotEqual; +import org.apache.paimon.predicate.NotIn; +import org.apache.paimon.predicate.NotLike; +import org.apache.paimon.predicate.Or; import org.apache.paimon.predicate.Predicate; import org.apache.paimon.predicate.PredicateBuilder; +import org.apache.paimon.predicate.StartsWith; import org.apache.paimon.types.DataField; +import org.apache.paimon.types.RowType; import java.util.ArrayList; +import java.util.Arrays; +import java.util.Collections; import java.util.Comparator; import java.util.List; import java.util.Optional; +import java.util.TreeSet; +import java.util.stream.Collectors; -/** Matches a full conjunction of equalities in physical composite-key order. */ +/** + * Equality/IN prefixes followed by one range, with remaining key predicates checked before + * postings. + */ public final class CompositeBTreePredicate { + public static final int MAX_INTERVALS = 256; + private CompositeBTreePredicate() {} - /** SQL equalities with NULL, or distinct values for the same key field, cannot match. */ - public static boolean isContradictory(List fields, Predicate predicate) { - List matched = match(fields, predicate).get(); - for (int i = 0; i < fields.size(); i++) { - Object value = matched.get(i).literals().get(0); - if (value == null) { - return true; + public static Optional plan(List fields, Predicate predicate) { + List originals = new ArrayList<>(); + List filters = new ArrayList<>(); + for (Predicate child : PredicateBuilder.splitAnd(predicate)) { + LeafPredicate leaf = asLeaf(child); + if (leaf == null + || !supported(leaf.function()) + || !leaf.fieldRefOptional().isPresent()) { + continue; + } + for (int i = 0; i < fields.size(); i++) { + DataField field = fields.get(i); + if (field.name().equals(leaf.fieldRefOptional().get().name()) + && field.type().equalsIgnoreNullable(leaf.type())) { + originals.add(child); + filters.add( + new LeafPredicate( + leaf.function(), + field.type().copy(true), + i, + field.name(), + leaf.literals())); + break; + } + } + } + List prefixes = new ArrayList<>(); + prefixes.add(new Object[0]); + int equalColumns = 0; + for (; equalColumns < fields.size(); equalColumns++) { + final int position = equalColumns; + List column = + filters.stream() + .filter(leaf -> leaf.fieldRefOptional().get().index() == position) + .collect(Collectors.toList()); + Optional> domain = pointValues(fields.get(position), column); + if (!domain.isPresent()) { + break; + } + if ((long) prefixes.size() * domain.get().size() > MAX_INTERVALS) { + return Optional.empty(); + } + List expanded = new ArrayList<>(); + for (Object[] prefix : prefixes) { + for (Object value : domain.get()) { + expanded.add(append(prefix, value)); + } } - Comparator comparator = - KeySerializer.create(fields.get(i).type()).createComparator(); - for (Predicate child : PredicateBuilder.splitAnd(predicate)) { - if (child instanceof LeafPredicate) { - LeafPredicate leaf = (LeafPredicate) child; - if (leaf.function() instanceof Equal - && leaf.fieldRefOptional().equals(matched.get(i).fieldRefOptional())) { - Object other = leaf.literals().get(0); - if (other == null || comparator.compare(value, other) != 0) { - return true; + prefixes = expanded; + } + List intervals = new ArrayList<>(); + int boundColumns = equalColumns; + for (Object[] prefix : prefixes) { + Bound lower = new Bound(fields, prefix, false); + Bound upper = new Bound(fields, prefix, true); + if (equalColumns < fields.size()) { + for (LeafPredicate leaf : filters) { + if (leaf.fieldRefOptional().get().index() != equalColumns) { + continue; + } + LeafFunction function = leaf.function(); + if (function instanceof GreaterThan + || function instanceof GreaterOrEqual + || function instanceof LessThan + || function instanceof LessOrEqual + || function instanceof Between + || function instanceof IsNotNull) { + boundColumns = equalColumns + 1; + if (!(function instanceof IsNotNull) && leaf.literals().contains(null)) { + lower = null; + break; + } + if (function instanceof GreaterThan + || function instanceof GreaterOrEqual + || function instanceof Between + || function instanceof IsNotNull) { + Object value = + function instanceof IsNotNull ? null : leaf.literals().get(0); + Bound next = + new Bound( + fields, + append(prefix, value), + function instanceof GreaterThan + || function instanceof IsNotNull); + if (next.compareTo(lower) > 0) { + lower = next; + } + } + if (function instanceof LessThan + || function instanceof LessOrEqual + || function instanceof Between) { + Object value = leaf.literals().get(function instanceof Between ? 1 : 0); + Bound next = + new Bound( + fields, + append(prefix, value), + !(function instanceof LessThan)); + if (next.compareTo(upper) < 0) { + upper = next; + } } } } } + if (lower != null && lower.compareTo(upper) < 0) { + intervals.add(new Interval(lower, upper)); + } + } + if (boundColumns == 0) { + return Optional.empty(); } - return false; + return Optional.of( + new Plan( + fields, + originals, + filters, + intervals, + boundColumns, + equalColumns == fields.size())); } - public static Optional> match(List fields, Predicate predicate) { - List conjuncts = PredicateBuilder.splitAnd(predicate); - List matched = new ArrayList<>(); - for (DataField field : fields) { - LeafPredicate equality = null; - for (Predicate conjunct : conjuncts) { - if (conjunct instanceof LeafPredicate) { - LeafPredicate leaf = (LeafPredicate) conjunct; - if (leaf.function() instanceof Equal - && leaf.fieldRefOptional().isPresent() - && field.name().equals(leaf.fieldRefOptional().get().name()) - && field.type().equalsIgnoreNullable(leaf.type())) { - equality = leaf; - break; + private static Optional> pointValues( + DataField field, List filters) { + for (LeafPredicate leaf : filters) { + LeafFunction function = leaf.function(); + if (!(function instanceof Equal) + && !(function instanceof In) + && !(function instanceof IsNull)) { + continue; + } + List values = + function instanceof IsNull ? Collections.singletonList(null) : leaf.literals(); + Comparator comparator = KeySerializer.create(field.type()).createComparator(); + TreeSet unique = + new TreeSet<>( + (a, b) -> + a == null + ? (b == null ? 0 : -1) + : b == null ? 1 : comparator.compare(a, b)); + for (Object value : values) { + if ((value == null && !(function instanceof IsNull)) + || (value == null && !field.type().isNullable())) { + continue; + } + if (filters.stream() + .allMatch( + filter -> + filter.function() + .test(filter.type(), value, filter.literals()))) { + unique.add(value); + if (unique.size() > MAX_INTERVALS) { + return Optional.of(new ArrayList<>(unique)); } } } - if (equality == null) { - return Optional.empty(); + return Optional.of(new ArrayList<>(unique)); + } + return Optional.empty(); + } + + // PredicateBuilder represents small IN lists as ORs of equalities. + private static LeafPredicate asLeaf(Predicate predicate) { + if (predicate instanceof LeafPredicate) { + return (LeafPredicate) predicate; + } + CompoundPredicate compound = (CompoundPredicate) predicate; + if (!(compound.function() instanceof Or)) { + return null; + } + LeafPredicate first = null; + List literals = new ArrayList<>(); + for (Predicate child : PredicateBuilder.splitOr(predicate)) { + if (!(child instanceof LeafPredicate)) { + return null; + } + LeafPredicate leaf = (LeafPredicate) child; + if (!(leaf.function() instanceof Equal) + || !leaf.fieldRefOptional().isPresent() + || (first != null + && !first.fieldRefOptional().equals(leaf.fieldRefOptional()))) { + return null; + } + first = leaf; + literals.add(leaf.literals().get(0)); + } + return first == null + ? null + : new LeafPredicate( + In.INSTANCE, + first.type(), + first.fieldRefOptional().get().index(), + first.fieldRefOptional().get().name(), + literals); + } + + private static boolean supported(LeafFunction function) { + return function instanceof Equal + || function instanceof In + || function instanceof IsNull + || function instanceof IsNotNull + || function instanceof GreaterThan + || function instanceof GreaterOrEqual + || function instanceof LessThan + || function instanceof LessOrEqual + || function instanceof Between + || function instanceof NotEqual + || function instanceof NotIn + || function instanceof NotBetween + || function instanceof StartsWith + || function instanceof EndsWith + || function instanceof Contains + || function instanceof Like + || function instanceof NotLike; + } + + private static Object[] append(Object[] prefix, Object value) { + Object[] result = Arrays.copyOf(prefix, prefix.length + 1); + result[prefix.length] = value; + return result; + } + + /** A virtual boundary immediately before or after every key with the specified prefix. */ + public static final class Bound { + private final Object[] values; + private final boolean after; + private final InternalRow.FieldGetter[] getters; + private final Comparator[] comparators; + + @SuppressWarnings("unchecked") + private Bound(List fields, Object[] values, boolean after) { + this.values = values; + this.after = after; + this.getters = new InternalRow.FieldGetter[values.length]; + this.comparators = new Comparator[values.length]; + for (int i = 0; i < values.length; i++) { + getters[i] = InternalRow.createFieldGetter(fields.get(i).type().copy(true), i); + comparators[i] = KeySerializer.create(fields.get(i).type()).createComparator(); + } + } + + public int compareKey(InternalRow row) { + for (int i = 0; i < values.length; i++) { + Object key = getters[i].getFieldOrNull(row); + int comparison = compare(i, key, values[i]); + if (comparison != 0) { + return comparison; + } + } + return after ? -1 : 1; + } + + private int compare(int position, Object left, Object right) { + return left == null + ? (right == null ? 0 : -1) + : right == null ? 1 : comparators[position].compare(left, right); + } + + private int compareTo(Bound other) { + int count = Math.min(values.length, other.values.length); + for (int i = 0; i < count; i++) { + int comparison = compare(i, values[i], other.values[i]); + if (comparison != 0) { + return comparison; + } + } + if (values.length != other.values.length) { + return values.length < other.values.length + ? (after ? 1 : -1) + : (other.after ? -1 : 1); + } + return Boolean.compare(after, other.after); + } + } + + /** A bounded tuple interval with virtual, unencoded endpoints. */ + public static final class Interval { + private final Bound lower; + private final Bound upper; + + private Interval(Bound lower, Bound upper) { + this.lower = lower; + this.upper = upper; + } + + public Bound lower() { + return lower; + } + + public Bound upper() { + return upper; + } + + public GenericRow pointKey() { + return GenericRow.of(lower.values); + } + } + + /** Scan intervals and key predicates, rebuilt from the query and ordered index fields. */ + public static final class Plan { + private final List fields; + private final List predicates; + private final List filters; + private final List intervals; + private final int boundColumns; + private final boolean pointLookup; + + private Plan( + List fields, + List predicates, + List filters, + List intervals, + int boundColumns, + boolean pointLookup) { + this.fields = fields; + this.predicates = predicates; + this.filters = filters; + this.intervals = intervals; + this.boundColumns = boundColumns; + this.pointLookup = pointLookup; + } + + public List predicates() { + return predicates; + } + + public List intervals() { + return intervals; + } + + public int boundColumns() { + return boundColumns; + } + + public int filterColumns() { + return (int) + filters.stream() + .map(leaf -> leaf.fieldRefOptional().get().index()) + .distinct() + .count(); + } + + public boolean isPointLookup() { + return pointLookup; + } + + public boolean test(InternalRow key) { + for (LeafPredicate leaf : filters) { + if (!leaf.test(key)) { + return false; + } + } + return true; + } + + /** Point probes are unrestricted; prefix/range scans use the selected-file byte budget. */ + public boolean canScan(List selected, long scanBudget) { + if (pointLookup || selected.isEmpty()) { + return true; + } + if (scanBudget <= 0) { + return false; + } + long remaining = scanBudget; + for (GlobalIndexIOMeta file : selected) { + if (file.fileSize() > remaining) { + return false; + } + remaining -= file.fileSize(); + } + return true; + } + + public List selectFiles(List files) { + CompositeKeySerializer serializer = new CompositeKeySerializer(new RowType(fields)); + List result = new ArrayList<>(); + for (GlobalIndexIOMeta file : files) { + if (intervals.isEmpty()) { + break; + } + if (file.metadata() == null) { + result.add(file); + continue; + } + SortedIndexFileMeta meta = SortedIndexFileMeta.deserialize(file.metadata()); + if (meta.firstKey() == null || meta.lastKey() == null) { + result.add(file); + continue; + } + InternalRow first = + (InternalRow) serializer.deserialize(MemorySlice.wrap(meta.firstKey())); + InternalRow last = + (InternalRow) serializer.deserialize(MemorySlice.wrap(meta.lastKey())); + if (intervals.stream() + .anyMatch( + interval -> + interval.lower.compareKey(last) > 0 + && interval.upper.compareKey(first) < 0)) { + result.add(file); + } } - matched.add(equality); + return result; } - return Optional.of(matched); } } diff --git a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/LazyFilteredBTreeReader.java b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/LazyFilteredBTreeReader.java index 2f945f5fbb44..0826519a7fc9 100644 --- a/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/LazyFilteredBTreeReader.java +++ b/paimon-common/src/main/java/org/apache/paimon/globalindex/btree/LazyFilteredBTreeReader.java @@ -60,6 +60,8 @@ public class LazyFilteredBTreeReader extends SortedFileGlobalIndexReader comparator; private final long totalRowCount; @Nullable private final Pair fullRangeBounds; + private final List indexFiles; + private final long fallbackScanMaxSize; public LazyFilteredBTreeReader( List files, @@ -74,6 +76,8 @@ public LazyFilteredBTreeReader( this.cacheManager = cacheManager; this.fileReader = fileReader; this.keySerializer = keySerializer; + this.indexFiles = files; + this.fallbackScanMaxSize = fallbackScanMaxSize; this.rowIdFilter = rowRanges == null ? null : GlobalIndexResult.fromRanges(rowRanges).results(); this.comparator = keySerializer.createComparator(); @@ -81,6 +85,26 @@ public LazyFilteredBTreeReader( this.fullRangeBounds = fullRangeBounds(files); } + @Override + public CompletableFuture> visitComposite( + org.apache.paimon.predicate.Predicate predicate) { + if (!(keySerializer instanceof CompositeKeySerializer)) { + return CompletableFuture.completedFuture(Optional.empty()); + } + Optional planned = + CompositeBTreePredicate.plan( + ((CompositeKeySerializer) keySerializer).rowType().getFields(), predicate); + if (!planned.isPresent()) { + return CompletableFuture.completedFuture(Optional.empty()); + } + CompositeBTreePredicate.Plan plan = planned.get(); + List selected = plan.selectFiles(indexFiles); + if (!plan.canScan(selected, fallbackScanMaxSize)) { + return CompletableFuture.completedFuture(Optional.empty()); + } + return visitSelectedFiles(Optional.of(selected), reader -> reader.visitComposite(plan)); + } + @Nullable private Pair fullRangeBounds(List files) { if (totalRowCount == 0 || files.isEmpty()) { diff --git a/paimon-common/src/main/java/org/apache/paimon/sst/BlockIterator.java b/paimon-common/src/main/java/org/apache/paimon/sst/BlockIterator.java index b477f8547d14..861f892f818e 100644 --- a/paimon-common/src/main/java/org/apache/paimon/sst/BlockIterator.java +++ b/paimon-common/src/main/java/org/apache/paimon/sst/BlockIterator.java @@ -24,6 +24,7 @@ import java.util.Iterator; import java.util.Map; import java.util.NoSuchElementException; +import java.util.function.ToIntFunction; /** An {@link Iterator} for a block. */ public class BlockIterator implements Iterator> { @@ -64,6 +65,11 @@ public void remove() { } public boolean seekTo(MemorySlice targetKey) { + return seekTo(key -> reader.comparator().compare(key, targetKey)); + } + + /** Seek using a monotone comparison to a key or virtual prefix boundary. */ + public boolean seekTo(ToIntFunction compareToTarget) { int left = 0; int right = reader.recordCount() - 1; polledPosition = -1; @@ -72,7 +78,7 @@ public boolean seekTo(MemorySlice targetKey) { while (left <= right) { int mid = left + (right - left) / 2; - int compare = reader.comparator().compare(readKey(mid), targetKey); + int compare = compareToTarget.applyAsInt(readKey(mid)); if (compare == 0) { polledPosition = mid; diff --git a/paimon-common/src/main/java/org/apache/paimon/sst/SstFileReader.java b/paimon-common/src/main/java/org/apache/paimon/sst/SstFileReader.java index 8346464f3b4f..ea0575457e4d 100644 --- a/paimon-common/src/main/java/org/apache/paimon/sst/SstFileReader.java +++ b/paimon-common/src/main/java/org/apache/paimon/sst/SstFileReader.java @@ -32,6 +32,7 @@ import java.io.Closeable; import java.io.IOException; import java.util.Comparator; +import java.util.function.ToIntFunction; import static org.apache.paimon.sst.SstFileUtils.crc32c; import static org.apache.paimon.utils.Preconditions.checkArgument; @@ -279,15 +280,19 @@ public class SstFileIterator { */ public void seekTo(byte[] key) { MemorySlice keySlice = MemorySlice.wrap(key); + seekTo(candidate -> comparator.compare(candidate, keySlice)); + } - indexIterator.seekTo(keySlice); + /** Seek to a virtual boundary without encoding it as a persisted key. */ + public void seekTo(ToIntFunction compareToTarget) { + indexIterator.seekTo(compareToTarget); if (indexIterator.hasNext()) { seekedDataBlock = getNextBlock(indexIterator); // The index block entry key is the last key of the corresponding data block. // If there is some index entry key >= targetKey, the related data block must // also contain some key >= target key, which means seekedDataBlock.hasNext() // must be true - seekedDataBlock.seekTo(keySlice); + seekedDataBlock.seekTo(compareToTarget); Preconditions.checkState(seekedDataBlock.hasNext()); } else { seekedDataBlock = null; diff --git a/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java b/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java index a54d3b980126..90ac3701a15b 100644 --- a/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java +++ b/paimon-common/src/test/java/org/apache/paimon/globalindex/btree/CompositeBTreeIndexTest.java @@ -31,13 +31,22 @@ import org.apache.paimon.globalindex.KeySerializer; import org.apache.paimon.globalindex.ResultEntry; import org.apache.paimon.globalindex.SortedGlobalIndexer; +import org.apache.paimon.globalindex.SortedIndexFileMeta; import org.apache.paimon.globalindex.io.GlobalIndexFileWriter; import org.apache.paimon.memory.MemorySlice; +import org.apache.paimon.options.MemorySize; import org.apache.paimon.options.Options; +import org.apache.paimon.predicate.In; +import org.apache.paimon.predicate.LeafPredicate; +import org.apache.paimon.predicate.Predicate; +import org.apache.paimon.predicate.PredicateBuilder; +import org.apache.paimon.sst.SstFileWriter; import org.apache.paimon.types.DataField; import org.apache.paimon.types.DataTypes; import org.apache.paimon.types.RowType; +import org.apache.paimon.utils.LongArrayList; import org.apache.paimon.utils.Range; +import org.apache.paimon.utils.RoaringNavigableMap64; import org.junit.jupiter.api.Test; import org.junit.jupiter.api.io.TempDir; @@ -47,8 +56,11 @@ import java.util.Arrays; import java.util.Collections; import java.util.Comparator; +import java.util.List; import java.util.UUID; import java.util.concurrent.ExecutorService; +import java.util.stream.Collectors; +import java.util.stream.IntStream; import static org.apache.paimon.shade.guava30.com.google.common.util.concurrent.MoreExecutors.newDirectExecutorService; import static org.assertj.core.api.Assertions.assertThat; @@ -170,6 +182,347 @@ public PositionOutputStream newOutputStream(String name) } } + @ParameterizedTest + @ValueSource(ints = {1, 2}) + void testPrefixesRangesInNullsAndKeyFilters(int version) throws Exception { + RowType type = + new RowType( + Arrays.asList( + new DataField(10, "category", DataTypes.STRING()), + new DataField(20, "item_number", DataTypes.INT()), + new DataField(30, "tag", DataTypes.STRING()))); + List rows = + Arrays.asList( + row("category-a", Integer.MIN_VALUE, "x"), + row("category-a", -1, "x"), + row("category-a", 0, "y"), + row("category-a", 7, "x"), + row("category-a", 7, "y"), + row("category-a", 8, "x"), + row("category-a", Integer.MAX_VALUE, "x"), + row("category-b", 7, "x"), + row(null, 7, "x"), + row("category-a", 9, null), + row("category-a", 10, "y"), + GenericRow.of( + BinaryString.fromString("category-a"), + null, + BinaryString.fromString("x")), + row("category-b", -1, null)); + Options options = new Options(); + options.set(BTreeIndexOptions.BTREE_INDEX_FILE_VERSION, version); + options.set(BTreeIndexOptions.BTREE_INDEX_BLOOM_FILTER_ENABLED, true); + options.set(BTreeIndexOptions.BTREE_INDEX_BLOCK_SIZE, MemorySize.ofBytes(128)); + GlobalIndexer indexer = + GlobalIndexer.create( + "btree", type.getFields().get(0), type.getFields().subList(1, 3), options); + LocalFileIO io = LocalFileIO.create(); + Path directory = new Path(tempPath.toUri()); + GlobalIndexFileWriter files = + new GlobalIndexFileWriter() { + @Override + public String newFileName(String prefix) { + return prefix + UUID.randomUUID(); + } + + @Override + public PositionOutputStream newOutputStream(String name) + throws java.io.IOException { + return io.newOutputStream(new Path(directory, name), false); + } + }; + GlobalIndexSingleColumnWriter writer = + (GlobalIndexSingleColumnWriter) indexer.createWriter(files); + List sortedRows = + IntStream.range(0, rows.size()).boxed().collect(Collectors.toList()); + Comparator comparator = new CompositeKeySerializer(type).createComparator(); + sortedRows.sort((left, right) -> comparator.compare(rows.get(left), rows.get(right))); + for (int rowId : sortedRows) { + writer.write(rows.get(rowId), rowId); + } + ResultEntry result = writer.finish().get(0); + Path path = new Path(directory, result.fileName()); + GlobalIndexIOMeta meta = + new GlobalIndexIOMeta(path, io.getFileSize(path), result.rowCount(), result.meta()); + PredicateBuilder b = new PredicateBuilder(type); + Predicate a = b.equal(0, BinaryString.fromString("category-a")); + List queries = + Arrays.asList( + a, + PredicateBuilder.and(a, b.greaterThan(1, 7)), + PredicateBuilder.and(b.lessThan(1, 0), a), + PredicateBuilder.and(a, b.greaterOrEqual(1, 0), b.lessOrEqual(1, 7)), + PredicateBuilder.and(a, b.between(1, 7, 8)), + PredicateBuilder.and(a, b.greaterThan(1, 7), b.lessOrEqual(1, 7)), + PredicateBuilder.and(a, b.greaterOrEqual(1, 7), b.lessOrEqual(1, 7)), + PredicateBuilder.and(a, b.greaterThan(1, Integer.MAX_VALUE)), + PredicateBuilder.and(a, b.lessThan(1, Integer.MIN_VALUE)), + PredicateBuilder.and(a, b.lessOrEqual(1, Integer.MAX_VALUE)), + PredicateBuilder.and(a, b.in(1, Arrays.asList(7, 7, null, 8))), + PredicateBuilder.and( + b.in( + 0, + Arrays.asList( + BinaryString.fromString("category-a"), + BinaryString.fromString("category-b"))), + b.in(1, Arrays.asList(7, 8))), + PredicateBuilder.and(a, b.isNull(1)), + PredicateBuilder.and(b.isNull(0), b.equal(1, 7)), + PredicateBuilder.and(a, b.isNotNull(1)), + PredicateBuilder.and( + a, b.greaterThan(1, 7), b.equal(2, BinaryString.fromString("x"))), + PredicateBuilder.and(a, b.equal(2, BinaryString.fromString("x"))), + PredicateBuilder.and( + a, b.equal(1, 7), b.notLike(2, BinaryString.fromString("x%"))), + PredicateBuilder.and(a, b.equal(1, 7), b.isNull(2)), + PredicateBuilder.and(a, b.equal(1, 7), b.equal(1, 8)), + PredicateBuilder.and(a, b.equal(1, null)), + PredicateBuilder.and(a, b.greaterThan(1, null))); + ExecutorService executor = newDirectExecutorService(); + for (List allowed : + Arrays.asList(null, Collections.singletonList(new Range(3, 10)))) { + try (GlobalIndexReader reader = + indexer.createReader( + file -> io.newInputStream(file.filePath()), + Collections.singletonList(meta), + rows.size(), + allowed, + executor)) { + for (Predicate query : queries) { + RoaringNavigableMap64 expected = new RoaringNavigableMap64(); + for (int i = 0; i < rows.size(); i++) { + if (query.test(rows.get(i)) && (allowed == null || (i >= 3 && i <= 10))) { + expected.add(i); + } + } + assertThat(reader.visitComposite(query).get().get().results().toRangeList()) + .as("predicate %s; allowed %s", query, allowed) + .containsExactlyElementsOf(expected.toRangeList()); + } + assertThat(reader.visitComposite(b.equal(1, 7)).get()).isEmpty(); + Predicate largeIn = + new LeafPredicate( + In.INSTANCE, + DataTypes.INT(), + 1, + "item_number", + IntStream.range(0, CompositeBTreePredicate.MAX_INTERVALS + 1) + .boxed() + .map(value -> (Object) value) + .collect(Collectors.toList())); + assertThat(reader.visitComposite(PredicateBuilder.and(a, largeIn)).get()).isEmpty(); + } + } + Options noScan = new Options(); + noScan.set(BTreeIndexOptions.BTREE_INDEX_FALLBACK_SCAN_MAX_SIZE, MemorySize.ofBytes(0)); + GlobalIndexer budgeted = + GlobalIndexer.create( + "btree", type.getFields().get(0), type.getFields().subList(1, 3), noScan); + try (GlobalIndexReader reader = + budgeted.createReader( + file -> io.newInputStream(file.filePath()), + Collections.singletonList(meta), + rows.size(), + null, + executor)) { + assertThat(reader.visitComposite(PredicateBuilder.and(a, b.greaterThan(1, 7))).get()) + .isEmpty(); + assertThat( + reader.visitComposite( + PredicateBuilder.and( + b.equal(0, BinaryString.fromString("missing")), + b.greaterThan(1, 7))) + .get() + .get() + .results() + .toRangeList()) + .isEmpty(); + assertThat( + reader.visitComposite( + PredicateBuilder.and( + a, + b.equal(1, 7), + b.equal(2, BinaryString.fromString("x")))) + .get() + .get() + .results() + .toRangeList()) + .containsExactly(new Range(3, 3)); + } finally { + executor.shutdownNow(); + } + } + + @Test + void testFileOverlapAndInExpansionBoundaries() { + RowType type = + new RowType( + Arrays.asList( + new DataField(0, "a", DataTypes.INT().notNull()), + new DataField(1, "b", DataTypes.INT().notNull()))); + CompositeKeySerializer serializer = new CompositeKeySerializer(type); + PredicateBuilder b = new PredicateBuilder(type); + Predicate range = + PredicateBuilder.and(b.equal(0, 1), b.greaterThan(1, 7), b.lessOrEqual(1, 9)); + GlobalIndexIOMeta below = + metadata("below", serializer, GenericRow.of(1, 0), GenericRow.of(1, 7)); + GlobalIndexIOMeta inside = + metadata("inside", serializer, GenericRow.of(1, 8), GenericRow.of(1, 9)); + GlobalIndexIOMeta above = + metadata("above", serializer, GenericRow.of(1, 10), GenericRow.of(2, 0)); + GlobalIndexIOMeta before = + metadata( + "before", + serializer, + GenericRow.of(0, 0), + GenericRow.of(0, Integer.MAX_VALUE)); + GlobalIndexIOMeta after = + metadata( + "after", + serializer, + GenericRow.of(2, Integer.MIN_VALUE), + GenericRow.of(2, 0)); + GlobalIndexIOMeta noEndpoints = metadata("no-endpoints", serializer, null, null); + GlobalIndexIOMeta noLast = metadata("no-last", serializer, GenericRow.of(1, 8), null); + GlobalIndexIOMeta noMetadata = new GlobalIndexIOMeta(new Path("unknown"), 100, 1, null); + List files = + Arrays.asList(below, inside, above, before, after, noEndpoints, noLast, noMetadata); + CompositeBTreePredicate.Plan plan = + CompositeBTreePredicate.plan(type.getFields(), range).get(); + List selected = plan.selectFiles(files); + assertThat(selected).containsExactly(inside, noEndpoints, noLast, noMetadata); + assertThat(plan.canScan(selected, 400)).isTrue(); + assertThat(plan.canScan(selected, 399)).isFalse(); + assertThat(plan.canScan(selected, 0)).isFalse(); + assertThat(plan.canScan(Collections.emptyList(), 0)).isTrue(); + assertThat( + CompositeBTreePredicate.plan( + type.getFields(), + PredicateBuilder.and(range, b.lessOrEqual(1, 7))) + .get() + .selectFiles(files)) + .isEmpty(); + assertThat( + CompositeBTreePredicate.plan(type.getFields(), b.isNull(0)) + .get() + .selectFiles(files)) + .isEmpty(); + List sixteen = + IntStream.range(0, 16).boxed().map(i -> (Object) i).collect(Collectors.toList()); + List seventeen = + IntStream.range(0, 17).boxed().map(i -> (Object) i).collect(Collectors.toList()); + CompositeBTreePredicate.Plan points = + CompositeBTreePredicate.plan( + type.getFields(), + PredicateBuilder.and(b.in(0, sixteen), b.in(1, sixteen))) + .get(); + assertThat(points.intervals()).hasSize(256); + assertThat(points.isPointLookup()).isTrue(); + assertThat(points.canScan(files, 0)).isTrue(); + assertThat( + CompositeBTreePredicate.plan( + type.getFields(), + PredicateBuilder.and(b.in(0, sixteen), b.in(1, seventeen)))) + .isEmpty(); + } + + @Test + void testKeyFiltersSkipRejectedPostingDecodeAndHonorExactBudget() throws Exception { + RowType type = + new RowType( + Arrays.asList( + new DataField(0, "category", DataTypes.STRING().notNull()), + new DataField(1, "item_number", DataTypes.INT().notNull()), + new DataField(2, "tag", DataTypes.STRING().notNull()))); + CompositeKeySerializer serializer = new CompositeKeySerializer(type); + GenericRow accepted = row("category-a", 8, "x"); + GenericRow rejected = row("category-a", 8, "y"); + LocalFileIO io = LocalFileIO.create(); + Path path = new Path(tempPath.resolve("filtered-posting").toUri()); + try (PositionOutputStream out = io.newOutputStream(path, false)) { + SstFileWriter writer = new SstFileWriter(out, 128, null, null); + LongArrayList posting = new LongArrayList(1); + posting.add(0); + writer.put(serializer.serialize(accepted), BTreePostingList.serialize(posting)); + // Invalid payload: scanning this rejected key must never decode its posting. + writer.put(serializer.serialize(rejected), new byte[] {99}); + writer.flush(); + writer.writeSlice( + BTreeFileFooter.writeFooter( + new BTreeFileFooter( + BTreeFileFooter.VERSION_2, + null, + writer.writeIndexBlock(), + null))); + } + GlobalIndexIOMeta meta = + new GlobalIndexIOMeta( + path, + io.getFileSize(path), + 2, + new SortedIndexFileMeta( + serializer.serialize(accepted), + serializer.serialize(rejected), + false) + .serialize()); + PredicateBuilder b = new PredicateBuilder(type); + Predicate range = + PredicateBuilder.and( + b.equal(0, BinaryString.fromString("category-a")), b.greaterThan(1, 7)); + Predicate filtered = PredicateBuilder.and(range, b.equal(2, BinaryString.fromString("x"))); + ExecutorService executor = newDirectExecutorService(); + try { + for (long budget : Arrays.asList(meta.fileSize(), meta.fileSize() - 1)) { + Options options = new Options(); + options.set( + BTreeIndexOptions.BTREE_INDEX_FALLBACK_SCAN_MAX_SIZE, + MemorySize.ofBytes(budget)); + GlobalIndexer indexer = + GlobalIndexer.create( + "btree", + type.getFields().get(0), + type.getFields().subList(1, 3), + options); + try (GlobalIndexReader reader = + indexer.createReader( + file -> io.newInputStream(file.filePath()), + Collections.singletonList(meta), + 2, + null, + executor)) { + if (budget == meta.fileSize()) { + assertThat( + reader.visitComposite(filtered) + .get() + .get() + .results() + .toRangeList()) + .containsExactly(new Range(0, 0)); + assertThatThrownBy(() -> reader.visitComposite(range).get()) + .hasStackTraceContaining("Unknown BTree posting list type: 99"); + } else { + assertThat(reader.visitComposite(filtered).get()).isEmpty(); + } + } + } + } finally { + executor.shutdownNow(); + } + } + + private GlobalIndexIOMeta metadata( + String name, CompositeKeySerializer serializer, GenericRow first, GenericRow last) { + return new GlobalIndexIOMeta( + new Path(name), + 100, + 1, + new SortedIndexFileMeta( + first == null ? null : serializer.serialize(first), + last == null ? null : serializer.serialize(last), + false) + .serialize()); + } + @Test void testCompositeKeysPreserveTypesBoundariesAndNulls() { RowType type = diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java index 00b3260e5f0d..8b84683a5d7f 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionBatchScan.java @@ -388,10 +388,14 @@ private Plan planWithIndexQuery() { table.rowType(), indexFilter, indexFiles, - table.store().pathFactory().globalIndexFileFactory()); - // Scalar reader support can depend on global file coverage and scan budgets. Resolve it - // before split pruning so unsupported residuals cannot discard composite matches. - if (indexQuery == null || (indexQuery.hasCompositeQuery() && indexQuery.hasScalarQuery())) { + table.store().pathFactory().globalIndexFileFactory(), + table.coreOptions().toConfiguration()); + // Scalar and composite range support can depend on coverage and scan budgets. Resolve it + // before split pruning, including scalar alternatives after a composite scan is rejected. + if (indexQuery == null + || (indexQuery.requiresPlanningEvaluation() + && indexFiles.stream() + .anyMatch(DataEvolutionGlobalIndexScanner::isCompositeBTree))) { return planEagerIndex(dataPlan, snapshot, partitionFilter, indexFiles, indexFilter); } List unindexed = diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java index 740e3ac14d54..74a4413bb7ab 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/DataEvolutionGlobalIndexScanner.java @@ -380,7 +380,7 @@ static Filter indexFileFilter( return indexFileFilter; } - private static boolean isCompositeBTree(IndexFileMeta file) { + static boolean isCompositeBTree(IndexFileMeta file) { GlobalIndexMeta meta = file.globalIndexMeta(); return "btree".equals(file.indexType()) && meta != null @@ -411,7 +411,8 @@ public Optional scanWithCoverage(Predicate pred .noneMatch( DataEvolutionGlobalIndexScanner::isCompositeBTree) ? null - : GlobalIndexQuery.create(rowType, predicate, indexFiles, indexPathFactory); + : GlobalIndexQuery.create( + rowType, predicate, indexFiles, indexPathFactory, options); if (query != null && query.hasCompositeQuery()) { try { return query.evaluateWithCoverage( @@ -446,9 +447,13 @@ public GlobalIndexResult unindexedRows(Predicate predicate) { GlobalIndexQuery query = predicate == null ? null - : GlobalIndexQuery.create(rowType, predicate, indexFiles, indexPathFactory); - if (query != null && query.hasCompositeQuery() && query.hasScalarQuery()) { - // Scalar shards can decline a predicate at runtime, so the legacy API must use + : GlobalIndexQuery.create( + rowType, predicate, indexFiles, indexPathFactory, options); + if (query != null + && query.requiresPlanningEvaluation() + && indexFiles.stream() + .anyMatch(DataEvolutionGlobalIndexScanner::isCompositeBTree)) { + // Scalar predicates and composite scans can decline at runtime, so use // evaluated coverage as well. Internal callers retain the evaluation directly. Optional evaluation = scanWithCoverage(predicate); return evaluation.isPresent() diff --git a/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java b/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java index b530824ec70a..8d28f65f68c6 100644 --- a/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java +++ b/paimon-core/src/main/java/org/apache/paimon/globalindex/GlobalIndexQuery.java @@ -22,6 +22,7 @@ import org.apache.paimon.fs.Path; import org.apache.paimon.globalindex.DataEvolutionGlobalIndexScanner.IndexMetaFileGroup; import org.apache.paimon.globalindex.GlobalIndexEvaluator.Evaluation; +import org.apache.paimon.globalindex.btree.BTreeIndexOptions; import org.apache.paimon.globalindex.btree.CompositeBTreePredicate; import org.apache.paimon.index.IndexFileMeta; import org.apache.paimon.index.IndexPathFactory; @@ -30,7 +31,6 @@ import org.apache.paimon.options.Options; import org.apache.paimon.predicate.And; import org.apache.paimon.predicate.CompoundPredicate; -import org.apache.paimon.predicate.Equal; import org.apache.paimon.predicate.FieldRef; import org.apache.paimon.predicate.GreaterOrEqual; import org.apache.paimon.predicate.LeafPredicate; @@ -107,6 +107,16 @@ static GlobalIndexQuery create( Predicate predicate, List files, IndexPathFactory pathFactory) { + return create(rowType, predicate, files, pathFactory, new Options()); + } + + @Nullable + static GlobalIndexQuery create( + RowType rowType, + Predicate predicate, + List files, + IndexPathFactory pathFactory, + Options options) { Map> groupsByField = DataEvolutionGlobalIndexScanner.groupIndexFiles( files.stream() @@ -124,88 +134,39 @@ static GlobalIndexQuery create( } groups.put(fieldId, indexGroups); }); - return createForPredicate(predicate, rowType, groups); + return createForPredicate(predicate, rowType, groups, options); } @Nullable private static GlobalIndexQuery createForPredicate( - Predicate predicate, RowType rowType, Map> groups) { + Predicate predicate, + RowType rowType, + Map> groups, + Options options) { if (predicate instanceof LeafPredicate) { LeafPredicate leaf = (LeafPredicate) predicate; Optional field = leaf.fieldRefOptional(); if (!field.isPresent()) { return null; } - return createForIndexedField(leaf, field.get(), rowType, groups); + CompositeCandidate composite = selectComposite(leaf, groups, options); + return composite != null + ? composite.query() + : createForIndexedField(leaf, field.get(), rowType, groups); } CompoundPredicate compound = (CompoundPredicate) predicate; boolean union = compound.function() instanceof Or; List children = new ArrayList<>(); List predicates = GlobalIndexEvaluator.normalizedChildren(compound); if (!union) { - // Prefer the longest full composite key before any single-column posting is read. while (!predicates.isEmpty()) { - IndexGroup selected = null; - List matched = null; - Predicate conjunction = PredicateBuilder.and(predicates); - for (List fieldGroups : groups.values()) { - for (IndexGroup group : fieldGroups) { - if (!group.isCompositeBTree() - || (selected != null - && group.extraFields.size() - <= selected.extraFields.size())) { - continue; - } - Optional> match = - CompositeBTreePredicate.match(group.fields(), conjunction); - if (match.isPresent()) { - selected = group; - matched = match.get(); - } - } - } + CompositeCandidate selected = + selectComposite(PredicateBuilder.and(predicates), groups, options); if (selected == null) { break; } - List covered = new ArrayList<>(); - for (Predicate child : predicates) { - if (child instanceof LeafPredicate) { - LeafPredicate leaf = (LeafPredicate) child; - if (leaf.function() instanceof Equal - && matched.stream() - .anyMatch( - equality -> - equality.fieldRefOptional() - .equals(leaf.fieldRefOptional()))) { - covered.add(child); - } - } - } - Predicate composite = PredicateBuilder.and(covered); - List selectedGroups = new ArrayList<>(); - for (IndexGroup group : groups.get(selected.field.id())) { - if (group.type.equals(selected.type) - && group.fields().equals(selected.fields())) { - selectedGroups.add(group.selectFiles(composite)); - } - } - children.add( - new GlobalIndexQuery( - composite, false, Collections.emptyList(), selectedGroups)); - predicates.removeAll(covered); - // The tuple equality already narrows these columns to one value. Leave their - // remaining conditions as data filters instead of reading scalar postings. - List keyEqualities = matched; - predicates.removeIf( - child -> { - if (!(child instanceof LeafPredicate)) { - return false; - } - Optional field = ((LeafPredicate) child).fieldRefOptional(); - return keyEqualities.stream() - .anyMatch( - equality -> equality.fieldRefOptional().equals(field)); - }); + children.add(selected.query()); + predicates.removeAll(selected.plan.predicates()); } } for (int i = 0; i < predicates.size(); i++) { @@ -238,7 +199,7 @@ private static GlobalIndexQuery createForPredicate( } } if (query == null) { - query = createForPredicate(child, rowType, groups); + query = createForPredicate(child, rowType, groups, options); } if (query == null) { if (union) { @@ -253,6 +214,146 @@ private static GlobalIndexQuery createForPredicate( : new GlobalIndexQuery(null, union, children, Collections.emptyList()); } + @Nullable + private static CompositeCandidate selectComposite( + Predicate predicate, Map> groups, Options options) { + CompositeCandidate selected = null; + Set> seen = new HashSet<>(); + for (List fieldGroups : groups.values()) { + for (IndexGroup group : fieldGroups) { + if (!group.isCompositeBTree() || !seen.add(group.fields())) { + continue; + } + Optional planned = + CompositeBTreePredicate.plan(group.fields(), predicate); + if (!planned.isPresent()) { + continue; + } + List definition = + groups.get(group.field.id()).stream() + .filter( + other -> + other.type.equals(group.type) + && other.fields().equals(group.fields())) + .collect(Collectors.toList()); + CompositeCandidate candidate = + new CompositeCandidate(group, planned.get(), definition); + long scanBudget = + options.get(BTreeIndexOptions.BTREE_INDEX_FALLBACK_SCAN_MAX_SIZE) + .getBytes(); + if (candidate.selectedGroups.stream() + .anyMatch(part -> !candidate.plan.canScan(part.files, scanBudget))) { + continue; + } + // A scalar point/range has no tuple expansion when only its leading column is + // filtered. + if (candidate.plan.filterColumns() == 1 + && groups.get(group.field.id()).stream() + .filter( + other -> + "btree".equals(other.type) + && other.extraFields.isEmpty()) + .map(other -> other.range) + .collect( + Collectors.collectingAndThen( + Collectors.toList(), + scalarCoverage -> + Range.and( + candidate.coverage, + Range.sortAndMergeOverlap( + scalarCoverage, + true)) + .equals(candidate.coverage)))) { + continue; + } + if (selected == null || candidate.compareTo(selected) < 0) { + selected = candidate; + } + } + } + return selected; + } + + /** + * Metadata-only preference: useful prefix, key filtering, coverage, probes and selected bytes. + */ + private static class CompositeCandidate implements Comparable { + private final IndexGroup group; + private final CompositeBTreePredicate.Plan plan; + private final List selectedGroups; + private final List coverage; + private final long coveredRows; + private final long bytes; + + private CompositeCandidate( + IndexGroup group, CompositeBTreePredicate.Plan plan, List definition) { + this.group = group; + this.plan = plan; + Predicate predicate = PredicateBuilder.and(plan.predicates()); + this.selectedGroups = + definition.stream() + .map(part -> part.selectFiles(predicate)) + .collect(Collectors.toList()); + this.coverage = + Range.sortAndMergeOverlap( + definition.stream() + .map(part -> part.range) + .collect(Collectors.toList()), + true); + long rows = 0; + for (Range range : coverage) { + rows = saturatedAdd(rows, range.count()); + } + this.coveredRows = rows; + long size = 0; + for (IndexGroup part : selectedGroups) { + for (GlobalIndexIOMeta file : part.files) { + size = saturatedAdd(size, file.fileSize()); + } + } + this.bytes = size; + } + + private GlobalIndexQuery query() { + return new GlobalIndexQuery( + PredicateBuilder.and(plan.predicates()), + false, + Collections.emptyList(), + selectedGroups); + } + + @Override + public int compareTo(CompositeCandidate other) { + int comparison = Integer.compare(other.plan.boundColumns(), plan.boundColumns()); + if (comparison == 0) { + comparison = Integer.compare(other.plan.filterColumns(), plan.filterColumns()); + } + if (comparison == 0) { + comparison = Long.compare(other.coveredRows, coveredRows); + } + if (comparison == 0) { + comparison = Boolean.compare(other.plan.isPointLookup(), plan.isPointLookup()); + } + if (comparison == 0) { + comparison = + Integer.compare(plan.intervals().size(), other.plan.intervals().size()); + } + if (comparison == 0) { + comparison = Long.compare(bytes, other.bytes); + } + if (comparison == 0) { + comparison = Integer.compare(group.fields().size(), other.group.fields().size()); + } + return comparison != 0 + ? comparison + : group.fields().toString().compareTo(other.group.fields().toString()); + } + + private static long saturatedAdd(long left, long right) { + return right > Long.MAX_VALUE - left ? Long.MAX_VALUE : left + right; + } + } + @Nullable private static GlobalIndexQuery createForIndexedField( Predicate predicate, @@ -290,9 +391,17 @@ boolean hasCompositeQuery() { || children.stream().anyMatch(GlobalIndexQuery::hasCompositeQuery); } - boolean hasScalarQuery() { - return groups.stream().anyMatch(group -> !group.isCompositeBTree()) - || children.stream().anyMatch(GlobalIndexQuery::hasScalarQuery); + /** Range/prefix budgets must be resolved before computing deferred-scan coverage. */ + boolean requiresPlanningEvaluation() { + return groups.stream() + .anyMatch( + group -> + !group.isCompositeBTree() + || !CompositeBTreePredicate.plan( + group.fields(), predicate) + .get() + .isPointLookup()) + || children.stream().anyMatch(GlobalIndexQuery::requiresPlanningEvaluation); } /** Coverage of the selected query paths, including files pruned safely by key metadata. */ @@ -450,22 +559,7 @@ private Optional evaluateWithExecutor( private Function>> predicateQuery(IndexGroup group) { if (group.isCompositeBTree()) { - if (CompositeBTreePredicate.isContradictory(group.fields(), predicate)) { - return reader -> - CompletableFuture.completedFuture( - Optional.of(GlobalIndexResult.createEmpty())); - } - List equalities = - CompositeBTreePredicate.match(group.fields(), predicate) - .orElseThrow( - () -> - new IllegalArgumentException( - "Incomplete composite BTree predicate")); - List literals = - equalities.stream() - .map(leaf -> leaf.literals().get(0)) - .collect(Collectors.toList()); - return reader -> reader.visitCompositeEqual(literals); + return reader -> reader.visitComposite(predicate); } if (predicate instanceof LeafPredicate) { LeafPredicate leaf = (LeafPredicate) predicate; diff --git a/paimon-core/src/test/java/org/apache/paimon/globalindex/GlobalIndexQueryTest.java b/paimon-core/src/test/java/org/apache/paimon/globalindex/GlobalIndexQueryTest.java index 90e899c7d9e2..846c46b4504b 100644 --- a/paimon-core/src/test/java/org/apache/paimon/globalindex/GlobalIndexQueryTest.java +++ b/paimon-core/src/test/java/org/apache/paimon/globalindex/GlobalIndexQueryTest.java @@ -87,6 +87,72 @@ class GlobalIndexQueryTest { @TempDir java.nio.file.Path tempDir; + @Test + void testCompositeChoiceUsesUsefulPrefixCoverageAndBytes() throws Exception { + RowType type = RowType.of(DataTypes.INT(), DataTypes.INT(), DataTypes.INT()); + PredicateBuilder builder = new PredicateBuilder(type); + Predicate joint = PredicateBuilder.and(builder.equal(0, 7), builder.equal(1, 8)); + IndexPathFactory paths = mock(IndexPathFactory.class); + when(paths.toPath(any(IndexFileMeta.class))) + .thenAnswer( + invocation -> + new Path( + "index/" + + invocation + .getArgument(0) + .fileName())); + IndexFileMeta ab = compositeFile("ab", 1000, 99, 0, new int[] {1}); + IndexFileMeta abc = compositeFile("abc", 1, 99, 0, new int[] {1, 2}); + IndexFileMeta ba = compositeFile("ba", 100, 99, 1, new int[] {0}); + // AB is a point lookup, whereas ABC has an unconstrained trailing column. + assertQueryFiles(type, joint, paths, Arrays.asList(abc, ab), "ab", "abc"); + // Equivalent point lookups prefer fewer selected bytes, independently of metadata order. + assertQueryFiles(type, joint, paths, Arrays.asList(ab, ba), "ba", "ab"); + assertQueryFiles(type, joint, paths, Arrays.asList(ba, ab), "ba", "ab"); + IndexFileMeta covered = compositeFile("covered", 10000, 199, 1, new int[] {0}); + GlobalIndexQuery query = + GlobalIndexQuery.create(type, joint, Arrays.asList(ab, covered), paths); + assertThat(query.coveredRanges()).containsExactly(new Range(0, 199)); + assertQueryFiles(type, joint, paths, Arrays.asList(ab, covered), "covered", "ab"); + IndexFileMeta scalar = compositeFile("scalar", 10, 99, 0, null); + GlobalIndexQuery scalarQuery = + GlobalIndexQuery.create( + type, builder.equal(0, 7), Arrays.asList(ab, scalar), paths); + assertThat(scalarQuery.hasCompositeQuery()).isFalse(); + assertQueryFiles( + type, builder.equal(0, 7), paths, Arrays.asList(ab, scalar), "scalar", "ab"); + } + + private IndexFileMeta compositeFile( + String name, long bytes, long lastRow, int field, int[] extras) { + return new IndexFileMeta( + "btree", + name, + bytes, + lastRow + 1, + new GlobalIndexMeta(0, lastRow, field, extras, null), + null); + } + + private void assertQueryFiles( + RowType type, + Predicate predicate, + IndexPathFactory paths, + List files, + String selected, + String skipped) + throws Exception { + GlobalIndexQuery query = GlobalIndexQuery.create(type, predicate, files, paths); + DataOutputSerializer output = new DataOutputSerializer(256); + query.serialize(output); + String serialized = new String(output.getCopyOfBuffer(), StandardCharsets.ISO_8859_1); + assertThat(serialized).contains("index/" + selected).doesNotContain("index/" + skipped); + GlobalIndexQuery restored = + GlobalIndexQuery.deserialize( + new org.apache.paimon.io.DataInputDeserializer(output.getCopyOfBuffer())); + assertThat(restored).isEqualTo(query); + } + @ParameterizedTest @ValueSource(strings = {"btree", "bitmap"}) void testPrunesSortedFilesBeforeSplitSerialization(String indexType) throws Exception { diff --git a/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java b/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java index 659260099bcc..606423343834 100644 --- a/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java +++ b/paimon-core/src/test/java/org/apache/paimon/table/CompositeBTreeTableTest.java @@ -65,6 +65,175 @@ /** Composite lookups must avoid single-column postings and retain uncovered data. */ class CompositeBTreeTableTest extends DataEvolutionTestBase { + @Test + void testPrefixAndRangeLookupsAvoidSingleColumnPostings() throws Exception { + createTableDefault(); + FileStoreTable table = + table().copy(Collections.singletonMap("sorted-index.records-per-file", "13")); + append(table, 0, 100); + build(table, "f1"); + build(table, "f0"); + build(table, "f1", "f0"); + deleteSingleColumnFiles(table); + PredicateBuilder builder = new PredicateBuilder(table.rowType()); + Predicate prefix = builder.equal(1, BinaryString.fromString("category-a")); + Predicate range = PredicateBuilder.and(builder.greaterThan(0, 7), prefix); + try (DataEvolutionGlobalIndexScanner scanner = + DataEvolutionGlobalIndexScanner.create( + table, + table.store().newIndexFileHandler().scanEntries().stream() + .map(IndexManifestEntry::indexFile) + .collect(Collectors.toList())) + .get()) { + assertThat(scanner.scan(range).get().results().toRangeList()) + .containsExactly( + new Range(8, 9), + new Range(28, 29), + new Range(48, 49), + new Range(68, 69), + new Range(88, 89)); + } + for (boolean inReader : Arrays.asList(false, true)) { + assertThat(read(configured(table, "fast", inReader), range)) + .containsExactlyInAnyOrder( + "p8", "p9", "p28", "p29", "p48", "p49", "p68", "p69", "p88", "p89"); + } + } + + @Test + void testCompositeBudgetRetainsScalarPointCandidates() throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 20); + build(table, "f1"); + build(table, "f1", "f0"); + PredicateBuilder builder = new PredicateBuilder(table.rowType()); + Predicate query = + PredicateBuilder.and( + builder.equal(1, BinaryString.fromString("category-a")), + builder.greaterThan(0, 7)); + FileStoreTable budgeted = + table.copy(Collections.singletonMap("btree-index.fallback-scan-max-size", "0 b")); + try (DataEvolutionGlobalIndexScanner scanner = + DataEvolutionGlobalIndexScanner.create( + budgeted, + table.store().newIndexFileHandler().scanEntries().stream() + .map(IndexManifestEntry::indexFile) + .collect(Collectors.toList())) + .get()) { + // The unsupported composite scan must retain scalar equality pruning. + assertThat(scanner.scan(query)).isPresent(); + assertThat(scanner.scan(query).get().results().toRangeList()) + .containsExactly(new Range(0, 9)); + } + for (boolean inReader : Arrays.asList(false, true)) { + assertThat(read(configured(budgeted, "fast", inReader), query)) + .containsExactlyInAnyOrder("p8", "p9"); + } + } + + @Test + void testRangeCoverageBudgetsAndInFallback() throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 20); + build(table, "f1", "f0"); + append(table, 20, 40); + build(table, "f1"); + build(table, "f0"); + PredicateBuilder builder = new PredicateBuilder(table.rowType()); + Predicate prefix = builder.equal(1, BinaryString.fromString("category-a")); + Predicate range = PredicateBuilder.and(prefix, builder.greaterThan(0, 7)); + Predicate union = + PredicateBuilder.or( + range, + PredicateBuilder.and( + builder.equal(1, BinaryString.fromString("category-b")), + builder.lessThan(0, 1))); + FileStoreTable withoutIndex = + table.copy( + Collections.singletonMap(CoreOptions.GLOBAL_INDEX_ENABLED.key(), "false")); + Predicate largeIn = + new org.apache.paimon.predicate.LeafPredicate( + org.apache.paimon.predicate.In.INSTANCE, + table.rowType().getTypeAt(0), + 0, + "f0", + java.util.stream.IntStream.range(0, 257) + .boxed() + .map(i -> (Object) i) + .collect(Collectors.toList())); + for (boolean inReader : Arrays.asList(false, true)) { + assertThat(read(configured(table, "fast", inReader), range)) + .containsExactlyInAnyOrder("p8", "p9"); + for (String mode : Arrays.asList("full", "detail")) { + FileStoreTable configured = configured(table, mode, inReader); + assertThat(read(configured, range)) + .containsExactlyInAnyOrder("p8", "p9", "p28", "p29"); + assertThat(read(configured, union)) + .containsExactlyInAnyOrderElementsOf(read(withoutIndex, union)); + FileStoreTable budgeted = + configured.copy( + Collections.singletonMap( + "btree-index.fallback-scan-max-size", "0 b")); + assertThat(read(budgeted, range)) + .containsExactlyInAnyOrderElementsOf(read(withoutIndex, range)); + Predicate expanded = PredicateBuilder.and(prefix, largeIn); + assertThat(read(configured, expanded)) + .containsExactlyInAnyOrderElementsOf(read(withoutIndex, expanded)); + } + } + } + + @Test + void testPrefixAndTrailingKeyFiltersUseCompositeIndex() throws Exception { + createTableDefault(); + FileStoreTable table = table(); + append(table, 0, 60); + build(table, "f1", "f0", "f2"); + PredicateBuilder builder = new PredicateBuilder(table.rowType()); + Predicate prefix = builder.equal(1, BinaryString.fromString("category-a")); + Predicate suffix = builder.contains(2, BinaryString.fromString("8")); + List queries = + Arrays.asList( + prefix, + PredicateBuilder.and(prefix, suffix), + PredicateBuilder.and(prefix, builder.greaterThan(0, 5), suffix), + PredicateBuilder.and(prefix, builder.in(0, Arrays.asList(7, 8))), + PredicateBuilder.and( + builder.in( + 1, + Arrays.asList( + BinaryString.fromString("category-a"), + BinaryString.fromString("category-b"))), + builder.between(0, 7, 8))); + FileStoreTable withoutIndex = + table.copy( + Collections.singletonMap(CoreOptions.GLOBAL_INDEX_ENABLED.key(), "false")); + for (boolean inReader : Arrays.asList(false, true)) { + for (Predicate query : queries) { + assertThat(read(configured(table, "fast", inReader), query)) + .containsExactlyInAnyOrderElementsOf(read(withoutIndex, query)); + } + } + try (DataEvolutionGlobalIndexScanner scanner = + DataEvolutionGlobalIndexScanner.create( + table, + table.store().newIndexFileHandler().scanEntries().stream() + .map(IndexManifestEntry::indexFile) + .collect(Collectors.toList())) + .get()) { + assertThat( + scanner.scan( + PredicateBuilder.and( + prefix, builder.greaterThan(0, 5), suffix)) + .get() + .results() + .toRangeList()) + .containsExactly(new Range(8, 8), new Range(28, 28), new Range(48, 48)); + } + } + @Test void testCoexistsAndAvoidsSingleColumnPostings() throws Exception { createTableDefault(); @@ -373,7 +542,7 @@ void testPartialCompositeCoverageDoesNotBorrowSingleColumnCoverage() throws Exce } @Test - void testPartialSingleCoverageDoesNotBorrowCompositeCoverage() throws Exception { + void testCompositePrefixExtendsPartialSingleColumnCoverage() throws Exception { createTableDefault(); FileStoreTable table = table(); append(table, 0, 20); @@ -384,7 +553,9 @@ void testPartialSingleCoverageDoesNotBorrowCompositeCoverage() throws Exception new PredicateBuilder(table.rowType()) .equal(1, BinaryString.fromString("category-a")); for (boolean inReader : Arrays.asList(false, true)) { - assertThat(read(configured(table, "fast", inReader), scalar)).hasSize(10); + assertThat(read(configured(table, "fast", inReader), scalar)) + .hasSize(20) + .contains("p27"); for (String mode : Arrays.asList("full", "detail")) { assertThat(read(configured(table, mode, inReader), scalar)) .hasSize(20) diff --git a/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java b/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java index 9eb127cddb22..d5da664ffadb 100644 --- a/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java +++ b/paimon-flink/paimon-flink-common/src/test/java/org/apache/paimon/flink/SortedGlobalIndexITCase.java @@ -36,6 +36,7 @@ import org.apache.flink.types.Row; import org.junit.jupiter.api.Test; +import java.util.Arrays; import java.util.List; import java.util.Map; import java.util.Set; @@ -98,7 +99,10 @@ public void testCompositeBTreeIndex() throws Exception { (i / 10) % 2 == 0 ? "category-a" : "category-b", i % 10)) .collect(Collectors.joining(",")); - sql("INSERT INTO T_COMPOSITE VALUES " + values); + sql( + "INSERT INTO T_COMPOSITE VALUES " + + values + + ", (100, CAST(NULL AS STRING), 7), (101, 'category-a', CAST(NULL AS INT))"); sql( "CALL sys.create_global_index(`table` => 'default.T_COMPOSITE', index_column => 'category,item_number', index_type => 'btree')"); assertThat( @@ -113,11 +117,43 @@ public void testCompositeBTreeIndex() throws Exception { assertThat(entry.indexFile().globalIndexMeta().rowRange().count()) .isLessThanOrEqualTo(13); }); - sql("ALTER TABLE T_COMPOSITE SET ('global-index.query-in-reader.enabled' = 'true')"); - assertThat( - sql( - "SELECT id FROM T_COMPOSITE WHERE category = 'category-a' AND item_number = 7")) - .containsExactlyInAnyOrder(Row.of(7), Row.of(27)); + for (boolean inReader : Arrays.asList(false, true)) { + sql( + "ALTER TABLE T_COMPOSITE SET ('global-index.query-in-reader.enabled' = '" + + inReader + + "', 'scalar-index.search-mode' = 'fast')"); + assertThat( + sql( + "SELECT id FROM T_COMPOSITE WHERE category = 'category-a' AND item_number = 7")) + .containsExactlyInAnyOrder(Row.of(7), Row.of(27)); + assertThat( + sql( + "SELECT id FROM T_COMPOSITE WHERE item_number > 7 AND category = 'category-a'")) + .containsExactlyInAnyOrder(Row.of(8), Row.of(9), Row.of(28), Row.of(29)); + assertThat( + sql( + "SELECT id FROM T_COMPOSITE WHERE category = 'category-a' AND item_number IN (7, 8)")) + .containsExactlyInAnyOrder(Row.of(7), Row.of(8), Row.of(27), Row.of(28)); + assertThat( + sql( + "SELECT id FROM T_COMPOSITE WHERE category IS NOT NULL AND item_number BETWEEN 7 AND 8")) + .containsExactlyInAnyOrder( + Row.of(7), + Row.of(8), + Row.of(17), + Row.of(18), + Row.of(27), + Row.of(28), + Row.of(37), + Row.of(38)); + assertThat(sql("SELECT id FROM T_COMPOSITE WHERE category = 'category-a'")).hasSize(21); + assertThat(sql("SELECT id FROM T_COMPOSITE WHERE category IS NULL AND item_number = 7")) + .containsExactly(Row.of(100)); + assertThat( + sql( + "SELECT id FROM T_COMPOSITE WHERE category = 'category-a' AND item_number IS NULL")) + .containsExactly(Row.of(101)); + } } @Test diff --git a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala index 33d40458dad3..9549eec42603 100644 --- a/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala +++ b/paimon-spark/paimon-spark-ut/src/test/scala/org/apache/paimon/spark/procedure/CompositeBTreeIndexProcedureTest.scala @@ -78,6 +78,7 @@ class CompositeBTreeIndexProcedureTest extends PaimonSparkTestBase { s"CALL sys.create_global_index(table => 'test.T', index_column => '$columns', index_type => 'btree')") } insert(0, 40) + sql("INSERT INTO T VALUES (100, NULL, 7), (101, 'category-a', NULL)") build("category") build("item_number") build("category,item_number") @@ -85,10 +86,29 @@ class CompositeBTreeIndexProcedureTest extends PaimonSparkTestBase { assert(indexes.exists(_.indexFile().globalIndexMeta().getIndexedFieldIds().size() == 2)) for (inReader <- Seq(false, true)) { sql( - s"ALTER TABLE T SET TBLPROPERTIES ('global-index.query-in-reader.enabled' = '$inReader')") + s"ALTER TABLE T SET TBLPROPERTIES ('global-index.query-in-reader.enabled' = '$inReader', 'scalar-index.search-mode' = 'fast')") checkAnswer( sql("SELECT id FROM T WHERE item_number = 7 AND category = 'category-a'"), Seq(Row(7), Row(27))) + checkAnswer( + sql("SELECT id FROM T WHERE item_number > 7 AND category = 'category-a'"), + Seq(Row(8), Row(9), Row(28), Row(29))) + checkAnswer( + sql("SELECT id FROM T WHERE category = 'category-a' AND item_number IN (7, 8)"), + Seq(Row(7), Row(8), Row(27), Row(28))) + checkAnswer( + sql("SELECT id FROM T WHERE category IS NOT NULL AND item_number BETWEEN 7 AND 8"), + Seq(Row(7), Row(8), Row(17), Row(18), Row(27), Row(28), Row(37), Row(38)) + ) + checkAnswer( + sql("SELECT id FROM T WHERE category = 'category-a'"), + (0 until 40).filter(i => (i / 10) % 2 == 0).map(Row(_)) :+ Row(101)) + checkAnswer( + sql("SELECT id FROM T WHERE category IS NULL AND item_number = 7"), + Seq(Row(100))) + checkAnswer( + sql("SELECT id FROM T WHERE category = 'category-a' AND item_number IS NULL"), + Seq(Row(101))) } insert(40, 60) build("category,item_number")