From f0113e1f1da0522969ef6b353d82cbb5fc5b289f Mon Sep 17 00:00:00 2001 From: yangjie01 Date: Fri, 25 Sep 2026 10:19:46 +0800 Subject: [PATCH] [core] Scan raw data for full-text search when no index exists MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit DataEvolutionFullTextScan only added the raw-data compensation split when at least one full-text index file existed. On a table whose full-text index has never been built (or fully expired), the FULL and DETAIL search modes silently returned an empty plan — zero rows — although they promise to scan the raw data for unindexed rows; only FAST, which is index-only by design, may return nothing. Always compute the unindexed ranges — with zero index files every row is unindexed, so FULL mode plans a raw split covering the whole row-id space and DETAIL mode covers the data files' ranges; FAST still yields no split. The raw read side can then no longer take the index implementation from an index split: resolve it from the column's splits, then any split, and finally the fixed 'full-text' implementation that a never-indexed table implicitly uses. Assisted-by: GLM-5.3 --- .../source/DataEvolutionFullTextScan.java | 25 +++--- .../table/source/RawFullTextReadImpl.java | 23 +++-- .../source/FullTextSearchBuilderTest.java | 87 +++++++++++++++++++ 3 files changed, 118 insertions(+), 17 deletions(-) diff --git a/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextScan.java b/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextScan.java index a3636b3e424c..4e6aeb2279ea 100644 --- a/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextScan.java +++ b/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextScan.java @@ -166,18 +166,19 @@ public Plan scan() { Collections.singletonList(selection.fileRange)))); } - if (!fullTextIndexFiles.isEmpty()) { - List rawRowRanges = - new DataEvolutionGlobalIndexCoverage( - table, - snapshot, - partitionFilter, - fullTextIndexFiles, - table.coreOptions().fullTextIndexSearchMode()) - .unindexedRanges(textColumnIds, null); - if (!rawRowRanges.isEmpty()) { - splits.add(new RawFullTextSearchSplit(rawRowRanges)); - } + // With no full-text index file at all, every row is unindexed: FULL and DETAIL + // modes must still scan the raw data instead of silently returning nothing (FAST + // mode stays index-only — unindexedRanges returns empty there). + List rawRowRanges = + new DataEvolutionGlobalIndexCoverage( + table, + snapshot, + partitionFilter, + fullTextIndexFiles, + table.coreOptions().fullTextIndexSearchMode()) + .unindexedRanges(textColumnIds, null); + if (!rawRowRanges.isEmpty()) { + splits.add(new RawFullTextSearchSplit(rawRowRanges)); } @Nullable Snapshot planSnapshot = snapshot; diff --git a/paimon-core/src/main/java/org/apache/paimon/table/source/RawFullTextReadImpl.java b/paimon-core/src/main/java/org/apache/paimon/table/source/RawFullTextReadImpl.java index 44a7e63a3609..a9c51bc1c5bc 100644 --- a/paimon-core/src/main/java/org/apache/paimon/table/source/RawFullTextReadImpl.java +++ b/paimon-core/src/main/java/org/apache/paimon/table/source/RawFullTextReadImpl.java @@ -35,6 +35,7 @@ import org.apache.paimon.index.GlobalIndexMeta; import org.apache.paimon.index.IndexFileMeta; import org.apache.paimon.index.IndexPathFactory; +import org.apache.paimon.index.pkfulltext.PkFullTextIndexFile; import org.apache.paimon.options.Options; import org.apache.paimon.partition.PartitionPredicate; import org.apache.paimon.predicate.Predicate; @@ -204,12 +205,8 @@ private Map createRawFullTextIndexes( Map rawIndexes = new HashMap<>(); long rowRangeStart = rawRowRanges.get(0).from; long rowRangeEnd = rawRowRanges.get(rawRowRanges.size() - 1).to; - String fallbackIndexType = firstIndexType(splitsByColumn); String column = textColumn.name(); - String indexType = indexType(column, splitsByColumn); - if (indexType == null) { - indexType = checkNotNull(fallbackIndexType); - } + String indexType = resolveRawIndexType(column, splitsByColumn); GlobalIndexer globalIndexer = GlobalIndexerFactoryUtils.load(indexType).create(textColumn, rawSearchOptions()); try { @@ -237,6 +234,22 @@ private Map createRawFullTextIndexes( return rawIndexes; } + /** The type of the temporary raw index, resolved like the persistent index's type. */ + static String resolveRawIndexType( + String column, Map> splitsByColumn) { + String indexType = indexType(column, splitsByColumn); + if (indexType == null) { + indexType = firstIndexType(splitsByColumn); + } + if (indexType == null) { + // No full-text index file exists at all, so no split can name the + // implementation: full text has the fixed 'full-text' implementation (the + // module must be on the reader classpath anyway, like for indexed reads). + indexType = PkFullTextIndexFile.INDEX_TYPE; + } + return indexType; + } + @Nullable private static String indexType( String column, Map> splitsByColumn) { diff --git a/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java b/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java index 281f3ce32c31..45468fd74461 100644 --- a/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java +++ b/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java @@ -37,6 +37,7 @@ import org.apache.paimon.index.IndexFileMeta; import org.apache.paimon.index.pk.PrimaryKeyIndexSourceFile; import org.apache.paimon.index.pk.PrimaryKeyIndexSourceMeta; +import org.apache.paimon.index.pkfulltext.PkFullTextIndexFile; import org.apache.paimon.io.CompactIncrement; import org.apache.paimon.io.DataIncrement; import org.apache.paimon.options.Options; @@ -71,7 +72,9 @@ import java.util.ArrayList; import java.util.Arrays; import java.util.Collections; +import java.util.HashMap; import java.util.List; +import java.util.Map; import static org.apache.paimon.table.source.DeletionVectorTestUtils.commitDeletionVectors; import static org.assertj.core.api.Assertions.assertThat; @@ -203,6 +206,90 @@ public void testFullTextSearchPinsLiveRowFilterToPlanSnapshot() throws Exception assertThat(result.results()).containsExactlyInAnyOrder(0L, 1L, 2L, 3L); } + @Test + public void testFullTextWithoutIndexFilesScansRawDataInFullMode() throws Exception { + Identifier identifier = identifier("full_text_no_index_files"); + Schema schema = + Schema.newBuilder() + .column("id", DataTypes.INT()) + .column(TEXT_FIELD_NAME, DataTypes.STRING()) + .option(CoreOptions.BUCKET.key(), "-1") + .option(CoreOptions.ROW_TRACKING_ENABLED.key(), "true") + .option(CoreOptions.DATA_EVOLUTION_ENABLED.key(), "true") + .option(CoreOptions.FULL_TEXT_INDEX_SEARCH_MODE.key(), "full") + .build(); + catalog.createTable(identifier, schema, false); + FileStoreTable table = getTable(identifier); + + // no index has ever been built: every row is unindexed, and FULL mode promises to + // scan the raw data — it must not silently return an empty plan + writeDocuments(table, new String[] {"keyword here", "no match"}); + + FullTextSearchBuilder builder = + table.newFullTextSearchBuilder() + .withQuery(TEXT_FIELD_NAME, matchQuery("keyword")) + .withLimit(2); + FullTextScan.Plan plan = builder.newFullTextScan().scan(); + assertThat(plan.splits()).anyMatch(RawFullTextSearchSplit.class::isInstance); + + // with no index file to name the implementation, the temporary raw index uses + // the fixed 'full-text' implementation; splits without files do not change that + assertThat(RawFullTextReadImpl.resolveRawIndexType(TEXT_FIELD_NAME, Collections.emptyMap())) + .isEqualTo(PkFullTextIndexFile.INDEX_TYPE); + Map> emptySplits = new HashMap<>(); + emptySplits.put( + TEXT_FIELD_NAME, + Collections.singletonList( + new IndexFullTextSearchSplit( + TEXT_FIELD_NAME, 0, 10, Collections.emptyList()))); + assertThat(RawFullTextReadImpl.resolveRawIndexType(TEXT_FIELD_NAME, emptySplits)) + .isEqualTo(PkFullTextIndexFile.INDEX_TYPE); + Map> otherColumnSplits = new HashMap<>(); + otherColumnSplits.put( + "other", + Collections.singletonList( + new IndexFullTextSearchSplit("other", 0, 10, Collections.emptyList()))); + assertThat(RawFullTextReadImpl.resolveRawIndexType(TEXT_FIELD_NAME, otherColumnSplits)) + .isEqualTo(PkFullTextIndexFile.INDEX_TYPE); + + // DETAIL mode also recovers: its raw split covers the data files' ranges + FileStoreTable detailTable = + table.copy( + Collections.singletonMap( + CoreOptions.FULL_TEXT_INDEX_SEARCH_MODE.key(), "detail")); + FullTextScan.Plan detailPlan = + detailTable + .newFullTextSearchBuilder() + .withQuery(TEXT_FIELD_NAME, matchQuery("keyword")) + .withLimit(2) + .newFullTextScan() + .scan(); + assertThat(detailPlan.splits()).anyMatch(RawFullTextSearchSplit.class::isInstance); + + // FAST mode stays index-only: with zero index files its plan is empty + Identifier fastIdentifier = identifier("full_text_no_index_files_fast"); + Schema fastSchema = + Schema.newBuilder() + .column("id", DataTypes.INT()) + .column(TEXT_FIELD_NAME, DataTypes.STRING()) + .option(CoreOptions.BUCKET.key(), "-1") + .option(CoreOptions.ROW_TRACKING_ENABLED.key(), "true") + .option(CoreOptions.DATA_EVOLUTION_ENABLED.key(), "true") + .build(); + catalog.createTable(fastIdentifier, fastSchema, false); + FileStoreTable fastTable = getTable(fastIdentifier); + writeDocuments(fastTable, new String[] {"keyword here"}); + + FullTextScan.Plan fastPlan = + fastTable + .newFullTextSearchBuilder() + .withQuery(TEXT_FIELD_NAME, matchQuery("keyword")) + .withLimit(2) + .newFullTextScan() + .scan(); + assertThat(fastPlan.splits()).noneMatch(RawFullTextSearchSplit.class::isInstance); + } + @Test public void testFullTextRawFallbackPinsDataReadToPlanSnapshot() throws Exception { Identifier identifier = identifier("full_text_pinned_raw_fallback");