diff --git a/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextScan.java b/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextScan.java index a3636b3e424c..4e6aeb2279ea 100644 --- a/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextScan.java +++ b/paimon-core/src/main/java/org/apache/paimon/table/source/DataEvolutionFullTextScan.java @@ -166,18 +166,19 @@ public Plan scan() { Collections.singletonList(selection.fileRange)))); } - if (!fullTextIndexFiles.isEmpty()) { - List rawRowRanges = - new DataEvolutionGlobalIndexCoverage( - table, - snapshot, - partitionFilter, - fullTextIndexFiles, - table.coreOptions().fullTextIndexSearchMode()) - .unindexedRanges(textColumnIds, null); - if (!rawRowRanges.isEmpty()) { - splits.add(new RawFullTextSearchSplit(rawRowRanges)); - } + // With no full-text index file at all, every row is unindexed: FULL and DETAIL + // modes must still scan the raw data instead of silently returning nothing (FAST + // mode stays index-only — unindexedRanges returns empty there). + List rawRowRanges = + new DataEvolutionGlobalIndexCoverage( + table, + snapshot, + partitionFilter, + fullTextIndexFiles, + table.coreOptions().fullTextIndexSearchMode()) + .unindexedRanges(textColumnIds, null); + if (!rawRowRanges.isEmpty()) { + splits.add(new RawFullTextSearchSplit(rawRowRanges)); } @Nullable Snapshot planSnapshot = snapshot; diff --git a/paimon-core/src/main/java/org/apache/paimon/table/source/RawFullTextReadImpl.java b/paimon-core/src/main/java/org/apache/paimon/table/source/RawFullTextReadImpl.java index 44a7e63a3609..a9c51bc1c5bc 100644 --- a/paimon-core/src/main/java/org/apache/paimon/table/source/RawFullTextReadImpl.java +++ b/paimon-core/src/main/java/org/apache/paimon/table/source/RawFullTextReadImpl.java @@ -35,6 +35,7 @@ import org.apache.paimon.index.GlobalIndexMeta; import org.apache.paimon.index.IndexFileMeta; import org.apache.paimon.index.IndexPathFactory; +import org.apache.paimon.index.pkfulltext.PkFullTextIndexFile; import org.apache.paimon.options.Options; import org.apache.paimon.partition.PartitionPredicate; import org.apache.paimon.predicate.Predicate; @@ -204,12 +205,8 @@ private Map createRawFullTextIndexes( Map rawIndexes = new HashMap<>(); long rowRangeStart = rawRowRanges.get(0).from; long rowRangeEnd = rawRowRanges.get(rawRowRanges.size() - 1).to; - String fallbackIndexType = firstIndexType(splitsByColumn); String column = textColumn.name(); - String indexType = indexType(column, splitsByColumn); - if (indexType == null) { - indexType = checkNotNull(fallbackIndexType); - } + String indexType = resolveRawIndexType(column, splitsByColumn); GlobalIndexer globalIndexer = GlobalIndexerFactoryUtils.load(indexType).create(textColumn, rawSearchOptions()); try { @@ -237,6 +234,22 @@ private Map createRawFullTextIndexes( return rawIndexes; } + /** The type of the temporary raw index, resolved like the persistent index's type. */ + static String resolveRawIndexType( + String column, Map> splitsByColumn) { + String indexType = indexType(column, splitsByColumn); + if (indexType == null) { + indexType = firstIndexType(splitsByColumn); + } + if (indexType == null) { + // No full-text index file exists at all, so no split can name the + // implementation: full text has the fixed 'full-text' implementation (the + // module must be on the reader classpath anyway, like for indexed reads). + indexType = PkFullTextIndexFile.INDEX_TYPE; + } + return indexType; + } + @Nullable private static String indexType( String column, Map> splitsByColumn) { diff --git a/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java b/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java index 281f3ce32c31..45468fd74461 100644 --- a/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java +++ b/paimon-core/src/test/java/org/apache/paimon/table/source/FullTextSearchBuilderTest.java @@ -37,6 +37,7 @@ import org.apache.paimon.index.IndexFileMeta; import org.apache.paimon.index.pk.PrimaryKeyIndexSourceFile; import org.apache.paimon.index.pk.PrimaryKeyIndexSourceMeta; +import org.apache.paimon.index.pkfulltext.PkFullTextIndexFile; import org.apache.paimon.io.CompactIncrement; import org.apache.paimon.io.DataIncrement; import org.apache.paimon.options.Options; @@ -71,7 +72,9 @@ import java.util.ArrayList; import java.util.Arrays; import java.util.Collections; +import java.util.HashMap; import java.util.List; +import java.util.Map; import static org.apache.paimon.table.source.DeletionVectorTestUtils.commitDeletionVectors; import static org.assertj.core.api.Assertions.assertThat; @@ -203,6 +206,90 @@ public void testFullTextSearchPinsLiveRowFilterToPlanSnapshot() throws Exception assertThat(result.results()).containsExactlyInAnyOrder(0L, 1L, 2L, 3L); } + @Test + public void testFullTextWithoutIndexFilesScansRawDataInFullMode() throws Exception { + Identifier identifier = identifier("full_text_no_index_files"); + Schema schema = + Schema.newBuilder() + .column("id", DataTypes.INT()) + .column(TEXT_FIELD_NAME, DataTypes.STRING()) + .option(CoreOptions.BUCKET.key(), "-1") + .option(CoreOptions.ROW_TRACKING_ENABLED.key(), "true") + .option(CoreOptions.DATA_EVOLUTION_ENABLED.key(), "true") + .option(CoreOptions.FULL_TEXT_INDEX_SEARCH_MODE.key(), "full") + .build(); + catalog.createTable(identifier, schema, false); + FileStoreTable table = getTable(identifier); + + // no index has ever been built: every row is unindexed, and FULL mode promises to + // scan the raw data — it must not silently return an empty plan + writeDocuments(table, new String[] {"keyword here", "no match"}); + + FullTextSearchBuilder builder = + table.newFullTextSearchBuilder() + .withQuery(TEXT_FIELD_NAME, matchQuery("keyword")) + .withLimit(2); + FullTextScan.Plan plan = builder.newFullTextScan().scan(); + assertThat(plan.splits()).anyMatch(RawFullTextSearchSplit.class::isInstance); + + // with no index file to name the implementation, the temporary raw index uses + // the fixed 'full-text' implementation; splits without files do not change that + assertThat(RawFullTextReadImpl.resolveRawIndexType(TEXT_FIELD_NAME, Collections.emptyMap())) + .isEqualTo(PkFullTextIndexFile.INDEX_TYPE); + Map> emptySplits = new HashMap<>(); + emptySplits.put( + TEXT_FIELD_NAME, + Collections.singletonList( + new IndexFullTextSearchSplit( + TEXT_FIELD_NAME, 0, 10, Collections.emptyList()))); + assertThat(RawFullTextReadImpl.resolveRawIndexType(TEXT_FIELD_NAME, emptySplits)) + .isEqualTo(PkFullTextIndexFile.INDEX_TYPE); + Map> otherColumnSplits = new HashMap<>(); + otherColumnSplits.put( + "other", + Collections.singletonList( + new IndexFullTextSearchSplit("other", 0, 10, Collections.emptyList()))); + assertThat(RawFullTextReadImpl.resolveRawIndexType(TEXT_FIELD_NAME, otherColumnSplits)) + .isEqualTo(PkFullTextIndexFile.INDEX_TYPE); + + // DETAIL mode also recovers: its raw split covers the data files' ranges + FileStoreTable detailTable = + table.copy( + Collections.singletonMap( + CoreOptions.FULL_TEXT_INDEX_SEARCH_MODE.key(), "detail")); + FullTextScan.Plan detailPlan = + detailTable + .newFullTextSearchBuilder() + .withQuery(TEXT_FIELD_NAME, matchQuery("keyword")) + .withLimit(2) + .newFullTextScan() + .scan(); + assertThat(detailPlan.splits()).anyMatch(RawFullTextSearchSplit.class::isInstance); + + // FAST mode stays index-only: with zero index files its plan is empty + Identifier fastIdentifier = identifier("full_text_no_index_files_fast"); + Schema fastSchema = + Schema.newBuilder() + .column("id", DataTypes.INT()) + .column(TEXT_FIELD_NAME, DataTypes.STRING()) + .option(CoreOptions.BUCKET.key(), "-1") + .option(CoreOptions.ROW_TRACKING_ENABLED.key(), "true") + .option(CoreOptions.DATA_EVOLUTION_ENABLED.key(), "true") + .build(); + catalog.createTable(fastIdentifier, fastSchema, false); + FileStoreTable fastTable = getTable(fastIdentifier); + writeDocuments(fastTable, new String[] {"keyword here"}); + + FullTextScan.Plan fastPlan = + fastTable + .newFullTextSearchBuilder() + .withQuery(TEXT_FIELD_NAME, matchQuery("keyword")) + .withLimit(2) + .newFullTextScan() + .scan(); + assertThat(fastPlan.splits()).noneMatch(RawFullTextSearchSplit.class::isInstance); + } + @Test public void testFullTextRawFallbackPinsDataReadToPlanSnapshot() throws Exception { Identifier identifier = identifier("full_text_pinned_raw_fallback");