From 8086a5d415d850c80accf1f1043a285640110e0e Mon Sep 17 00:00:00 2001 From: Refrain Date: Wed, 2 Sep 2026 09:12:40 +0800 Subject: [PATCH] [fix](fe) Infer compression before skipping empty schema files ### What problem does this PR solve? Issue Number: None Related PR: None Problem Summary: Schema inference skips empty compressed CSV and JSON files using format-specific minimum sizes, but the check used only the explicitly configured compression type. With the default UNKNOWN type, an empty gzip file selected by a glob was treated as non-empty and prevented a later valid file from supplying the schema. Reuse the existing path-based compression inference before applying the empty-content threshold. ### Release note File table-valued functions now skip empty gzip CSV and JSON files during schema inference when compression is inferred from the path. ### Check List (For Author) - Test: Regression test and manual test - Regression test: test_empty_first_compressed_schema for CSV and JSON gzip globs - Manual test: local CSV and JSON DESC FUNCTION and queries returned the later valid file rows - Behavior changed: Yes. Empty compressed files no longer block schema inference for a matching glob. - Does this need documentation: No --- .../ExternalFileTableValuedFunction.java | 4 +- .../test_empty_first_compressed_schema.out | 9 ++++ .../csv/a_empty.csv.gz | Bin 0 -> 20 bytes .../csv/b_valid.csv.gz | Bin 0 -> 60 bytes .../json/a_empty.jsonl.gz | Bin 0 -> 20 bytes .../json/b_valid.jsonl.gz | Bin 0 -> 79 bytes .../test_empty_first_compressed_schema.groovy | 47 ++++++++++++++++++ 7 files changed, 59 insertions(+), 1 deletion(-) create mode 100644 regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema.out create mode 100644 regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/csv/a_empty.csv.gz create mode 100644 regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/csv/b_valid.csv.gz create mode 100644 regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/json/a_empty.jsonl.gz create mode 100644 regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/json/b_valid.jsonl.gz create mode 100644 regression-test/suites/external_table_p0/tvf/test_empty_first_compressed_schema.groovy diff --git a/fe/fe-core/src/main/java/org/apache/doris/tablefunction/ExternalFileTableValuedFunction.java b/fe/fe-core/src/main/java/org/apache/doris/tablefunction/ExternalFileTableValuedFunction.java index 01abd8b482123e..85e8c8aa0665a1 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/tablefunction/ExternalFileTableValuedFunction.java +++ b/fe/fe-core/src/main/java/org/apache/doris/tablefunction/ExternalFileTableValuedFunction.java @@ -560,7 +560,9 @@ private boolean isFileContentEmpty(TBrokerFileStatus fileStatus) { if (Util.isCsvFormat(fileFormatProperties.getFileFormatType()) || fileFormatProperties.getFileFormatType() == TFileFormatType.FORMAT_JSON) { int magicNumberBytes = 0; - switch (fileFormatProperties.getCompressionType()) { + TFileCompressType compressType = Util.getOrInferCompressType( + fileFormatProperties.getCompressionType(), fileStatus.getPath()); + switch (compressType) { case GZ: magicNumberBytes = 20; break; diff --git a/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema.out b/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema.out new file mode 100644 index 00000000000000..9afdca4650e56f --- /dev/null +++ b/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema.out @@ -0,0 +1,9 @@ +-- This file is automatically generated. You should know what you did if you want to edit this +-- !csv_empty_first_gzip -- +1 alpha 1.10 +2 beta 2.20 + +-- !json_empty_first_gzip -- +1 beijing 1.1 +2 shanghai 2.2 + diff --git a/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/csv/a_empty.csv.gz b/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/csv/a_empty.csv.gz new file mode 100644 index 0000000000000000000000000000000000000000..229151a5a27ab0cc4661f529cc0eda27e3c03e10 GIT binary patch literal 20 Rcmb2|=3oE=W@ZQtBmoVe0J#7F literal 0 HcmV?d00001 diff --git a/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/csv/b_valid.csv.gz b/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/csv/b_valid.csv.gz new file mode 100644 index 0000000000000000000000000000000000000000..5ac60b417f91b0ddf3c60ae73b6af44936a14427 GIT binary patch literal 60 zcmb2|=3oE==F>hGPkNv6y?WADSM!w5Gove>XEaZETrs?2Z1TY9lBbuh=M}?C#zqgA N7^YjTH`4%W0syP~7KH!+ literal 0 HcmV?d00001 diff --git a/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/json/a_empty.jsonl.gz b/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/json/a_empty.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..229151a5a27ab0cc4661f529cc0eda27e3c03e10 GIT binary patch literal 20 Rcmb2|=3oE=W@ZQtBmoVe0J#7F literal 0 HcmV?d00001 diff --git a/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/json/b_valid.jsonl.gz b/regression-test/data/external_table_p0/tvf/test_empty_first_compressed_schema/json/b_valid.jsonl.gz new file mode 100644 index 0000000000000000000000000000000000000000..d7dac3b26850ec90b09ac876114fbc62317dbb11 GIT binary patch literal 79 zcmb2|=3oE==G9@Rd;&KaT?zC$r?WC-Q;?U}*)ylkc!ysKJmssadCG_F)Jw^0MWM}8 hB@a#&O;>rgG+iZ4X{GTc<5f?Y7$oA;P5go8006qn9)17- literal 0 HcmV?d00001 diff --git a/regression-test/suites/external_table_p0/tvf/test_empty_first_compressed_schema.groovy b/regression-test/suites/external_table_p0/tvf/test_empty_first_compressed_schema.groovy new file mode 100644 index 00000000000000..2e164c2cbe9682 --- /dev/null +++ b/regression-test/suites/external_table_p0/tvf/test_empty_first_compressed_schema.groovy @@ -0,0 +1,47 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +suite("test_empty_first_compressed_schema", "p0,external") { + String ak = getS3AK() + String sk = getS3SK() + String s3Endpoint = getS3Endpoint() + String bucket = context.config.otherConfigs.get("s3BucketName") + String baseUri = "https://${bucket}.${s3Endpoint}/regression/tvf/test_empty_first_compressed_schema" + + order_qt_csv_empty_first_gzip """ + select id, name, metric from + s3( + "URI" = "${baseUri}/csv/*.csv.gz", + "s3.access_key" = "${ak}", + "s3.secret_key" = "${sk}", + "FORMAT" = "csv_with_names", + "column_separator" = ",", + "use_path_style" = "false") order by id; + """ + + order_qt_json_empty_first_gzip """ + select id, city, metric from + s3( + "URI" = "${baseUri}/json/*.jsonl.gz", + "s3.access_key" = "${ak}", + "s3.secret_key" = "${sk}", + "FORMAT" = "json", + "read_json_by_line" = "true", + "fuzzy_parse" = "true", + "use_path_style" = "false") order by id; + """ +}