be/src/exec/scan/file_scanner_v2.cpp
Line | Count | Source |
1 | | // Licensed to the Apache Software Foundation (ASF) under one |
2 | | // or more contributor license agreements. See the NOTICE file |
3 | | // distributed with this work for additional information |
4 | | // regarding copyright ownership. The ASF licenses this file |
5 | | // to you under the Apache License, Version 2.0 (the |
6 | | // "License"); you may not use this file except in compliance |
7 | | // with the License. You may obtain a copy of the License at |
8 | | // |
9 | | // http://www.apache.org/licenses/LICENSE-2.0 |
10 | | // |
11 | | // Unless required by applicable law or agreed to in writing, |
12 | | // software distributed under the License is distributed on an |
13 | | // "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY |
14 | | // KIND, either express or implied. See the License for the |
15 | | // specific language governing permissions and limitations |
16 | | // under the License. |
17 | | |
18 | | #include "exec/scan/file_scanner_v2.h" |
19 | | |
20 | | #include <gen_cpp/Exprs_types.h> |
21 | | #include <gen_cpp/PlanNodes_types.h> |
22 | | |
23 | | #include <algorithm> |
24 | | #include <map> |
25 | | #include <memory> |
26 | | #include <optional> |
27 | | #include <string> |
28 | | #include <utility> |
29 | | |
30 | | #include "common/cast_set.h" |
31 | | #include "common/config.h" |
32 | | #include "common/consts.h" |
33 | | #include "common/metrics/doris_metrics.h" |
34 | | #include "common/status.h" |
35 | | #include "core/assert_cast.h" |
36 | | #include "core/block/column_with_type_and_name.h" |
37 | | #include "core/column/column.h" |
38 | | #include "core/data_type/data_type.h" |
39 | | #include "core/data_type/data_type_nullable.h" |
40 | | #include "core/data_type_serde/data_type_serde.h" |
41 | | #include "core/string_ref.h" |
42 | | #include "exec/common/util.hpp" |
43 | | #include "exec/operator/scan_operator.h" |
44 | | #include "exec/scan/access_path_parser.h" |
45 | | #include "exec/scan/file_scan_io_context.h" |
46 | | #include "exprs/runtime_filter_expr.h" |
47 | | #include "exprs/vexpr.h" |
48 | | #include "exprs/vexpr_context.h" |
49 | | #include "exprs/vslot_ref.h" |
50 | | #include "format/format_common.h" |
51 | | #include "format/table/iceberg_scan_semantics.h" |
52 | | #include "format_v2/column_mapper.h" |
53 | | #include "format_v2/jni/iceberg_sys_table_reader.h" |
54 | | #include "format_v2/jni/jdbc_reader.h" |
55 | | #include "format_v2/jni/max_compute_jni_reader.h" |
56 | | #include "format_v2/jni/trino_connector_jni_reader.h" |
57 | | #include "format_v2/table/hive_reader.h" |
58 | | #include "format_v2/table/hudi_reader.h" |
59 | | #include "format_v2/table/iceberg_position_delete_sys_table_reader.h" |
60 | | #include "format_v2/table/iceberg_reader.h" |
61 | | #include "format_v2/table/paimon_reader.h" |
62 | | #include "format_v2/table/remote_doris_reader.h" |
63 | | #include "format_v2/table_reader.h" |
64 | | #include "io/cache/block_file_cache_profile.h" |
65 | | #include "io/fs/file_meta_cache.h" |
66 | | #include "io/io_common.h" |
67 | | #include "runtime/descriptors.h" |
68 | | #include "runtime/exec_env.h" |
69 | | #include "runtime/file_scan_profile.h" |
70 | | #include "runtime/runtime_state.h" |
71 | | #include "service/backend_options.h" |
72 | | #include "storage/id_manager.h" |
73 | | |
74 | | namespace doris { |
75 | | namespace { |
76 | | |
77 | | constexpr int kIcebergPositionDeleteContent = 1; |
78 | | constexpr int kIcebergDeletionVectorContent = 3; |
79 | | |
80 | 41 | std::string table_format_name(const TFileRangeDesc& range) { |
81 | 41 | return range.__isset.table_format_params ? range.table_format_params.table_format_type |
82 | 41 | : "NotSet"; |
83 | 41 | } |
84 | | |
85 | | TFileFormatType::type get_range_format_type(const TFileScanRangeParams& params, |
86 | 45 | const TFileRangeDesc& range) { |
87 | 45 | return range.__isset.format_type ? range.format_type : params.format_type; |
88 | 45 | } |
89 | | |
90 | 32 | bool is_supported_table_format(const TFileRangeDesc& range) { |
91 | 32 | const auto table_format = table_format_name(range); |
92 | 32 | if (table_format == "hudi" && range.__isset.table_format_params && |
93 | 32 | range.table_format_params.__isset.hudi_params && |
94 | 32 | range.table_format_params.hudi_params.__isset.delta_logs && |
95 | 32 | !range.table_format_params.hudi_params.delta_logs.empty()) { |
96 | | // Hudi MOR splits need log-file merge semantics and must stay on the existing JNI path. |
97 | | // FileScannerV2 currently supports native Parquet data files only. |
98 | 1 | return false; |
99 | 1 | } |
100 | 31 | return table_format == "NotSet" || table_format == "tvf" || table_format == "hive" || |
101 | 31 | table_format == "iceberg" || table_format == "paimon" || table_format == "hudi"; |
102 | 32 | } |
103 | | |
104 | 3 | bool is_supported_arrow_table_format(const TFileRangeDesc& range) { |
105 | 3 | return table_format_name(range) == "remote_doris"; |
106 | 3 | } |
107 | | |
108 | 5 | bool is_supported_jni_table_format(const TFileRangeDesc& range) { |
109 | 5 | const auto table_format = table_format_name(range); |
110 | 5 | if (table_format == "paimon") { |
111 | 2 | return range.__isset.table_format_params && |
112 | 2 | range.table_format_params.__isset.paimon_params && |
113 | 2 | range.table_format_params.paimon_params.__isset.reader_type && |
114 | 2 | range.table_format_params.paimon_params.reader_type == TPaimonReaderType::PAIMON_JNI; |
115 | 2 | } |
116 | 3 | return table_format == "jdbc" || table_format == "iceberg" || table_format == "hudi" || |
117 | 3 | table_format == "max_compute" || table_format == "trino_connector"; |
118 | 5 | } |
119 | | |
120 | 0 | bool is_iceberg_position_deletes_sys_table(const TFileRangeDesc& range) { |
121 | 0 | return range.__isset.table_format_params && |
122 | 0 | range.table_format_params.table_format_type == "iceberg" && |
123 | 0 | range.table_format_params.__isset.iceberg_params && |
124 | 0 | range.table_format_params.iceberg_params.__isset.content && |
125 | 0 | (range.table_format_params.iceberg_params.content == kIcebergPositionDeleteContent || |
126 | 0 | range.table_format_params.iceberg_params.content == kIcebergDeletionVectorContent); |
127 | 0 | } |
128 | | |
129 | 19 | bool is_csv_format(TFileFormatType::type format_type) { |
130 | 19 | switch (format_type) { |
131 | 2 | case TFileFormatType::FORMAT_CSV_PLAIN: |
132 | 3 | case TFileFormatType::FORMAT_CSV_GZ: |
133 | 4 | case TFileFormatType::FORMAT_CSV_BZ2: |
134 | 5 | case TFileFormatType::FORMAT_CSV_LZ4FRAME: |
135 | 6 | case TFileFormatType::FORMAT_CSV_LZ4BLOCK: |
136 | 7 | case TFileFormatType::FORMAT_CSV_LZOP: |
137 | 8 | case TFileFormatType::FORMAT_CSV_DEFLATE: |
138 | 9 | case TFileFormatType::FORMAT_CSV_SNAPPYBLOCK: |
139 | 10 | case TFileFormatType::FORMAT_PROTO: |
140 | 10 | return true; |
141 | 9 | default: |
142 | 9 | return false; |
143 | 19 | } |
144 | 19 | } |
145 | | |
146 | 9 | bool is_text_format(TFileFormatType::type format_type) { |
147 | 9 | return format_type == TFileFormatType::FORMAT_TEXT; |
148 | 9 | } |
149 | | |
150 | 7 | bool is_json_format(TFileFormatType::type format_type) { |
151 | 7 | return format_type == TFileFormatType::FORMAT_JSON; |
152 | 7 | } |
153 | | |
154 | 5 | bool is_native_format(TFileFormatType::type format_type) { |
155 | 5 | return format_type == TFileFormatType::FORMAT_NATIVE; |
156 | 5 | } |
157 | | |
158 | 6 | bool is_partition_slot(const TFileScanSlotInfo& slot_info, const std::string& column_name) { |
159 | 6 | if (column_name.starts_with(BeConsts::GLOBAL_ROWID_COL) || |
160 | 6 | column_name == BeConsts::ICEBERG_ROWID_COL) { |
161 | 2 | return false; |
162 | 2 | } |
163 | 4 | return slot_info.__isset.category ? slot_info.category == TColumnCategory::PARTITION_KEY |
164 | 4 | : !slot_info.is_file_slot; |
165 | 6 | } |
166 | | |
167 | 8 | bool is_data_file_slot(const TFileScanSlotInfo& slot_info, const std::string& column_name) { |
168 | 8 | if (column_name.starts_with(BeConsts::GLOBAL_ROWID_COL) || |
169 | 8 | column_name == BeConsts::ICEBERG_ROWID_COL) { |
170 | 2 | return false; |
171 | 2 | } |
172 | | // CSV and other non-self-describing formats need FE slot descriptors for only the columns that |
173 | | // are physically read from the file. Partition/default/virtual columns stay in TableReader's |
174 | | // mapping layer and are materialized after the file-local block is read. New FE provides an |
175 | | // explicit category; old FE falls back to `is_file_slot`. |
176 | 6 | if (slot_info.__isset.category) { |
177 | 4 | return slot_info.category == TColumnCategory::REGULAR || |
178 | 4 | slot_info.category == TColumnCategory::GENERATED; |
179 | 4 | } |
180 | 2 | return slot_info.is_file_slot; |
181 | 6 | } |
182 | | |
183 | | Status rewrite_slot_refs_to_global_index( |
184 | | VExprSPtr* expr, |
185 | 7 | const std::unordered_map<int32_t, format::GlobalIndex>& slot_id_to_global_index) { |
186 | 7 | DORIS_CHECK(expr != nullptr); |
187 | 7 | if (*expr == nullptr) { |
188 | 0 | return Status::OK(); |
189 | 0 | } |
190 | 7 | if (auto* runtime_filter = dynamic_cast<RuntimeFilterExpr*>(expr->get()); |
191 | 7 | runtime_filter != nullptr) { |
192 | 1 | auto impl = runtime_filter->get_impl(); |
193 | 1 | DORIS_CHECK(impl != nullptr); |
194 | 1 | RETURN_IF_ERROR(rewrite_slot_refs_to_global_index(&impl, slot_id_to_global_index)); |
195 | 1 | runtime_filter->set_impl(std::move(impl)); |
196 | 1 | return Status::OK(); |
197 | 1 | } |
198 | 6 | if ((*expr)->is_slot_ref()) { |
199 | 4 | const auto* slot_ref = assert_cast<const VSlotRef*>(expr->get()); |
200 | 4 | const auto global_index_it = slot_id_to_global_index.find(slot_ref->slot_id()); |
201 | 4 | if (global_index_it == slot_id_to_global_index.end()) { |
202 | 1 | return Status::InternalError( |
203 | 1 | "Can not resolve source slot id {} to a table global index for column {}", |
204 | 1 | slot_ref->slot_id(), slot_ref->column_name()); |
205 | 1 | } |
206 | 3 | const auto global_index = global_index_it->second; |
207 | 3 | *expr = VSlotRef::create_shared(cast_set<int>(global_index.value()), |
208 | 3 | cast_set<int>(global_index.value()), -1, |
209 | 3 | slot_ref->data_type(), slot_ref->column_name()); |
210 | 3 | RETURN_IF_ERROR(expr->get()->prepare(nullptr, RowDescriptor(), nullptr)); |
211 | 3 | return Status::OK(); |
212 | 3 | } |
213 | 2 | auto children = (*expr)->children(); |
214 | 2 | for (auto& child : children) { |
215 | 2 | if (child == nullptr) { |
216 | 0 | continue; |
217 | 0 | } |
218 | 2 | RETURN_IF_ERROR(rewrite_slot_refs_to_global_index(&child, slot_id_to_global_index)); |
219 | 2 | } |
220 | 2 | (*expr)->set_children(std::move(children)); |
221 | 2 | return Status::OK(); |
222 | 2 | } |
223 | | |
224 | | } // namespace |
225 | | |
226 | | #ifdef BE_TEST |
227 | | FileScannerV2::FileScannerV2(RuntimeState* state, RuntimeProfile* profile, |
228 | | std::unique_ptr<format::TableReader> table_reader) |
229 | 1 | : Scanner(state, profile), _table_reader(std::move(table_reader)) {} |
230 | | |
231 | | Status FileScannerV2::TEST_validate_scan_range(const TFileScanRangeParams& params, |
232 | 2 | const TFileRangeDesc& range) { |
233 | 2 | return _validate_scan_range(params, range); |
234 | 2 | } |
235 | | |
236 | | Status FileScannerV2::TEST_to_file_format(TFileFormatType::type format_type, |
237 | 16 | format::FileFormat* file_format) { |
238 | 16 | return _to_file_format(format_type, file_format); |
239 | 16 | } |
240 | | |
241 | | bool FileScannerV2::TEST_is_partition_slot(const TFileScanSlotInfo& slot_info, |
242 | 6 | const std::string& column_name) { |
243 | 6 | return is_partition_slot(slot_info, column_name); |
244 | 6 | } |
245 | | |
246 | | bool FileScannerV2::TEST_is_data_file_slot(const TFileScanSlotInfo& slot_info, |
247 | 8 | const std::string& column_name) { |
248 | 8 | return is_data_file_slot(slot_info, column_name); |
249 | 8 | } |
250 | | |
251 | | Status FileScannerV2::TEST_rewrite_slot_refs_to_global_index( |
252 | | VExprSPtr* expr, |
253 | 4 | const std::unordered_map<int32_t, format::GlobalIndex>& slot_id_to_global_index) { |
254 | 4 | return rewrite_slot_refs_to_global_index(expr, slot_id_to_global_index); |
255 | 4 | } |
256 | | |
257 | | FileScannerV2::RealtimeCounterDeltas FileScannerV2::TEST_collect_realtime_counter_deltas( |
258 | | const io::FileReaderStats& file_reader_stats, |
259 | | const io::FileCacheStatistics& file_cache_statistics, |
260 | | UncachedReaderBytesStorage uncached_reader_bytes_storage, int64_t* last_read_bytes, |
261 | | int64_t* last_read_rows, int64_t* last_bytes_read_from_local, |
262 | 7 | int64_t* last_bytes_read_from_remote) { |
263 | 7 | return _collect_realtime_counter_deltas(file_reader_stats, file_cache_statistics, |
264 | 7 | uncached_reader_bytes_storage, last_read_bytes, |
265 | 7 | last_read_rows, last_bytes_read_from_local, |
266 | 7 | last_bytes_read_from_remote); |
267 | 7 | } |
268 | | |
269 | | void FileScannerV2::TEST_report_file_cache_profile( |
270 | 1 | RuntimeProfile* profile, const io::FileCacheStatistics& file_cache_statistics) { |
271 | 1 | _report_file_cache_profile(profile, file_cache_statistics); |
272 | 1 | } |
273 | | |
274 | 4 | bool FileScannerV2::TEST_should_skip_not_found(const Status& status, bool ignore_not_found) { |
275 | 4 | return _should_skip_not_found(status, ignore_not_found); |
276 | 4 | } |
277 | | |
278 | 4 | bool FileScannerV2::TEST_should_skip_empty(const Status& status, bool stopped) { |
279 | 4 | return _should_skip_empty(status, stopped); |
280 | 4 | } |
281 | | #endif |
282 | | |
283 | 44 | bool FileScannerV2::is_supported(const TFileScanRangeParams& params, const TFileRangeDesc& range) { |
284 | 44 | const auto format_type = get_range_format_type(params, range); |
285 | 44 | if (format_type == TFileFormatType::FORMAT_PARQUET || |
286 | 44 | format_type == TFileFormatType::FORMAT_ORC) { |
287 | 17 | return is_supported_table_format(range); |
288 | 27 | } else if (format_type == TFileFormatType::FORMAT_ARROW) { |
289 | 3 | return is_supported_arrow_table_format(range); |
290 | 24 | } else if (format_type == TFileFormatType::FORMAT_JNI) { |
291 | 5 | return is_supported_jni_table_format(range); |
292 | 19 | } else if (is_csv_format(format_type) || is_text_format(format_type) || |
293 | 19 | is_json_format(format_type) || is_native_format(format_type)) { |
294 | 15 | return is_supported_table_format(range); |
295 | 15 | } else { |
296 | 4 | LOG(WARNING) << "Unsupported file format type " << format_type << " for file scanner v2"; |
297 | 4 | return false; |
298 | 4 | } |
299 | 44 | } |
300 | | |
301 | | Status FileScannerV2::_validate_scan_range(const TFileScanRangeParams& params, |
302 | 2 | const TFileRangeDesc& range) { |
303 | 2 | if (!is_supported(params, range)) { |
304 | 1 | return Status::NotSupported( |
305 | 1 | "FileScannerV2 does not support table format {} with file format {}", |
306 | 1 | table_format_name(range), to_string(get_range_format_type(params, range))); |
307 | 1 | } |
308 | 1 | return Status::OK(); |
309 | 2 | } |
310 | | |
311 | | FileScannerV2::FileScannerV2(RuntimeState* state, FileScanLocalState* local_state, int64_t limit, |
312 | | std::shared_ptr<SplitSourceConnector> split_source, |
313 | | RuntimeProfile* profile, ShardedKVCache* kv_cache, |
314 | | const std::unordered_map<std::string, int>* colname_to_slot_id) |
315 | 0 | : Scanner(state, local_state, limit, profile), |
316 | 0 | _split_source(std::move(split_source)), |
317 | 0 | _kv_cache(kv_cache) { |
318 | 0 | (void)colname_to_slot_id; |
319 | 0 | if (state->get_query_ctx() != nullptr && |
320 | 0 | state->get_query_ctx()->file_scan_range_params_map.count(local_state->parent_id()) > 0) { |
321 | 0 | _params = &(state->get_query_ctx()->file_scan_range_params_map[local_state->parent_id()]); |
322 | 0 | } else { |
323 | 0 | _params = _split_source->get_params(); |
324 | 0 | } |
325 | 0 | } |
326 | | |
327 | 0 | Status FileScannerV2::init(RuntimeState* state, const VExprContextSPtrs& conjuncts) { |
328 | 0 | RETURN_IF_ERROR(Scanner::init(state, conjuncts)); |
329 | 0 | auto* profile = _local_state->scanner_profile(); |
330 | 0 | const auto hierarchy = file_scan_profile::ensure_hierarchy(profile); |
331 | 0 | _scanner_total_timer = hierarchy.scanner; |
332 | 0 | _io_timer = hierarchy.io; |
333 | 0 | _init_timer = ADD_CHILD_TIMER_WITH_LEVEL(profile, "FileScannerV2InitTime", |
334 | 0 | file_scan_profile::SCANNER, 1); |
335 | 0 | _open_timer = ADD_CHILD_TIMER_WITH_LEVEL(profile, "FileScannerV2OpenTime", |
336 | 0 | file_scan_profile::SCANNER, 1); |
337 | 0 | _get_block_timer = ADD_CHILD_TIMER_WITH_LEVEL(profile, "FileScannerV2GetBlockTime", |
338 | 0 | file_scan_profile::SCANNER, 1); |
339 | 0 | _prepare_split_timer = ADD_CHILD_TIMER_WITH_LEVEL(profile, "FileScannerV2PrepareSplitTime", |
340 | 0 | file_scan_profile::SCANNER, 1); |
341 | 0 | _get_next_range_timer = ADD_CHILD_TIMER_WITH_LEVEL(profile, "FileScannerV2GetNextRangeTime", |
342 | 0 | file_scan_profile::SCANNER, 1); |
343 | 0 | _close_timer = ADD_CHILD_TIMER_WITH_LEVEL(profile, "FileScannerV2CloseTime", |
344 | 0 | file_scan_profile::SCANNER, 1); |
345 | 0 | _empty_file_counter = ADD_CHILD_COUNTER_WITH_LEVEL(profile, "EmptyFileNum", TUnit::UNIT, |
346 | 0 | file_scan_profile::SCANNER, 1); |
347 | 0 | _not_found_file_counter = ADD_CHILD_COUNTER_WITH_LEVEL(profile, "NotFoundFileNum", TUnit::UNIT, |
348 | 0 | file_scan_profile::SCANNER, 1); |
349 | 0 | _file_counter = ADD_CHILD_COUNTER_WITH_LEVEL(profile, "FileNumber", TUnit::UNIT, |
350 | 0 | file_scan_profile::SCANNER, 1); |
351 | 0 | _file_read_bytes_counter = ADD_CHILD_COUNTER_WITH_LEVEL(profile, "FileReadBytes", TUnit::BYTES, |
352 | 0 | file_scan_profile::IO, 1); |
353 | 0 | _file_read_calls_counter = ADD_CHILD_COUNTER_WITH_LEVEL(profile, "FileReadCalls", TUnit::UNIT, |
354 | 0 | file_scan_profile::IO, 1); |
355 | 0 | _file_read_time_counter = |
356 | 0 | ADD_CHILD_TIMER_WITH_LEVEL(profile, "FileReadTime", file_scan_profile::IO, 1); |
357 | 0 | _adaptive_batch_predicted_rows_counter = ADD_CHILD_COUNTER_WITH_LEVEL( |
358 | 0 | profile, "AdaptiveBatchPredictedRows", TUnit::UNIT, file_scan_profile::SCANNER, 1); |
359 | 0 | _adaptive_batch_actual_bytes_counter = ADD_CHILD_COUNTER_WITH_LEVEL( |
360 | 0 | profile, "AdaptiveBatchActualBytes", TUnit::BYTES, file_scan_profile::SCANNER, 1); |
361 | 0 | _adaptive_batch_probe_count_counter = ADD_CHILD_COUNTER_WITH_LEVEL( |
362 | 0 | profile, "AdaptiveBatchProbeCount", TUnit::UNIT, file_scan_profile::SCANNER, 1); |
363 | 0 | SCOPED_TIMER(_scanner_total_timer); |
364 | 0 | SCOPED_TIMER(_init_timer); |
365 | 0 | _file_cache_statistics = std::make_unique<io::FileCacheStatistics>(); |
366 | 0 | _file_reader_stats = std::make_unique<io::FileReaderStats>(); |
367 | 0 | RETURN_IF_ERROR(_init_io_ctx()); |
368 | 0 | _io_ctx->file_cache_stats = _file_cache_statistics.get(); |
369 | 0 | _io_ctx->file_reader_stats = _file_reader_stats.get(); |
370 | 0 | _io_ctx->is_disposable = _state->query_options().disable_file_cache; |
371 | 0 | return Status::OK(); |
372 | 0 | } |
373 | | |
374 | 0 | Status FileScannerV2::_open_impl(RuntimeState* state) { |
375 | 0 | SCOPED_TIMER(_scanner_total_timer); |
376 | 0 | SCOPED_TIMER(_open_timer); |
377 | 0 | RETURN_IF_CANCELLED(state); |
378 | 0 | RETURN_IF_ERROR(Scanner::_open_impl(state)); |
379 | 0 | RETURN_IF_ERROR(_get_next_scan_range(&_first_scan_range)); |
380 | 0 | if (_first_scan_range) { |
381 | 0 | RETURN_IF_ERROR(_create_table_reader_for_format(_current_range, &_table_reader)); |
382 | 0 | DORIS_CHECK(_table_reader != nullptr); |
383 | 0 | RETURN_IF_ERROR(_init_expr_ctxes()); |
384 | 0 | RETURN_IF_ERROR(_init_table_reader(_current_range)); |
385 | 0 | } |
386 | 0 | return Status::OK(); |
387 | 0 | } |
388 | | |
389 | 0 | Status FileScannerV2::_get_next_scan_range(bool* has_next) { |
390 | 0 | SCOPED_TIMER(_get_next_range_timer); |
391 | 0 | DORIS_CHECK(has_next != nullptr); |
392 | 0 | RETURN_IF_ERROR(_split_source->get_next(has_next, &_current_range)); |
393 | 0 | if (*has_next) { |
394 | 0 | RETURN_IF_ERROR(_validate_scan_range(*_params, _current_range)); |
395 | 0 | } |
396 | 0 | return Status::OK(); |
397 | 0 | } |
398 | | |
399 | 0 | Status FileScannerV2::_get_block_impl(RuntimeState* state, Block* block, bool* eof) { |
400 | 0 | SCOPED_TIMER(_scanner_total_timer); |
401 | 0 | SCOPED_TIMER(_get_block_timer); |
402 | 0 | while (true) { |
403 | 0 | RETURN_IF_CANCELLED(state); |
404 | 0 | if (!_has_prepared_split) { |
405 | 0 | RETURN_IF_ERROR(_prepare_next_split(eof)); |
406 | 0 | if (*eof) { |
407 | 0 | return Status::OK(); |
408 | 0 | } |
409 | 0 | } |
410 | | |
411 | 0 | { |
412 | 0 | if (_should_run_adaptive_batch_size()) { |
413 | 0 | _table_reader->set_batch_size(_predict_reader_batch_rows()); |
414 | 0 | } |
415 | 0 | const auto status = _table_reader->get_block(block, eof); |
416 | 0 | if (_should_skip_not_found(status, config::ignore_not_found_file_in_external_table)) { |
417 | 0 | RETURN_IF_ERROR(_table_reader->abort_split()); |
418 | 0 | COUNTER_UPDATE(_not_found_file_counter, 1); |
419 | 0 | _state->update_num_finished_scan_range(1); |
420 | 0 | _has_prepared_split = false; |
421 | 0 | block->clear_column_data(cast_set<int64_t>(_projected_columns.size())); |
422 | 0 | *eof = false; |
423 | 0 | continue; |
424 | 0 | } |
425 | 0 | if (_should_skip_empty(status, _should_stop || _io_ctx->should_stop)) { |
426 | | // END_OF_FILE here means the reader discovered a valid split with no data while |
427 | | // opening or probing it, not that the Scanner has exhausted all splits. Examples |
428 | | // are a zero-byte CSV with an explicit schema and a Doris Native file containing |
429 | | // only its 12-byte header. Treat it like V1's empty-file path: finish this range, |
430 | | // discard partial reader state, and let the loop fetch the next split. |
431 | 0 | RETURN_IF_ERROR(_table_reader->abort_split()); |
432 | 0 | COUNTER_UPDATE(_empty_file_counter, 1); |
433 | 0 | _state->update_num_finished_scan_range(1); |
434 | 0 | _has_prepared_split = false; |
435 | 0 | block->clear_column_data(cast_set<int64_t>(_projected_columns.size())); |
436 | 0 | *eof = false; |
437 | 0 | continue; |
438 | 0 | } |
439 | 0 | RETURN_IF_ERROR(status); |
440 | 0 | } |
441 | 0 | if (*eof) { |
442 | 0 | _state->update_num_finished_scan_range(1); |
443 | 0 | _has_prepared_split = false; |
444 | 0 | *eof = false; |
445 | 0 | continue; |
446 | 0 | } |
447 | 0 | _update_adaptive_batch_size(*block); |
448 | 0 | return Status::OK(); |
449 | 0 | } |
450 | 0 | } |
451 | | |
452 | 0 | Status FileScannerV2::_filter_output_block(Block* block) { |
453 | 0 | return _contextualize_output_filter_status(Scanner::_filter_output_block(block), |
454 | 0 | _get_current_format_type()); |
455 | 0 | } |
456 | | |
457 | | Status FileScannerV2::_contextualize_output_filter_status(Status status, |
458 | 2 | TFileFormatType::type format_type) { |
459 | 2 | if (!status.ok() && format_type == TFileFormatType::FORMAT_ORC) { |
460 | | // Error-preserving expressions cannot be reordered into the ORC reader and therefore run |
461 | | // at the scanner boundary; keep their error context identical to ORC callback failures. |
462 | 1 | status.prepend("Orc row reader nextBatch failed. reason = "); |
463 | 1 | } |
464 | 2 | return status; |
465 | 2 | } |
466 | | |
467 | 0 | Status FileScannerV2::_prepare_next_split(bool* eos) { |
468 | 0 | SCOPED_TIMER(_prepare_split_timer); |
469 | 0 | while (true) { |
470 | 0 | bool has_next = _first_scan_range; |
471 | 0 | if (!_first_scan_range) { |
472 | 0 | RETURN_IF_ERROR(_get_next_scan_range(&has_next)); |
473 | 0 | } |
474 | 0 | _first_scan_range = false; |
475 | 0 | if (!has_next || _should_stop) { |
476 | 0 | *eos = true; |
477 | 0 | return Status::OK(); |
478 | 0 | } |
479 | 0 | DORIS_CHECK(_table_reader != nullptr); |
480 | 0 | _current_range_path = _current_range.path; |
481 | |
|
482 | 0 | const auto format_type = get_range_format_type(*_params, _current_range); |
483 | 0 | _init_adaptive_batch_size_state(format_type); |
484 | 0 | if (_block_size_predictor != nullptr) { |
485 | | // JNI readers open eagerly in prepare_split(). Always seed the probe before preparing |
486 | | // the next split: its metadata-COUNT decision is not available yet, and the state |
487 | | // exposed by TableReader can still describe the preceding split. Metadata shortcuts |
488 | | // ignore this batch size, while row-scan fallbacks need it for their first physical |
489 | | // read batch. |
490 | 0 | _table_reader->set_batch_size(_predict_reader_batch_rows()); |
491 | 0 | } |
492 | 0 | std::map<std::string, Field> partition_values; |
493 | 0 | RETURN_IF_ERROR(_generate_partition_values(_current_range, &partition_values)); |
494 | 0 | const auto status = |
495 | 0 | _prepare_table_reader_split(_current_range, std::move(partition_values)); |
496 | 0 | if (_should_skip_not_found(status, config::ignore_not_found_file_in_external_table)) { |
497 | 0 | RETURN_IF_ERROR(_table_reader->abort_split()); |
498 | 0 | COUNTER_UPDATE(_not_found_file_counter, 1); |
499 | 0 | _state->update_num_finished_scan_range(1); |
500 | 0 | continue; |
501 | 0 | } |
502 | 0 | if (_should_skip_empty(status, _should_stop || _io_ctx->should_stop)) { |
503 | | // Schema discovery can reach EOF before a split becomes prepared. A header-only Native |
504 | | // file follows this path, while a reader that discovers emptiness on its first |
505 | | // get_block() follows the symmetric branch in _get_block_impl(). Both paths must |
506 | | // advance exactly one scan range and preserve later files in the same scan. |
507 | 0 | RETURN_IF_ERROR(_table_reader->abort_split()); |
508 | 0 | COUNTER_UPDATE(_empty_file_counter, 1); |
509 | 0 | _state->update_num_finished_scan_range(1); |
510 | 0 | continue; |
511 | 0 | } |
512 | 0 | RETURN_IF_ERROR(status); |
513 | 0 | if (_table_reader->current_split_pruned()) { |
514 | 0 | _state->update_num_finished_scan_range(1); |
515 | 0 | continue; |
516 | 0 | } |
517 | 0 | COUNTER_UPDATE(_file_counter, 1); |
518 | 0 | _has_prepared_split = true; |
519 | 0 | *eos = false; |
520 | 0 | return Status::OK(); |
521 | 0 | } |
522 | 0 | } |
523 | | |
524 | 0 | Status FileScannerV2::_init_table_reader(const TFileRangeDesc& range) { |
525 | 0 | const auto format_type = get_range_format_type(*_params, range); |
526 | 0 | format::FileFormat file_format; |
527 | 0 | RETURN_IF_ERROR(_to_file_format(format_type, &file_format)); |
528 | 0 | DORIS_CHECK(_table_reader != nullptr); |
529 | |
|
530 | 0 | VExprContextSPtrs table_conjuncts; |
531 | 0 | RETURN_IF_ERROR(_build_table_conjuncts(&table_conjuncts)); |
532 | 0 | std::optional<std::vector<format::GlobalIndex>> push_down_count_columns; |
533 | 0 | const auto& push_down_count_slot_ids = _local_state->get_push_down_count_slot_ids(); |
534 | 0 | if (push_down_count_slot_ids.has_value()) { |
535 | 0 | push_down_count_columns.emplace(); |
536 | 0 | push_down_count_columns->reserve(push_down_count_slot_ids->size()); |
537 | 0 | for (const auto slot_id : *push_down_count_slot_ids) { |
538 | 0 | const auto global_index_it = _slot_id_to_global_index.find(slot_id); |
539 | 0 | if (global_index_it == _slot_id_to_global_index.end()) { |
540 | 0 | return Status::InternalError( |
541 | 0 | "Pushed-down COUNT argument is not a projected file scan slot, slot_id={}", |
542 | 0 | slot_id); |
543 | 0 | } |
544 | 0 | push_down_count_columns->push_back(global_index_it->second); |
545 | 0 | } |
546 | 0 | } |
547 | 0 | RETURN_IF_ERROR(_table_reader->init({ |
548 | 0 | .projected_columns = _projected_columns, |
549 | 0 | .conjuncts = std::move(table_conjuncts), |
550 | 0 | .format = file_format, |
551 | 0 | .scan_params = const_cast<TFileScanRangeParams*>(_params), |
552 | 0 | .io_ctx = _io_ctx, |
553 | 0 | .runtime_state = _state, |
554 | 0 | .scanner_profile = _local_state->scanner_profile(), |
555 | 0 | .file_slot_descs = &_file_slot_descs, |
556 | 0 | .push_down_agg_type = _local_state->get_push_down_agg_type(), |
557 | 0 | .push_down_count_columns = std::move(push_down_count_columns), |
558 | 0 | .condition_cache_digest = _local_state->get_condition_cache_digest(), |
559 | 0 | })); |
560 | 0 | return Status::OK(); |
561 | 0 | } |
562 | | |
563 | | Status FileScannerV2::_create_table_reader_for_format( |
564 | 0 | const TFileRangeDesc& range, std::unique_ptr<format::TableReader>* reader) const { |
565 | 0 | DORIS_CHECK(reader != nullptr); |
566 | 0 | const auto table_format = table_format_name(range); |
567 | 0 | if (table_format == "NotSet" || table_format == "tvf") { |
568 | 0 | *reader = std::make_unique<format::TableReader>(); |
569 | 0 | } else if (table_format == "hive") { |
570 | 0 | *reader = format::hive::HiveReader::create_unique(); |
571 | 0 | } else if (table_format == "iceberg") { |
572 | 0 | if (is_iceberg_position_deletes_sys_table(range)) { |
573 | 0 | *reader = std::make_unique<format::iceberg::IcebergPositionDeleteSysTableV2Reader>(); |
574 | 0 | } else if (get_range_format_type(*_params, range) == TFileFormatType::FORMAT_JNI) { |
575 | 0 | *reader = std::make_unique<format::iceberg::IcebergSysTableJniReader>(); |
576 | 0 | } else { |
577 | 0 | *reader = std::make_unique<format::iceberg::IcebergTableReader>(); |
578 | 0 | } |
579 | 0 | } else if (table_format == "paimon") { |
580 | 0 | *reader = std::make_unique<format::paimon::PaimonHybridReader>(); |
581 | 0 | } else if (table_format == "hudi") { |
582 | 0 | *reader = std::make_unique<format::hudi::HudiHybridReader>(); |
583 | 0 | } else if (table_format == "jdbc") { |
584 | 0 | *reader = std::make_unique<format::jdbc::JdbcJniReader>(); |
585 | 0 | } else if (table_format == "max_compute") { |
586 | 0 | const auto* mc_desc = |
587 | 0 | static_cast<const MaxComputeTableDescriptor*>(_output_tuple_desc->table_desc()); |
588 | 0 | RETURN_IF_ERROR(mc_desc->init_status()); |
589 | 0 | *reader = std::make_unique<format::max_compute::MaxComputeJniReader>(mc_desc); |
590 | 0 | } else if (table_format == "trino_connector") { |
591 | 0 | *reader = std::make_unique<format::trino_connector::TrinoConnectorJniReader>(); |
592 | 0 | } else if (table_format == "remote_doris") { |
593 | 0 | *reader = std::make_unique<format::remote_doris::RemoteDorisReader>(); |
594 | 0 | } else { |
595 | 0 | return Status::NotSupported("FileScannerV2 does not support table format {}", table_format); |
596 | 0 | } |
597 | 0 | return Status::OK(); |
598 | 0 | } |
599 | | |
600 | | Status FileScannerV2::_prepare_table_reader_split(const TFileRangeDesc& range, |
601 | 0 | std::map<std::string, Field> partition_values) { |
602 | 0 | format::FileFormat current_split_format; |
603 | 0 | RETURN_IF_ERROR(_to_file_format(get_range_format_type(*_params, range), ¤t_split_format)); |
604 | 0 | VExprContextSPtrs conjuncts; |
605 | 0 | RETURN_IF_ERROR(_build_table_conjuncts(&conjuncts)); |
606 | 0 | VExprContextSPtrs partition_prune_conjuncts; |
607 | 0 | if (_state->query_options().enable_runtime_filter_partition_prune) { |
608 | 0 | RETURN_IF_ERROR(_build_table_conjuncts(&partition_prune_conjuncts)); |
609 | 0 | } |
610 | 0 | RETURN_IF_ERROR(_table_reader->prepare_split({ |
611 | 0 | .partition_values = std::move(partition_values), |
612 | 0 | .conjuncts = std::move(conjuncts), |
613 | 0 | .partition_prune_conjuncts = std::move(partition_prune_conjuncts), |
614 | | // A metadata COUNT split may span scheduler turns. Do not enter that irreversible |
615 | | // synthetic-row path while a runtime filter can still arrive between batches. |
616 | 0 | .all_runtime_filters_applied = _applied_rf_num == _total_rf_num, |
617 | 0 | .condition_cache_digest = _current_condition_cache_digest(), |
618 | 0 | .cache = _kv_cache, |
619 | 0 | .current_range = range, |
620 | 0 | .current_split_format = current_split_format, |
621 | 0 | .global_rowid_context = _create_global_rowid_context(range), |
622 | 0 | })); |
623 | 0 | return Status::OK(); |
624 | 0 | } |
625 | | |
626 | 4 | bool FileScannerV2::_should_skip_not_found(const Status& status, bool ignore_not_found) { |
627 | 4 | return ignore_not_found && status.is<ErrorCode::NOT_FOUND>(); |
628 | 4 | } |
629 | | |
630 | 4 | bool FileScannerV2::_should_skip_empty(const Status& status, bool stopped) { |
631 | | // Several readers use END_OF_FILE both for a valid zero-row split and for an interrupted IO. |
632 | | // For example, DeletionVectorReader returns END_OF_FILE("stop read.") after try_stop() marks |
633 | | // the shared IOContext. That status must unwind the stopped scanner; counting it as an empty |
634 | | // file would incorrectly finish the scan range and increment EmptyFileNum. |
635 | 4 | return !stopped && status.is<ErrorCode::END_OF_FILE>(); |
636 | 4 | } |
637 | | |
638 | 0 | bool FileScannerV2::_should_enable_file_meta_cache() const { |
639 | 0 | return ExecEnv::GetInstance()->file_meta_cache()->enabled() && |
640 | 0 | _split_source->num_scan_ranges() < config::max_external_file_meta_cache_num / 3; |
641 | 0 | } |
642 | | |
643 | | std::optional<format::GlobalRowIdContext> FileScannerV2::_create_global_rowid_context( |
644 | 0 | const TFileRangeDesc& range) const { |
645 | 0 | if (!_need_global_rowid_column) { |
646 | 0 | return std::nullopt; |
647 | 0 | } |
648 | 0 | auto& id_file_map = _state->get_id_file_map(); |
649 | 0 | DORIS_CHECK(id_file_map != nullptr); |
650 | 0 | const auto file_id = id_file_map->get_file_mapping_id( |
651 | 0 | std::make_shared<FileMapping>(_local_state->cast<FileScanLocalState>().parent_id(), |
652 | 0 | range, _should_enable_file_meta_cache())); |
653 | 0 | return format::GlobalRowIdContext { |
654 | 0 | .version = IdManager::ID_VERSION, |
655 | 0 | .backend_id = BackendOptions::get_backend_id(), |
656 | 0 | .file_id = file_id, |
657 | 0 | }; |
658 | 0 | } |
659 | | |
660 | | Status FileScannerV2::_generate_partition_values( |
661 | 0 | const TFileRangeDesc& range, std::map<std::string, Field>* partition_values) const { |
662 | 0 | DORIS_CHECK(partition_values != nullptr); |
663 | 0 | partition_values->clear(); |
664 | 0 | if (!range.__isset.columns_from_path_keys || !range.__isset.columns_from_path) { |
665 | 0 | return Status::OK(); |
666 | 0 | } |
667 | 0 | DORIS_CHECK(range.columns_from_path_keys.size() == range.columns_from_path.size()); |
668 | 0 | for (size_t idx = 0; idx < range.columns_from_path_keys.size(); ++idx) { |
669 | 0 | const auto& key = range.columns_from_path_keys[idx]; |
670 | 0 | const auto it = _partition_slot_descs.find(key); |
671 | 0 | if (it == _partition_slot_descs.end()) { |
672 | 0 | continue; |
673 | 0 | } |
674 | 0 | const auto& value = range.columns_from_path[idx]; |
675 | 0 | const bool is_null = range.__isset.columns_from_path_is_null && |
676 | 0 | idx < range.columns_from_path_is_null.size() && |
677 | 0 | range.columns_from_path_is_null[idx]; |
678 | 0 | Field field; |
679 | 0 | DORIS_CHECK(it->second.slot_desc != nullptr); |
680 | 0 | RETURN_IF_ERROR(_parse_partition_value(it->second.slot_desc, value, is_null, &field)); |
681 | 0 | partition_values->emplace(it->second.canonical_name, std::move(field)); |
682 | 0 | } |
683 | 0 | return Status::OK(); |
684 | 0 | } |
685 | | |
686 | | Status FileScannerV2::_parse_partition_value(const SlotDescriptor* slot_desc, |
687 | | const std::string& value, bool is_null, |
688 | 0 | Field* field) const { |
689 | 0 | DORIS_CHECK(slot_desc != nullptr); |
690 | 0 | DORIS_CHECK(field != nullptr); |
691 | 0 | if (is_null) { |
692 | 0 | *field = Field::create_field<TYPE_NULL>(Null()); |
693 | 0 | return Status::OK(); |
694 | 0 | } |
695 | 0 | const auto data_type = remove_nullable(slot_desc->get_data_type_ptr()); |
696 | 0 | auto column = data_type->create_column(); |
697 | 0 | auto serde = data_type->get_serde(); |
698 | 0 | DataTypeSerDe::FormatOptions options; |
699 | 0 | options.converted_from_string = true; |
700 | 0 | StringRef ref(value.data(), value.size()); |
701 | 0 | RETURN_IF_ERROR(serde->from_string(ref, *column, options)); |
702 | 0 | DORIS_CHECK(column->size() == 1); |
703 | 0 | *field = (*column)[0]; |
704 | 0 | return Status::OK(); |
705 | 0 | } |
706 | | |
707 | 0 | Status FileScannerV2::_init_expr_ctxes() { |
708 | 0 | _slot_id_to_desc.clear(); |
709 | 0 | _slot_id_to_global_index.clear(); |
710 | 0 | _partition_slot_descs.clear(); |
711 | 0 | _file_slot_descs.clear(); |
712 | 0 | for (const auto* slot_desc : _output_tuple_desc->slots()) { |
713 | 0 | _slot_id_to_desc.emplace(slot_desc->id(), slot_desc); |
714 | 0 | } |
715 | 0 | DORIS_CHECK(_table_reader != nullptr); |
716 | 0 | RETURN_IF_ERROR(_build_projected_columns(*_table_reader)); |
717 | 0 | return Status::OK(); |
718 | 0 | } |
719 | | |
720 | 0 | Status FileScannerV2::_build_projected_columns(const format::TableReader& table_reader) { |
721 | 0 | _projected_columns.clear(); |
722 | 0 | _projected_columns.reserve(_params->required_slots.size()); |
723 | 0 | _need_global_rowid_column = false; |
724 | 0 | format::ProjectedColumnBuildContext build_context { |
725 | 0 | .scan_params = _params, |
726 | 0 | .range = &_current_range, |
727 | 0 | .runtime_state = _state, |
728 | 0 | }; |
729 | | // Field 34 is the rollout boundary for root and nested exact-name precedence. |
730 | 0 | const bool prefer_exact_name_match = |
731 | 0 | !_params->__isset.history_schema_info || supports_iceberg_scan_semantics_v1(_params); |
732 | |
|
733 | 0 | for (size_t slot_idx = 0; slot_idx < _params->required_slots.size(); ++slot_idx) { |
734 | 0 | const auto& slot_info = _params->required_slots[slot_idx]; |
735 | 0 | const auto it = _slot_id_to_desc.find(slot_info.slot_id); |
736 | 0 | if (it == _slot_id_to_desc.end()) { |
737 | 0 | return Status::InternalError("Unknown source slot descriptor, slot_id={}", |
738 | 0 | slot_info.slot_id); |
739 | 0 | } |
740 | 0 | auto column = _build_table_column(it->second); |
741 | 0 | if (column.name.starts_with(BeConsts::GLOBAL_ROWID_COL)) { |
742 | 0 | _need_global_rowid_column = true; |
743 | 0 | } |
744 | 0 | RETURN_IF_ERROR(_build_default_expr(slot_info, &column.default_expr)); |
745 | 0 | build_context.schema_column.reset(); |
746 | 0 | RETURN_IF_ERROR(table_reader.annotate_projected_column(slot_info, &build_context, &column)); |
747 | | // Build nested children from access paths generated by the slot's access-path |
748 | | // expressions. A projected column can therefore contain only a subset of the schema |
749 | | // column's nested children. |
750 | 0 | RETURN_IF_ERROR(AccessPathParser::build_nested_children( |
751 | 0 | &column, it->second, |
752 | 0 | build_context.schema_column.has_value() ? &*build_context.schema_column : nullptr, |
753 | 0 | prefer_exact_name_match)); |
754 | 0 | if (is_partition_slot(slot_info, column.name)) { |
755 | 0 | column.is_partition_key = true; |
756 | 0 | _partition_slot_descs.emplace( |
757 | 0 | column.name, |
758 | 0 | PartitionSlotInfo {.slot_desc = it->second, .canonical_name = column.name}); |
759 | 0 | for (const auto& alias : column.name_mapping) { |
760 | 0 | _partition_slot_descs.emplace( |
761 | 0 | alias, |
762 | 0 | PartitionSlotInfo {.slot_desc = it->second, .canonical_name = column.name}); |
763 | 0 | } |
764 | 0 | } else if (is_data_file_slot(slot_info, column.name)) { |
765 | 0 | _file_slot_descs.push_back(const_cast<SlotDescriptor*>(it->second)); |
766 | 0 | } |
767 | 0 | const auto global_index = format::GlobalIndex(slot_idx); |
768 | 0 | _slot_id_to_global_index.emplace(slot_info.slot_id, global_index); |
769 | 0 | _projected_columns.push_back(std::move(column)); |
770 | 0 | } |
771 | 0 | RETURN_IF_ERROR(table_reader.validate_projected_columns(build_context)); |
772 | 0 | return Status::OK(); |
773 | 0 | } |
774 | | |
775 | | Status FileScannerV2::_build_default_expr(const TFileScanSlotInfo& slot_info, |
776 | 0 | VExprContextSPtr* ctx) const { |
777 | 0 | DORIS_CHECK(ctx != nullptr); |
778 | 0 | if (slot_info.__isset.default_value_expr && !slot_info.default_value_expr.nodes.empty()) { |
779 | 0 | return VExpr::create_expr_tree(slot_info.default_value_expr, *ctx); |
780 | 0 | } |
781 | | |
782 | 0 | if (_params->__isset.default_value_of_src_slot) { |
783 | 0 | const auto it = _params->default_value_of_src_slot.find(slot_info.slot_id); |
784 | 0 | if (it != _params->default_value_of_src_slot.end() && !it->second.nodes.empty()) { |
785 | 0 | return VExpr::create_expr_tree(it->second, *ctx); |
786 | 0 | } |
787 | 0 | } |
788 | 0 | return Status::OK(); |
789 | 0 | } |
790 | | |
791 | 0 | format::ColumnDefinition FileScannerV2::_build_table_column(const SlotDescriptor* slot_desc) { |
792 | 0 | DORIS_CHECK(slot_desc != nullptr); |
793 | 0 | format::ColumnDefinition column; |
794 | | // TODO(gabriel): why always BY_NAME here? |
795 | 0 | column.identifier = Field::create_field<TYPE_STRING>(slot_desc->col_name()); |
796 | 0 | column.name = slot_desc->col_name(); |
797 | 0 | column.type = slot_desc->get_data_type_ptr(); |
798 | 0 | return column; |
799 | 0 | } |
800 | | |
801 | 0 | Status FileScannerV2::_build_table_conjuncts(VExprContextSPtrs* conjuncts) const { |
802 | 0 | DORIS_CHECK(conjuncts != nullptr); |
803 | 0 | conjuncts->clear(); |
804 | 0 | conjuncts->reserve(_conjuncts.size()); |
805 | 0 | for (const auto& conjunct : _conjuncts) { |
806 | 0 | VExprSPtr root; |
807 | 0 | RETURN_IF_ERROR(format::clone_table_expr_tree(conjunct->root(), &root)); |
808 | 0 | RETURN_IF_ERROR(rewrite_slot_refs_to_global_index(&root, _slot_id_to_global_index)); |
809 | 0 | conjuncts->push_back(VExprContext::create_shared(std::move(root))); |
810 | 0 | } |
811 | 0 | return Status::OK(); |
812 | 0 | } |
813 | | |
814 | 0 | TFileFormatType::type FileScannerV2::_get_current_format_type() const { |
815 | 0 | return get_range_format_type(*_params, _current_range); |
816 | 0 | } |
817 | | |
818 | | Status FileScannerV2::_to_file_format(TFileFormatType::type format_type, |
819 | 16 | format::FileFormat* file_format) { |
820 | 16 | DORIS_CHECK(file_format != nullptr); |
821 | 16 | switch (format_type) { |
822 | 1 | case TFileFormatType::FORMAT_PARQUET: |
823 | 1 | *file_format = format::FileFormat::PARQUET; |
824 | 1 | return Status::OK(); |
825 | 1 | case TFileFormatType::FORMAT_ORC: |
826 | 1 | *file_format = format::FileFormat::ORC; |
827 | 1 | return Status::OK(); |
828 | 1 | case TFileFormatType::FORMAT_JNI: |
829 | 1 | *file_format = format::FileFormat::JNI; |
830 | 1 | return Status::OK(); |
831 | 1 | case TFileFormatType::FORMAT_CSV_PLAIN: |
832 | 2 | case TFileFormatType::FORMAT_CSV_GZ: |
833 | 3 | case TFileFormatType::FORMAT_CSV_BZ2: |
834 | 4 | case TFileFormatType::FORMAT_CSV_LZ4FRAME: |
835 | 5 | case TFileFormatType::FORMAT_CSV_LZ4BLOCK: |
836 | 6 | case TFileFormatType::FORMAT_CSV_LZOP: |
837 | 7 | case TFileFormatType::FORMAT_CSV_DEFLATE: |
838 | 8 | case TFileFormatType::FORMAT_CSV_SNAPPYBLOCK: |
839 | 9 | case TFileFormatType::FORMAT_PROTO: |
840 | 9 | *file_format = format::FileFormat::CSV; |
841 | 9 | return Status::OK(); |
842 | 1 | case TFileFormatType::FORMAT_TEXT: |
843 | 1 | *file_format = format::FileFormat::TEXT; |
844 | 1 | return Status::OK(); |
845 | 1 | case TFileFormatType::FORMAT_JSON: |
846 | 1 | *file_format = format::FileFormat::JSON; |
847 | 1 | return Status::OK(); |
848 | 1 | case TFileFormatType::FORMAT_NATIVE: |
849 | 1 | *file_format = format::FileFormat::NATIVE; |
850 | 1 | return Status::OK(); |
851 | 1 | case TFileFormatType::FORMAT_ARROW: |
852 | 1 | *file_format = format::FileFormat::ARROW; |
853 | 1 | return Status::OK(); |
854 | 0 | default: |
855 | 0 | return Status::NotSupported("FileScannerV2 does not support file format {}", |
856 | 0 | to_string(format_type)); |
857 | 16 | } |
858 | 16 | } |
859 | | |
860 | 0 | Status FileScannerV2::_init_io_ctx() { |
861 | 0 | _io_ctx = create_file_scan_io_context(_state); |
862 | 0 | return Status::OK(); |
863 | 0 | } |
864 | | |
865 | 0 | void FileScannerV2::_reset_adaptive_batch_size_state() { |
866 | 0 | _block_size_predictor.reset(); |
867 | 0 | COUNTER_SET(_adaptive_batch_predicted_rows_counter, int64_t(0)); |
868 | 0 | COUNTER_SET(_adaptive_batch_actual_bytes_counter, int64_t(0)); |
869 | 0 | } |
870 | | |
871 | 0 | void FileScannerV2::_init_adaptive_batch_size_state(TFileFormatType::type format_type) { |
872 | 0 | _reset_adaptive_batch_size_state(); |
873 | 0 | if (!_should_enable_adaptive_batch_size(format_type)) { |
874 | 0 | return; |
875 | 0 | } |
876 | | |
877 | | // V2 native file readers do not have reliable row-width hints before the first batch. Start |
878 | | // every split with a small probe, then learn bytes-per-row from the materialized table block |
879 | | // and keep later batches close to RuntimeState::preferred_block_size_bytes(). |
880 | 0 | _block_size_predictor = std::make_unique<AdaptiveBlockSizePredictor>( |
881 | 0 | _state->preferred_block_size_bytes(), 0.0, ADAPTIVE_BATCH_INITIAL_PROBE_ROWS, |
882 | 0 | _state->batch_size()); |
883 | 0 | } |
884 | | |
885 | 0 | bool FileScannerV2::_should_enable_adaptive_batch_size(TFileFormatType::type format_type) const { |
886 | 0 | if (!config::enable_adaptive_batch_size) { |
887 | 0 | return false; |
888 | 0 | } |
889 | 0 | switch (format_type) { |
890 | 0 | case TFileFormatType::FORMAT_PARQUET: |
891 | 0 | case TFileFormatType::FORMAT_ORC: |
892 | 0 | case TFileFormatType::FORMAT_CSV_PLAIN: |
893 | 0 | case TFileFormatType::FORMAT_CSV_GZ: |
894 | 0 | case TFileFormatType::FORMAT_CSV_BZ2: |
895 | 0 | case TFileFormatType::FORMAT_CSV_LZ4FRAME: |
896 | 0 | case TFileFormatType::FORMAT_CSV_LZ4BLOCK: |
897 | 0 | case TFileFormatType::FORMAT_CSV_LZOP: |
898 | 0 | case TFileFormatType::FORMAT_CSV_DEFLATE: |
899 | 0 | case TFileFormatType::FORMAT_CSV_SNAPPYBLOCK: |
900 | 0 | case TFileFormatType::FORMAT_PROTO: |
901 | 0 | case TFileFormatType::FORMAT_TEXT: |
902 | 0 | case TFileFormatType::FORMAT_JSON: |
903 | 0 | case TFileFormatType::FORMAT_JNI: |
904 | 0 | return true; |
905 | 0 | default: |
906 | 0 | return false; |
907 | 0 | } |
908 | 0 | } |
909 | | |
910 | 0 | bool FileScannerV2::_should_run_adaptive_batch_size() const { |
911 | 0 | DORIS_CHECK(_table_reader != nullptr); |
912 | 0 | return _should_run_adaptive_batch_size(_block_size_predictor != nullptr, |
913 | 0 | _table_reader->current_split_uses_metadata_count()); |
914 | 0 | } |
915 | | |
916 | | bool FileScannerV2::_should_run_adaptive_batch_size(bool predictor_initialized, |
917 | 3 | bool current_split_uses_metadata_count) { |
918 | | // Metadata COUNT emits synthetic rows and has no physical row width to learn from. A raw COUNT |
919 | | // opcode is not sufficient here: unsupported argument counts, mappings, filters, or deletes |
920 | | // make TableReader fall back to materializing normal rows, which still need adaptive batching. |
921 | 3 | return predictor_initialized && !current_split_uses_metadata_count; |
922 | 3 | } |
923 | | |
924 | 0 | size_t FileScannerV2::_predict_reader_batch_rows() { |
925 | 0 | DORIS_CHECK(_block_size_predictor != nullptr); |
926 | | // Before history exists this returns the probe row count; after update(), it returns roughly |
927 | | // preferred_block_size_bytes / EWMA(bytes_per_row), capped by RuntimeState::batch_size(). |
928 | 0 | const size_t predicted_rows = _block_size_predictor->predict_next_rows(); |
929 | 0 | COUNTER_SET(_adaptive_batch_predicted_rows_counter, static_cast<int64_t>(predicted_rows)); |
930 | 0 | return predicted_rows; |
931 | 0 | } |
932 | | |
933 | 0 | void FileScannerV2::_update_adaptive_batch_size(const Block& block) { |
934 | 0 | if (!_should_run_adaptive_batch_size()) { |
935 | 0 | return; |
936 | 0 | } |
937 | 0 | COUNTER_SET(_adaptive_batch_actual_bytes_counter, static_cast<int64_t>(block.bytes())); |
938 | 0 | if (block.rows() == 0) { |
939 | 0 | return; |
940 | 0 | } |
941 | | // The sample is taken after TableReader has finalized file-local columns to table columns. |
942 | | // This matches the memory shape seen by upstream operators and catches very wide nested |
943 | | // columns, such as map/string payloads, after the first probe batch. |
944 | 0 | if (!_block_size_predictor->has_history()) { |
945 | 0 | COUNTER_UPDATE(_adaptive_batch_probe_count_counter, 1); |
946 | 0 | } |
947 | 0 | _block_size_predictor->update(block); |
948 | 0 | } |
949 | | |
950 | 3 | Status FileScannerV2::close(RuntimeState* state) { |
951 | 3 | SCOPED_TIMER(_scanner_total_timer); |
952 | 3 | SCOPED_TIMER(_close_timer); |
953 | 3 | if (!_try_close()) { |
954 | 1 | return Status::OK(); |
955 | 1 | } |
956 | 2 | if (_table_reader != nullptr) { |
957 | 2 | const auto close_status = _table_reader->close(); |
958 | 2 | if (!close_status.ok()) { |
959 | | // Reserve the close attempt with _try_close(), but commit the scanner-level closed |
960 | | // state only after the retained table reader has completed its retryable cleanup. |
961 | 1 | _is_closed.store(false); |
962 | 1 | return close_status; |
963 | 1 | } |
964 | 1 | _report_condition_cache_profile(); |
965 | 1 | _table_reader.reset(); |
966 | 1 | } |
967 | 1 | return Scanner::close(state); |
968 | 2 | } |
969 | | |
970 | 0 | void FileScannerV2::try_stop() { |
971 | 0 | Scanner::try_stop(); |
972 | 0 | if (_io_ctx) { |
973 | 0 | _io_ctx->should_stop = true; |
974 | 0 | } |
975 | 0 | } |
976 | | |
977 | 0 | void FileScannerV2::update_realtime_counters() { |
978 | 0 | if (_file_reader_stats == nullptr) { |
979 | 0 | return; |
980 | 0 | } |
981 | 0 | DORIS_CHECK(_file_cache_statistics != nullptr); |
982 | 0 | const int64_t bytes_read = cast_set<int64_t>(_file_reader_stats->read_bytes); |
983 | 0 | auto* local_state = static_cast<FileScanLocalState*>(_local_state); |
984 | 0 | const auto file_type = |
985 | 0 | _current_range.__isset.file_type |
986 | 0 | ? _current_range.file_type |
987 | 0 | : (_params != nullptr && _params->__isset.file_type ? _params->file_type |
988 | 0 | : TFileType::FILE_LOCAL); |
989 | 0 | const auto deltas = _collect_realtime_counter_deltas( |
990 | 0 | *_file_reader_stats, *_file_cache_statistics, _uncached_reader_bytes_storage(file_type), |
991 | 0 | &_last_read_bytes, &_last_read_rows, &_last_bytes_read_from_local, |
992 | 0 | &_last_bytes_read_from_remote); |
993 | |
|
994 | 0 | COUNTER_UPDATE(local_state->_scan_bytes, deltas.scan_bytes); |
995 | 0 | COUNTER_UPDATE(local_state->_scan_rows, deltas.scan_rows); |
996 | |
|
997 | 0 | _state->get_query_ctx()->resource_ctx()->io_context()->update_scan_rows(deltas.scan_rows); |
998 | 0 | _state->get_query_ctx()->resource_ctx()->io_context()->update_scan_bytes(deltas.scan_bytes); |
999 | 0 | _state->get_query_ctx()->resource_ctx()->io_context()->update_scan_bytes_from_local_storage( |
1000 | 0 | deltas.scan_bytes_from_local_storage); |
1001 | 0 | _state->get_query_ctx()->resource_ctx()->io_context()->update_scan_bytes_from_remote_storage( |
1002 | 0 | deltas.scan_bytes_from_remote_storage); |
1003 | |
|
1004 | 0 | COUNTER_SET(_file_read_bytes_counter, bytes_read); |
1005 | 0 | COUNTER_SET(_file_read_calls_counter, cast_set<int64_t>(_file_reader_stats->read_calls)); |
1006 | 0 | COUNTER_SET(_file_read_time_counter, cast_set<int64_t>(_file_reader_stats->read_time_ns)); |
1007 | |
|
1008 | 0 | DorisMetrics::instance()->query_scan_bytes->increment(deltas.scan_bytes); |
1009 | 0 | DorisMetrics::instance()->query_scan_rows->increment(deltas.scan_rows); |
1010 | 0 | DorisMetrics::instance()->query_scan_bytes_from_local->increment( |
1011 | 0 | deltas.scan_bytes_from_local_storage); |
1012 | 0 | DorisMetrics::instance()->query_scan_bytes_from_remote->increment( |
1013 | 0 | deltas.scan_bytes_from_remote_storage); |
1014 | 0 | } |
1015 | | |
1016 | | FileScannerV2::RealtimeCounterDeltas FileScannerV2::_collect_realtime_counter_deltas( |
1017 | | const io::FileReaderStats& file_reader_stats, |
1018 | | const io::FileCacheStatistics& file_cache_statistics, |
1019 | | UncachedReaderBytesStorage uncached_reader_bytes_storage, int64_t* last_read_bytes, |
1020 | | int64_t* last_read_rows, int64_t* last_bytes_read_from_local, |
1021 | 7 | int64_t* last_bytes_read_from_remote) { |
1022 | 7 | DORIS_CHECK(last_read_bytes != nullptr); |
1023 | 7 | DORIS_CHECK(last_read_rows != nullptr); |
1024 | 7 | DORIS_CHECK(last_bytes_read_from_local != nullptr); |
1025 | 7 | DORIS_CHECK(last_bytes_read_from_remote != nullptr); |
1026 | | |
1027 | 7 | const int64_t read_bytes = cast_set<int64_t>(file_reader_stats.read_bytes); |
1028 | 7 | const int64_t read_rows = cast_set<int64_t>(file_reader_stats.read_rows); |
1029 | 7 | const int64_t bytes_read_from_local = file_cache_statistics.bytes_read_from_local; |
1030 | 7 | const int64_t bytes_read_from_remote = file_cache_statistics.bytes_read_from_remote; |
1031 | 7 | DORIS_CHECK(read_bytes >= *last_read_bytes); |
1032 | 7 | DORIS_CHECK(read_rows >= *last_read_rows); |
1033 | 7 | DORIS_CHECK(bytes_read_from_local >= *last_bytes_read_from_local); |
1034 | 7 | DORIS_CHECK(bytes_read_from_remote >= *last_bytes_read_from_remote); |
1035 | | |
1036 | 7 | RealtimeCounterDeltas deltas; |
1037 | 7 | deltas.scan_rows = read_rows - *last_read_rows; |
1038 | 7 | deltas.scan_bytes = read_bytes - *last_read_bytes; |
1039 | | // Peer cache is a known cache source, but it is not remote object storage. |
1040 | 7 | const bool has_cache_source_stats = file_cache_statistics.num_local_io_total != 0 || |
1041 | 7 | file_cache_statistics.num_remote_io_total != 0 || |
1042 | 7 | file_cache_statistics.num_peer_io_total != 0 || |
1043 | 7 | bytes_read_from_local != 0 || bytes_read_from_remote != 0 || |
1044 | 7 | file_cache_statistics.bytes_read_from_peer != 0; |
1045 | 7 | if (!has_cache_source_stats) { |
1046 | 4 | switch (uncached_reader_bytes_storage) { |
1047 | 1 | case UncachedReaderBytesStorage::LOCAL: |
1048 | 1 | deltas.scan_bytes_from_local_storage = deltas.scan_bytes; |
1049 | 1 | break; |
1050 | 3 | case UncachedReaderBytesStorage::REMOTE: |
1051 | 3 | deltas.scan_bytes_from_remote_storage = deltas.scan_bytes; |
1052 | 3 | break; |
1053 | 0 | case UncachedReaderBytesStorage::NONE: |
1054 | 0 | break; |
1055 | 4 | } |
1056 | 4 | } else { |
1057 | 3 | deltas.scan_bytes_from_local_storage = bytes_read_from_local - *last_bytes_read_from_local; |
1058 | 3 | deltas.scan_bytes_from_remote_storage = |
1059 | 3 | bytes_read_from_remote - *last_bytes_read_from_remote; |
1060 | 3 | } |
1061 | | |
1062 | 7 | *last_read_bytes = read_bytes; |
1063 | 7 | *last_read_rows = read_rows; |
1064 | 7 | *last_bytes_read_from_local = bytes_read_from_local; |
1065 | 7 | *last_bytes_read_from_remote = bytes_read_from_remote; |
1066 | 7 | return deltas; |
1067 | 7 | } |
1068 | | |
1069 | | FileScannerV2::UncachedReaderBytesStorage FileScannerV2::_uncached_reader_bytes_storage( |
1070 | 0 | TFileType::type file_type) { |
1071 | 0 | switch (file_type) { |
1072 | 0 | case TFileType::FILE_LOCAL: |
1073 | 0 | return UncachedReaderBytesStorage::LOCAL; |
1074 | 0 | case TFileType::FILE_STREAM: |
1075 | 0 | return UncachedReaderBytesStorage::NONE; |
1076 | 0 | case TFileType::FILE_BROKER: |
1077 | 0 | case TFileType::FILE_S3: |
1078 | 0 | case TFileType::FILE_HDFS: |
1079 | 0 | case TFileType::FILE_NET: |
1080 | 0 | case TFileType::FILE_HTTP: |
1081 | 0 | return UncachedReaderBytesStorage::REMOTE; |
1082 | 0 | } |
1083 | 0 | DORIS_CHECK(false) << "unknown file type: " << file_type; |
1084 | 0 | return UncachedReaderBytesStorage::NONE; |
1085 | 0 | } |
1086 | | |
1087 | 0 | void FileScannerV2::_collect_profile_before_close() { |
1088 | 0 | _report_file_reader_predicate_filtered_rows(); |
1089 | 0 | Scanner::_collect_profile_before_close(); |
1090 | 0 | if (config::enable_file_cache && _state->query_options().enable_file_cache && |
1091 | 0 | _profile != nullptr) { |
1092 | 0 | auto file_cache_delta = io::diff_file_cache_statistics(*_file_cache_statistics, |
1093 | 0 | _reported_file_cache_statistics); |
1094 | | // Profile collection can run more than once. Keep additive fields incremental while |
1095 | | // publishing high-water gauges and peer identities from the latest complete snapshot. |
1096 | 0 | file_cache_delta.remote_only_on_miss_triggered = |
1097 | 0 | _file_cache_statistics->remote_only_on_miss_triggered; |
1098 | 0 | file_cache_delta.remote_only_on_miss_threshold_bytes = |
1099 | 0 | _file_cache_statistics->remote_only_on_miss_threshold_bytes; |
1100 | 0 | file_cache_delta.peer_hosts = _file_cache_statistics->peer_hosts; |
1101 | 0 | _report_file_cache_profile(_profile, file_cache_delta); |
1102 | 0 | _state->get_query_ctx()->resource_ctx()->io_context()->update_bytes_write_into_cache( |
1103 | 0 | file_cache_delta.bytes_write_into_cache); |
1104 | 0 | _reported_file_cache_statistics = *_file_cache_statistics; |
1105 | 0 | } |
1106 | 0 | if (_file_reader_stats != nullptr) { |
1107 | 0 | COUNTER_SET(_file_read_bytes_counter, cast_set<int64_t>(_file_reader_stats->read_bytes)); |
1108 | 0 | COUNTER_SET(_file_read_calls_counter, cast_set<int64_t>(_file_reader_stats->read_calls)); |
1109 | 0 | COUNTER_SET(_file_read_time_counter, cast_set<int64_t>(_file_reader_stats->read_time_ns)); |
1110 | 0 | const auto read_time = cast_set<int64_t>(_file_reader_stats->read_time_ns); |
1111 | 0 | DORIS_CHECK(read_time >= _reported_io_read_time); |
1112 | | // Some transports (for example Arrow Flight) record directly into IO, while filesystem |
1113 | | // reads arrive through FileReaderStats. Add only the new traced delta so both paths remain |
1114 | | // visible without double counting repeated profile publication. |
1115 | 0 | COUNTER_UPDATE(_io_timer, read_time - _reported_io_read_time); |
1116 | 0 | _reported_io_read_time = read_time; |
1117 | 0 | } |
1118 | | // Query profiles can be collected before Scanner::close() runs. Publish condition-cache |
1119 | | // counters here as well, using deltas so this method and close() cannot double count. |
1120 | 0 | _report_condition_cache_profile(); |
1121 | 0 | } |
1122 | | |
1123 | | void FileScannerV2::_report_file_cache_profile( |
1124 | 1 | RuntimeProfile* profile, const io::FileCacheStatistics& file_cache_statistics) { |
1125 | 1 | file_scan_profile::ensure_hierarchy(profile); |
1126 | 1 | io::FileCacheProfileReporter cache_profile(profile, file_scan_profile::IO); |
1127 | 1 | cache_profile.update(&file_cache_statistics); |
1128 | 1 | } |
1129 | | |
1130 | 0 | bool FileScannerV2::_should_update_load_counters() const { |
1131 | 0 | if (_is_load) { |
1132 | 0 | return true; |
1133 | 0 | } |
1134 | | // TVF based loads (e.g. http_stream, group commit relay) plan the load source as a |
1135 | | // tvf query scan without src tuple desc, so _is_load is false. But rows filtered by |
1136 | | // the load's WHERE clause still need to be reported as unselected rows. FILE_STREAM |
1137 | | // is only reachable from such load entries, never from normal queries, so use it to |
1138 | | // identify these scanners. |
1139 | 0 | return (_params != nullptr && _params->__isset.file_type && |
1140 | 0 | _params->file_type == TFileType::FILE_STREAM) || |
1141 | 0 | (_current_range.__isset.file_type && _current_range.file_type == TFileType::FILE_STREAM); |
1142 | 0 | } |
1143 | | |
1144 | 0 | void FileScannerV2::_report_file_reader_predicate_filtered_rows() { |
1145 | 0 | const int64_t filtered_rows = _io_ctx != nullptr ? _io_ctx->predicate_filtered_rows : 0; |
1146 | 0 | const int64_t filtered_delta = filtered_rows - _reported_predicate_filtered_rows; |
1147 | 0 | if (filtered_delta > 0) { |
1148 | | // File readers can evaluate localized conjuncts before a block reaches Scanner. Count |
1149 | | // those rows as scanner-level unselected rows so load statistics stay identical no matter |
1150 | | // whether a predicate is pushed down or evaluated by Scanner::_filter_output_block(). |
1151 | 0 | _counter.num_rows_unselected += filtered_delta; |
1152 | 0 | _reported_predicate_filtered_rows = filtered_rows; |
1153 | 0 | } |
1154 | 0 | } |
1155 | | |
1156 | 1 | void FileScannerV2::_report_condition_cache_profile() { |
1157 | 1 | auto* local_state = static_cast<FileScanLocalState*>(_local_state); |
1158 | 1 | const int64_t hit_count = |
1159 | 1 | _table_reader != nullptr ? _table_reader->condition_cache_hit_count() : 0; |
1160 | 1 | const int64_t hit_delta = hit_count - _reported_condition_cache_hit_count; |
1161 | 1 | if (hit_delta > 0) { |
1162 | 0 | COUNTER_UPDATE(local_state->_condition_cache_hit_counter, hit_delta); |
1163 | 0 | _reported_condition_cache_hit_count = hit_count; |
1164 | 0 | } |
1165 | 1 | const int64_t filtered_rows = _io_ctx != nullptr ? _io_ctx->condition_cache_filtered_rows : 0; |
1166 | 1 | const int64_t filtered_delta = filtered_rows - _reported_condition_cache_filtered_rows; |
1167 | 1 | if (filtered_delta > 0) { |
1168 | 0 | COUNTER_UPDATE(local_state->_condition_cache_filtered_rows_counter, filtered_delta); |
1169 | 0 | _reported_condition_cache_filtered_rows = filtered_rows; |
1170 | 0 | } |
1171 | 1 | } |
1172 | | |
1173 | | } // namespace doris |