Coverage Report

Created: 2026-09-29 19:42

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
be/src/exec/pipeline/dependency.h
Line
Count
Source
1
// Licensed to the Apache Software Foundation (ASF) under one
2
// or more contributor license agreements.  See the NOTICE file
3
// distributed with this work for additional information
4
// regarding copyright ownership.  The ASF licenses this file
5
// to you under the Apache License, Version 2.0 (the
6
// "License"); you may not use this file except in compliance
7
// with the License.  You may obtain a copy of the License at
8
//
9
//   http://www.apache.org/licenses/LICENSE-2.0
10
//
11
// Unless required by applicable law or agreed to in writing,
12
// software distributed under the License is distributed on an
13
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
14
// KIND, either express or implied.  See the License for the
15
// specific language governing permissions and limitations
16
// under the License.
17
18
#pragma once
19
20
#ifdef __APPLE__
21
#include <netinet/in.h>
22
#include <sys/_types/_u_int.h>
23
#endif
24
25
#include <gen_cpp/Partitions_types.h>
26
#include <gen_cpp/internal_service.pb.h>
27
#include <sqltypes.h>
28
29
#include <atomic>
30
#include <condition_variable>
31
#include <functional>
32
#include <list>
33
#include <memory>
34
#include <mutex>
35
#include <queue>
36
#include <set>
37
#include <thread>
38
#include <utility>
39
40
#include "common/cast_set.h"
41
#include "common/config.h"
42
#include "common/exception.h"
43
#include "common/factory_creator.h"
44
#include "common/logging.h"
45
#include "common/thread_safety_annotations.h"
46
#include "core/arena.h"
47
#include "core/block/block.h"
48
#include "core/types.h"
49
#include "exec/common/join_op_utils.h"
50
#include "exec/operator/data_queue.h"
51
#include "exec/sort/partition_sorter.h"
52
#include "exec/sort/sorter.h"
53
#include "exec/spill/spill_file.h"
54
#include "exprs/vexpr_fwd.h"
55
#include "runtime/runtime_profile_counter_names.h"
56
#include "util/stack_util.h"
57
#include "util/stopwatch.hpp"
58
59
namespace doris {
60
class AggFnEvaluator;
61
class VSlotRef;
62
// Heavy hash-table variant machinery (exec/common/*_utils.h) is referenced
63
// through pointers only; keep it out of this header's parse cost.
64
struct AggregatedDataVariants;
65
struct AggregateDataContainer;
66
struct BucketedAggDataVariants;
67
struct JoinDataVariants;
68
struct SetDataVariants;
69
using AggregateDataPtr = char*;
70
} // namespace doris
71
72
namespace doris {
73
class Dependency;
74
class PipelineTask;
75
struct BasicSharedState;
76
using DependencySPtr = std::shared_ptr<Dependency>;
77
class LocalExchangeSourceLocalState;
78
79
static constexpr auto SLOW_DEPENDENCY_THRESHOLD = 60 * 1000L * 1000L * 1000L;
80
static constexpr auto TIME_UNIT_DEPENDENCY_LOG = 30 * 1000L * 1000L * 1000L;
81
static_assert(TIME_UNIT_DEPENDENCY_LOG < SLOW_DEPENDENCY_THRESHOLD);
82
83
struct BasicSharedState {
84
    ENABLE_FACTORY_CREATOR(BasicSharedState)
85
86
    template <class TARGET>
87
2.65M
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
2.65M
        return reinterpret_cast<TARGET*>(this);
92
2.65M
    }
_ZN5doris16BasicSharedState4castINS_19HashJoinSharedStateEEEPT_v
Line
Count
Source
87
195k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
195k
        return reinterpret_cast<TARGET*>(this);
92
195k
    }
_ZN5doris16BasicSharedState4castINS_30PartitionedHashJoinSharedStateEEEPT_v
Line
Count
Source
87
3
    TARGET* cast() {
88
3
        DCHECK(dynamic_cast<TARGET*>(this))
89
0
                << " Mismatch type! Current type is " << typeid(*this).name()
90
0
                << " and expect type is" << typeid(TARGET).name();
91
3
        return reinterpret_cast<TARGET*>(this);
92
3
    }
_ZN5doris16BasicSharedState4castINS_15SortSharedStateEEEPT_v
Line
Count
Source
87
482k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
482k
        return reinterpret_cast<TARGET*>(this);
92
482k
    }
_ZN5doris16BasicSharedState4castINS_20SpillSortSharedStateEEEPT_v
Line
Count
Source
87
53
    TARGET* cast() {
88
53
        DCHECK(dynamic_cast<TARGET*>(this))
89
0
                << " Mismatch type! Current type is " << typeid(*this).name()
90
0
                << " and expect type is" << typeid(TARGET).name();
91
53
        return reinterpret_cast<TARGET*>(this);
92
53
    }
_ZN5doris16BasicSharedState4castINS_25NestedLoopJoinSharedStateEEEPT_v
Line
Count
Source
87
13.2k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
13.2k
        return reinterpret_cast<TARGET*>(this);
92
13.2k
    }
_ZN5doris16BasicSharedState4castINS_19AnalyticSharedStateEEEPT_v
Line
Count
Source
87
17.2k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
17.2k
        return reinterpret_cast<TARGET*>(this);
92
17.2k
    }
_ZN5doris16BasicSharedState4castINS_14AggSharedStateEEEPT_v
Line
Count
Source
87
197k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
197k
        return reinterpret_cast<TARGET*>(this);
92
197k
    }
_ZN5doris16BasicSharedState4castINS_22BucketedAggSharedStateEEEPT_v
Line
Count
Source
87
169k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
169k
        return reinterpret_cast<TARGET*>(this);
92
169k
    }
_ZN5doris16BasicSharedState4castINS_25PartitionedAggSharedStateEEEPT_v
Line
Count
Source
87
294
    TARGET* cast() {
88
294
        DCHECK(dynamic_cast<TARGET*>(this))
89
0
                << " Mismatch type! Current type is " << typeid(*this).name()
90
0
                << " and expect type is" << typeid(TARGET).name();
91
294
        return reinterpret_cast<TARGET*>(this);
92
294
    }
_ZN5doris16BasicSharedState4castINS_16UnionSharedStateEEEPT_v
Line
Count
Source
87
12.6k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
12.6k
        return reinterpret_cast<TARGET*>(this);
92
12.6k
    }
_ZN5doris16BasicSharedState4castINS_28PartitionSortNodeSharedStateEEEPT_v
Line
Count
Source
87
716
    TARGET* cast() {
88
716
        DCHECK(dynamic_cast<TARGET*>(this))
89
0
                << " Mismatch type! Current type is " << typeid(*this).name()
90
0
                << " and expect type is" << typeid(TARGET).name();
91
716
        return reinterpret_cast<TARGET*>(this);
92
716
    }
_ZN5doris16BasicSharedState4castINS_20MultiCastSharedStateEEEPT_v
Line
Count
Source
87
9.54k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
9.54k
        return reinterpret_cast<TARGET*>(this);
92
9.54k
    }
_ZN5doris16BasicSharedState4castINS_14SetSharedStateEEEPT_v
Line
Count
Source
87
17.7k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
17.7k
        return reinterpret_cast<TARGET*>(this);
92
17.7k
    }
_ZN5doris16BasicSharedState4castINS_24LocalExchangeSharedStateEEEPT_v
Line
Count
Source
87
1.05M
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
1.05M
        return reinterpret_cast<TARGET*>(this);
92
1.05M
    }
_ZN5doris16BasicSharedState4castIS0_EEPT_v
Line
Count
Source
87
478k
    TARGET* cast() {
88
18.4E
        DCHECK(dynamic_cast<TARGET*>(this))
89
18.4E
                << " Mismatch type! Current type is " << typeid(*this).name()
90
18.4E
                << " and expect type is" << typeid(TARGET).name();
91
478k
        return reinterpret_cast<TARGET*>(this);
92
478k
    }
_ZN5doris16BasicSharedState4castINS_20DataQueueSharedStateEEEPT_v
Line
Count
Source
87
51
    TARGET* cast() {
88
51
        DCHECK(dynamic_cast<TARGET*>(this))
89
0
                << " Mismatch type! Current type is " << typeid(*this).name()
90
0
                << " and expect type is" << typeid(TARGET).name();
91
51
        return reinterpret_cast<TARGET*>(this);
92
51
    }
_ZN5doris16BasicSharedState4castINS_17RecCTESharedStateEEEPT_v
Line
Count
Source
87
507
    TARGET* cast() {
88
507
        DCHECK(dynamic_cast<TARGET*>(this))
89
0
                << " Mismatch type! Current type is " << typeid(*this).name()
90
0
                << " and expect type is" << typeid(TARGET).name();
91
507
        return reinterpret_cast<TARGET*>(this);
92
507
    }
93
    template <class TARGET>
94
    const TARGET* cast() const {
95
        DCHECK(dynamic_cast<const TARGET*>(this))
96
                << " Mismatch type! Current type is " << typeid(*this).name()
97
                << " and expect type is" << typeid(TARGET).name();
98
        return reinterpret_cast<const TARGET*>(this);
99
    }
100
    std::vector<DependencySPtr> source_deps;
101
    std::vector<DependencySPtr> sink_deps;
102
    int id = 0;
103
    std::set<int> related_op_ids;
104
105
1.70M
    virtual ~BasicSharedState() = default;
106
107
    void create_source_dependencies(int num_sources, int operator_id, int node_id,
108
                                    const std::string& name);
109
    Dependency* create_source_dependency(int operator_id, int node_id, const std::string& name);
110
111
    Dependency* create_sink_dependency(int dest_id, int node_id, const std::string& name);
112
870k
    std::vector<DependencySPtr> get_dep_by_channel_id(int channel_id) {
113
870k
        DCHECK_LT(channel_id, source_deps.size());
114
870k
        return {source_deps[channel_id]};
115
870k
    }
116
};
117
118
class Dependency : public std::enable_shared_from_this<Dependency> {
119
public:
120
    ENABLE_FACTORY_CREATOR(Dependency);
121
    Dependency(int id, int node_id, std::string name, bool ready = false)
122
6.25M
            : _id(id), _node_id(node_id), _name(std::move(name)), _ready(ready) {}
123
6.28M
    virtual ~Dependency() = default;
124
125
0
    [[nodiscard]] int id() const { return _id; }
126
7.76M
    [[nodiscard]] virtual std::string name() const { return _name; }
127
384k
    BasicSharedState* shared_state() { return _shared_state; }
128
2.57M
    void set_shared_state(BasicSharedState* shared_state) { _shared_state = shared_state; }
129
    virtual std::string debug_string(int indentation_level = 0);
130
895M
    bool ready() const { return _ready; }
131
132
    // Start the watcher. We use it to count how long this dependency block the current pipeline task.
133
4.74M
    void start_watcher() { _watcher.start(); }
134
7.57M
    [[nodiscard]] int64_t watcher_elapse_time() { return _watcher.elapsed_time(); }
135
136
    // Which dependency current pipeline task is blocked by. `nullptr` if this dependency is ready.
137
    [[nodiscard]] Dependency* is_blocked_by(std::shared_ptr<PipelineTask> task = nullptr);
138
    // Notify downstream pipeline tasks this dependency is ready.
139
    void set_ready();
140
1.19M
    void set_ready_to_read(int channel_id = 0) {
141
1.19M
        DCHECK_LT(channel_id, _shared_state->source_deps.size()) << debug_string();
142
1.19M
        _shared_state->source_deps[channel_id]->set_ready();
143
1.19M
    }
144
1.48k
    void set_ready_to_write() {
145
1.48k
        DCHECK_EQ(_shared_state->sink_deps.size(), 1) << debug_string();
146
1.48k
        _shared_state->sink_deps.front()->set_ready();
147
1.48k
    }
148
149
    // Notify downstream pipeline tasks this dependency is blocked.
150
1.52M
    void block() {
151
1.52M
        if (_always_ready) {
152
237k
            return;
153
237k
        }
154
1.28M
        std::unique_lock<std::mutex> lc(_always_ready_lock);
155
1.28M
        if (_always_ready) {
156
1
            return;
157
1
        }
158
1.28M
        _ready = false;
159
1.28M
    }
160
161
4.15M
    void set_always_ready() {
162
4.15M
        if (_always_ready) {
163
1.71M
            return;
164
1.71M
        }
165
2.44M
        std::unique_lock<std::mutex> lc(_always_ready_lock);
166
2.44M
        if (_always_ready) {
167
0
            return;
168
0
        }
169
2.44M
        _always_ready = true;
170
2.44M
        set_ready();
171
2.44M
    }
172
173
protected:
174
    void _add_block_task(std::shared_ptr<PipelineTask> task);
175
176
    const int _id;
177
    const int _node_id;
178
    const std::string _name;
179
    std::atomic<bool> _ready;
180
181
    BasicSharedState* _shared_state = nullptr;
182
    MonotonicStopWatch _watcher;
183
184
    std::mutex _task_lock;
185
    std::vector<std::weak_ptr<PipelineTask>> _blocked_task;
186
187
    // If `_always_ready` is true, `block()` will never block tasks.
188
    std::atomic<bool> _always_ready = false;
189
    std::mutex _always_ready_lock;
190
};
191
192
struct FakeSharedState final : public BasicSharedState {
193
    ENABLE_FACTORY_CREATOR(FakeSharedState)
194
};
195
196
class CountedFinishDependency final : public Dependency {
197
public:
198
    using SharedState = FakeSharedState;
199
    CountedFinishDependency(int id, int node_id, std::string name)
200
152k
            : Dependency(id, node_id, std::move(name), true) {}
201
202
988
    void add(uint32_t count = 1) {
203
988
        std::unique_lock<std::mutex> l(_mtx);
204
988
        if (!_counter) {
205
988
            block();
206
988
        }
207
988
        _counter += count;
208
988
    }
209
210
989
    void sub() {
211
989
        std::unique_lock<std::mutex> l(_mtx);
212
        // _counter is unsigned: a stray sub() when counter is already 0 would
213
        // underflow to UINT32_MAX and the dependency would never become ready,
214
        // hanging the query forever. Fail loudly instead.
215
989
        if (_counter == 0) [[unlikely]] {
216
2
            throw Exception(ErrorCode::INTERNAL_ERROR,
217
2
                            "CountedFinishDependency::sub() underflow on {}", debug_string());
218
2
        }
219
987
        _counter--;
220
987
        if (!_counter) {
221
984
            set_ready();
222
984
        }
223
987
    }
224
225
    std::string debug_string(int indentation_level = 0) override;
226
227
private:
228
    std::mutex _mtx;
229
    uint32_t _counter = 0;
230
};
231
232
struct RuntimeFilterTimerQueue;
233
class RuntimeFilterTimer {
234
public:
235
    RuntimeFilterTimer(int64_t registration_time, int32_t wait_time_ms,
236
                       std::shared_ptr<Dependency> parent, bool force_wait_timeout = false)
237
15.1k
            : _parent(std::move(parent)),
238
15.1k
              _registration_time(registration_time),
239
15.1k
              _wait_time_ms(wait_time_ms),
240
15.1k
              _force_wait_timeout(force_wait_timeout) {}
241
242
    // Called by runtime filter producer.
243
    void call_ready();
244
245
    // Called by RuntimeFilterTimerQueue which is responsible for checking if this rf is timeout.
246
    void call_timeout();
247
248
516k
    int64_t registration_time() const { return _registration_time; }
249
516k
    int32_t wait_time_ms() const { return _wait_time_ms; }
250
251
    void set_local_runtime_filter_dependencies(
252
4.91k
            const std::vector<std::shared_ptr<Dependency>>& deps) {
253
4.91k
        _local_runtime_filter_dependencies = deps;
254
4.91k
    }
255
256
    bool should_be_check_timeout();
257
258
530k
    bool force_wait_timeout() { return _force_wait_timeout; }
259
260
private:
261
    friend struct RuntimeFilterTimerQueue;
262
    std::shared_ptr<Dependency> _parent = nullptr;
263
    std::vector<std::shared_ptr<Dependency>> _local_runtime_filter_dependencies;
264
    std::mutex _lock;
265
    int64_t _registration_time;
266
    const int32_t _wait_time_ms;
267
    // true only for group_commit_scan_operator
268
    bool _force_wait_timeout;
269
};
270
271
struct RuntimeFilterTimerQueue {
272
    constexpr static int64_t interval = 10;
273
14
    void run() { _thread.detach(); }
274
    void start();
275
276
3
    void stop() {
277
3
        _stop = true;
278
3
        cv.notify_all();
279
3
        wait_for_shutdown();
280
3
    }
281
282
3
    void wait_for_shutdown() const {
283
6
        while (!_shutdown) {
284
3
            std::this_thread::sleep_for(std::chrono::milliseconds(interval));
285
3
        }
286
3
    }
287
288
3
    ~RuntimeFilterTimerQueue() = default;
289
14
    RuntimeFilterTimerQueue() { _thread = std::thread(&RuntimeFilterTimerQueue::start, this); }
290
8.91k
    void push_filter_timer(std::vector<std::shared_ptr<RuntimeFilterTimer>>&& filter) {
291
8.91k
        std::unique_lock<std::mutex> lc(_que_lock);
292
8.91k
        _que.insert(_que.end(), filter.begin(), filter.end());
293
8.91k
        cv.notify_all();
294
8.91k
    }
295
296
    std::thread _thread;
297
    std::condition_variable cv;
298
    std::mutex cv_m;
299
    std::mutex _que_lock;
300
    std::atomic_bool _stop = false;
301
    std::atomic_bool _shutdown = false;
302
    std::list<std::shared_ptr<RuntimeFilterTimer>> _que;
303
};
304
305
struct AggSharedState : public BasicSharedState {
306
    ENABLE_FACTORY_CREATOR(AggSharedState)
307
public:
308
    // Defined in dependency.cpp: the bodies touch AggregatedDataVariants'
309
    // full variant machinery, which every includer would otherwise
310
    // instantiate at parse time.
311
    AggSharedState();
312
    ~AggSharedState() override;
313
314
    Status reset_hash_table();
315
316
    bool do_limit_filter(Block* block, size_t num_rows, const std::vector<int>* key_locs = nullptr);
317
    void build_limit_heap(size_t hash_table_size);
318
319
    // We should call this function only at 1st phase.
320
    // 1st phase: is_merge=true, only have one SlotRef.
321
    // 2nd phase: is_merge=false, maybe have multiple exprs.
322
    static int get_slot_column_id(const AggFnEvaluator* evaluator);
323
324
    std::unique_ptr<AggregatedDataVariants> agg_data;
325
    std::unique_ptr<AggregateDataContainer> aggregate_data_container;
326
    std::vector<AggFnEvaluator*> aggregate_evaluators;
327
    // group by k1,k2
328
    VExprContextSPtrs probe_expr_ctxs;
329
    size_t input_num_rows = 0;
330
    std::vector<AggregateDataPtr> values;
331
    /// The total size of the row from the aggregate functions.
332
    size_t total_size_of_aggregate_states = 0;
333
    size_t align_aggregate_states = 1;
334
    /// The offset to the n-th aggregate function in a row of aggregate functions.
335
    std::vector<size_t> offsets_of_aggregate_states;
336
    std::vector<size_t> make_nullable_keys;
337
338
    bool agg_data_created_without_key = false;
339
    bool enable_spill = false;
340
    bool reach_limit = false;
341
342
    bool use_simple_count = false;
343
    int64_t limit = -1;
344
    bool do_sort_limit = false;
345
    MutableColumns limit_columns;
346
    int limit_columns_min = -1;
347
    PaddedPODArray<uint8_t> need_computes;
348
    std::vector<uint8_t> cmp_res;
349
    std::vector<int> order_directions;
350
    std::vector<int> null_directions;
351
352
    struct HeapLimitCursor {
353
        HeapLimitCursor(int row_id, MutableColumns& limit_columns,
354
                        std::vector<int>& order_directions, std::vector<int>& null_directions)
355
46.6k
                : _row_id(row_id),
356
46.6k
                  _limit_columns(limit_columns),
357
46.6k
                  _order_directions(order_directions),
358
46.6k
                  _null_directions(null_directions) {}
359
360
        HeapLimitCursor(const HeapLimitCursor& other) = default;
361
362
        HeapLimitCursor(HeapLimitCursor&& other) noexcept
363
286k
                : _row_id(other._row_id),
364
286k
                  _limit_columns(other._limit_columns),
365
286k
                  _order_directions(other._order_directions),
366
286k
                  _null_directions(other._null_directions) {}
367
368
0
        HeapLimitCursor& operator=(const HeapLimitCursor& other) noexcept {
369
0
            _row_id = other._row_id;
370
0
            return *this;
371
0
        }
372
373
595k
        HeapLimitCursor& operator=(HeapLimitCursor&& other) noexcept {
374
595k
            _row_id = other._row_id;
375
595k
            return *this;
376
595k
        }
377
378
549k
        bool operator<(const HeapLimitCursor& rhs) const {
379
984k
            for (int i = 0; i < _limit_columns.size(); ++i) {
380
984k
                const auto& _limit_column = _limit_columns[i];
381
984k
                auto res = _limit_column->compare_at(_row_id, rhs._row_id, *_limit_column,
382
984k
                                                     _null_directions[i]) *
383
984k
                           _order_directions[i];
384
984k
                if (res < 0) {
385
270k
                    return true;
386
714k
                } else if (res > 0) {
387
289k
                    return false;
388
289k
                }
389
984k
            }
390
18.4E
            return false;
391
549k
        }
392
393
        int _row_id;
394
        MutableColumns& _limit_columns;
395
        std::vector<int>& _order_directions;
396
        std::vector<int>& _null_directions;
397
    };
398
399
    std::priority_queue<HeapLimitCursor> limit_heap;
400
401
    // Refresh the top limit heap with a new row
402
    void refresh_top_limit(size_t row_id, const ColumnRawPtrs& key_columns);
403
404
    Arena agg_arena_pool;
405
    Arena agg_profile_arena;
406
407
private:
408
    MutableColumns _get_keys_hash_table();
409
410
    void _close_with_serialized_key();
411
    void _close_without_key();
412
    void _destroy_agg_status(AggregateDataPtr data);
413
};
414
415
static constexpr int BUCKETED_AGG_NUM_BUCKETS = 256;
416
417
/// Shared state for BucketedAggSinkOperatorX / BucketedAggSourceOperatorX.
418
///
419
/// Each sink pipeline instance owns 256 per-bucket hash tables (two-level hash table
420
/// approach, inspired by ClickHouse). During sink, each row is routed to bucket
421
/// (hash >> 24) & 0xFF.
422
///
423
/// Source-side merge is pipelined with sink completion: as each sink instance finishes,
424
/// it unblocks all source dependencies. Source instances scan buckets and merge data
425
/// from finished sink instances into the merge target (the first sink to finish).
426
/// Each bucket has a CAS lock so only one source works on a bucket at a time.
427
/// After all sinks finish and all buckets are merged + output, one source handles
428
/// null key merge and the pipeline completes.
429
///
430
/// Thread safety model:
431
///  - Sink phase: each instance writes only to its own per_instance_data[task_idx]. No locking.
432
///  - Source phase: per-bucket CAS lock (merge_in_progress). Under the lock, a source
433
///    scans all finished sink instances and merges their bucket data into the merge
434
///    target's bucket. Already-merged entries are nulled out to prevent re-processing.
435
///    Output is only done when all sinks have finished and the bucket is fully merged.
436
struct BucketedAggSharedState : public BasicSharedState {
437
    ENABLE_FACTORY_CREATOR(BucketedAggSharedState)
438
public:
439
15.9k
    BucketedAggSharedState() = default;
440
    ~BucketedAggSharedState() override; // defined in dependency.cpp (variant machinery)
441
442
    /// Per-instance data. One per sink pipeline instance.
443
    /// Each instance has 256 bucket hash tables + 1 shared arena.
444
    struct PerInstanceData {
445
        /// 256 per-bucket hash tables. Each bucket has its own BucketedAggDataVariants.
446
        /// Uses PHHashMap<StringRef> for string keys instead of StringHashMap.
447
        std::vector<std::unique_ptr<BucketedAggDataVariants>> bucket_agg_data;
448
        std::unique_ptr<Arena> arena;
449
450
        PerInstanceData(); // defined in dependency.cpp (variant machinery)
451
    };
452
453
    /// Per-bucket merge state for pipelined source-side processing.
454
    struct BucketMergeState {
455
        /// CAS lock: only one source instance can merge/output this bucket at a time.
456
        std::atomic<bool> merge_in_progress {false};
457
        /// Set to true once the bucket is fully merged and all rows have been output.
458
        std::atomic<bool> output_done {false};
459
        /// Tracks which sink instances have been merged into the merge target
460
        /// for this bucket. Accessed only under merge_in_progress CAS lock.
461
        /// Element i is true when instance i's data for this bucket has been merged.
462
        /// Sized to num_sink_instances in init_instances().
463
        std::vector<bool> merged_instances;
464
    };
465
466
    std::vector<PerInstanceData> per_instance_data;
467
    int num_sink_instances = 0;
468
469
    /// Tracks how many sinks have finished. Incremented by each sink on EOS.
470
    std::atomic<int> num_sinks_finished = 0;
471
472
    /// Per-sink completion flags. Set to true when each sink instance finishes.
473
    /// Source instances read these to know which sinks' data is safe to merge.
474
    std::unique_ptr<std::atomic<bool>[]> sink_finished;
475
476
    /// Index of the first sink instance to finish. Its bucket hash tables serve
477
    /// as the merge target — all other sinks' data is merged into it.
478
    /// Initialized to -1; the first sink to finish CAS-sets it to its instance idx.
479
    std::atomic<int> merge_target_instance = -1;
480
481
    /// Per-bucket merge state. Indexed by bucket id [0, 256).
482
    std::array<BucketMergeState, BUCKETED_AGG_NUM_BUCKETS> bucket_states;
483
484
    /// Arenas for memory allocated by aggregate function merges on the source side.
485
    /// One per source instance, indexed by the source task idx and sized in init_instances().
486
    /// Sink instance arenas cannot be used there because different buckets are merged
487
    /// concurrently by different source instances and Arena is not thread-safe, while a
488
    /// source instance runs on one thread at a time. Merged states may point into these
489
    /// arenas and may be output by any source instance, so they live as long as the
490
    /// shared state.
491
    std::vector<std::unique_ptr<Arena>> source_merge_arenas;
492
493
    // Aggregate function metadata (shared, read-only after init).
494
    std::vector<AggFnEvaluator*> aggregate_evaluators;
495
    VExprContextSPtrs probe_expr_ctxs;
496
    size_t total_size_of_aggregate_states = 0;
497
    size_t align_aggregate_states = 1;
498
    std::vector<size_t> offsets_of_aggregate_states;
499
    std::vector<size_t> make_nullable_keys;
500
501
    std::atomic<size_t> input_num_rows {0};
502
503
    /// When true, the aggregate has exactly one COUNT(*) function with no args.
504
    /// In this case, mapped values in the hash table store a UInt64 counter
505
    /// directly (reinterpret_cast<AggregateDataPtr>) instead of a pointer to
506
    /// allocated aggregate state. This eliminates create/merge/destroy overhead.
507
    bool use_simple_count = false;
508
509
    // ---- Source-side fields ----
510
511
    // Null key handling: null keys are stored separately (not in any bucket).
512
    // After all buckets are processed, one source instance merges and outputs
513
    // all null key data. This atomic ensures exactly one source instance does it.
514
    std::atomic<bool> null_key_output_claimed {false};
515
516
    /// Monotonically increasing counter bumped on every state change (bucket lock
517
    /// release, sink finish). Used by source instances to detect missed wakeups:
518
    /// if the generation changed between scan start and post-block() re-check,
519
    /// something happened and the source should unblock immediately.
520
    std::atomic<uint64_t> state_generation {0};
521
522
    /// Initialize per-instance data and optionally run a metadata init callback.
523
    /// The callback runs exactly once (under std::call_once), must return Status,
524
    /// and should populate shared metadata like probe_expr_ctxs, aggregate_evaluators, etc.
525
    /// All threads observe the same init status via _init_status.
526
    /// Once-per-query cold path; defined in dependency.cpp so the body (which
527
    /// materializes the per-bucket BucketedAggDataVariants storage) stays out
528
    /// of every includer's parse.
529
    Status init_instances(int num_instances, const std::function<Status()>& metadata_init);
530
531
private:
532
    std::once_flag _init_once;
533
    Status _init_status;
534
535
    void _close();
536
    void _close_one_agg_data(BucketedAggDataVariants& agg_data);
537
    void _destroy_agg_status(AggregateDataPtr data);
538
};
539
540
struct PartitionedAggSharedState : public BasicSharedState,
541
                                   public std::enable_shared_from_this<PartitionedAggSharedState> {
542
    ENABLE_FACTORY_CREATOR(PartitionedAggSharedState)
543
544
176
    PartitionedAggSharedState() = default;
545
176
    ~PartitionedAggSharedState() override { close(); }
546
547
    void close();
548
549
    AggSharedState* _in_mem_shared_state = nullptr;
550
    std::shared_ptr<BasicSharedState> _in_mem_shared_state_sptr;
551
552
    // partition count is no longer stored in shared state; operators maintain their own
553
    std::atomic<bool> _is_spilled = false;
554
    // This state is shared by the partitioned agg sink and source pipelines. Spill files left
555
    // here are owned by the shared state until the source moves them into its local queue, so the
556
    // cleanup must be tied to the shared state's lifetime and must be idempotent.
557
    std::atomic_bool is_closed = false;
558
    std::deque<SpillFileSPtr> _spill_partitions;
559
};
560
561
struct SortSharedState : public BasicSharedState {
562
    ENABLE_FACTORY_CREATOR(SortSharedState)
563
public:
564
    std::shared_ptr<Sorter> sorter;
565
};
566
567
struct SpillSortSharedState : public BasicSharedState,
568
                              public std::enable_shared_from_this<SpillSortSharedState> {
569
    ENABLE_FACTORY_CREATOR(SpillSortSharedState)
570
571
32
    SpillSortSharedState() = default;
572
32
    ~SpillSortSharedState() override = default;
573
574
468
    void update_spill_block_batch_row_count(RuntimeState* state, const Block* block) {
575
468
        auto rows = block->rows();
576
468
        if (rows > 0 && 0 == avg_row_bytes) {
577
19
            avg_row_bytes = std::max((std::size_t)1, block->bytes() / rows);
578
19
            spill_block_batch_row_count =
579
19
                    (state->spill_buffer_size_bytes() + avg_row_bytes - 1) / avg_row_bytes;
580
19
            LOG(INFO) << "spill sort block batch row count: " << spill_block_batch_row_count;
581
19
        }
582
468
    }
583
584
    void close();
585
586
    SortSharedState* in_mem_shared_state = nullptr;
587
    bool enable_spill = false;
588
    bool is_spilled = false;
589
    int64_t limit = -1;
590
    int64_t offset = 0;
591
    std::atomic_bool is_closed = false;
592
    std::shared_ptr<BasicSharedState> in_mem_shared_state_sptr;
593
594
    std::deque<SpillFileSPtr> sorted_spill_groups;
595
    size_t avg_row_bytes = 0;
596
    size_t spill_block_batch_row_count;
597
};
598
599
struct UnionSharedState : public BasicSharedState {
600
    ENABLE_FACTORY_CREATOR(UnionSharedState)
601
602
public:
603
3.96k
    UnionSharedState(int child_count = 1) : data_queue(child_count), _child_count(child_count) {};
604
0
    int child_count() const { return _child_count; }
605
    DataQueue data_queue;
606
    const int _child_count;
607
};
608
609
struct DataQueueSharedState : public BasicSharedState {
610
    ENABLE_FACTORY_CREATOR(DataQueueSharedState)
611
public:
612
    DataQueue data_queue;
613
};
614
615
class MultiCastDataStreamer;
616
617
struct MultiCastSharedState : public BasicSharedState,
618
                              public std::enable_shared_from_this<MultiCastSharedState> {
619
    MultiCastSharedState(ObjectPool* pool, int cast_sender_count, int node_id);
620
621
    std::unique_ptr<MultiCastDataStreamer> multi_cast_data_streamer;
622
};
623
624
struct AnalyticSharedState : public BasicSharedState {
625
    ENABLE_FACTORY_CREATOR(AnalyticSharedState)
626
627
public:
628
8.71k
    AnalyticSharedState() = default;
629
    std::queue<Block> blocks_buffer GUARDED_BY(buffer_mutex);
630
    AnnotatedMutex buffer_mutex;
631
    bool sink_eos GUARDED_BY(sink_eos_lock) = false;
632
    AnnotatedMutex sink_eos_lock;
633
    Arena agg_arena_pool;
634
};
635
636
struct JoinSharedState : public BasicSharedState {
637
    // For some join case, we can apply a short circuit strategy
638
    // 1. _has_null_in_build_side = true
639
    // 2. build side rows is empty, Join op is: inner join/right outer join/left semi/right semi/right anti
640
    bool _has_null_in_build_side = false;
641
    bool short_circuit_for_probe = false;
642
    // for some join, when build side rows is empty, we could return directly by add some additional null data in probe table.
643
    bool empty_right_table_need_probe_dispose = false;
644
    JoinOpVariants join_op_variants;
645
};
646
647
struct HashJoinSharedState : public JoinSharedState {
648
    ENABLE_FACTORY_CREATOR(HashJoinSharedState)
649
    // Defined in dependency.cpp (they materialize JoinDataVariants).
650
    HashJoinSharedState();
651
    HashJoinSharedState(int num_instances);
652
    std::shared_ptr<Arena> arena = std::make_shared<Arena>();
653
654
    const std::vector<TupleDescriptor*> build_side_child_desc;
655
    size_t build_exprs_size = 0;
656
    std::shared_ptr<Block> build_block;
657
    std::shared_ptr<std::vector<uint32_t>> build_indexes_null;
658
659
    // Used by shared hash table
660
    // For probe operator, hash table in _hash_table_variants is read-only if visited flags is not
661
    // used. (visited flags will be used only in right / full outer join).
662
    //
663
    // For broadcast join, although hash table is read-only, some states in `_hash_table_variants`
664
    // are still could be written. For example, serialized keys will be written in a continuous
665
    // memory in `_hash_table_variants`. So before execution, we should use a local _hash_table_variants
666
    // which has a shared hash table in it.
667
    std::vector<std::shared_ptr<JoinDataVariants>> hash_table_variant_vector;
668
669
    // whether left semi join could directly return
670
    // if runtime filters contains local in filter, we can make sure all input rows are matched
671
    // local filter will always be applied, and in filter could guarantee precise filtering
672
    // ATTN: we should disable always_true logic for in filter when we set this flag
673
    bool left_semi_direct_return = false;
674
675
    // ASOF JOIN specific fields
676
    // Whether the inequality is >= or > (true) vs <= or < (false)
677
    bool asof_inequality_is_greater = true;
678
    // Whether the inequality is strict (> or <) vs non-strict (>= or <=)
679
    bool asof_inequality_is_strict = false;
680
681
    // ASOF JOIN pre-sorted index with inline values for O(log K) branchless lookup
682
    // Typed AsofIndexGroups stored in a variant (uint32_t for DateV2, uint64_t for
683
    // DateTimeV2/TimestampTZ, int64_t for TimestampNs)
684
    AsofIndexVariant asof_index_groups;
685
    // build_row_index -> bucket_id for O(1) reverse lookup
686
    std::vector<uint32_t> asof_build_row_to_bucket;
687
};
688
689
struct PartitionedHashJoinSharedState
690
        : public HashJoinSharedState,
691
          public std::enable_shared_from_this<PartitionedHashJoinSharedState> {
692
    ENABLE_FACTORY_CREATOR(PartitionedHashJoinSharedState)
693
694
    std::unique_ptr<RuntimeState> _inner_runtime_state;
695
    std::shared_ptr<HashJoinSharedState> _inner_shared_state;
696
    std::vector<std::unique_ptr<MutableBlock>> _partitioned_build_blocks;
697
    std::vector<SpillFileSPtr> _spilled_build_groups;
698
    std::atomic<bool> _is_spilled = false;
699
};
700
701
struct NestedLoopJoinSharedState : public JoinSharedState {
702
    ENABLE_FACTORY_CREATOR(NestedLoopJoinSharedState)
703
    // if true, probe child has no more rows to process
704
    bool probe_side_eos = false;
705
    // Visited flags for each row in build side.
706
    MutableColumns build_side_visited_flags;
707
    // List of build blocks, constructed in prepare()
708
    Blocks build_blocks;
709
};
710
711
struct PartitionSortNodeSharedState : public BasicSharedState {
712
    ENABLE_FACTORY_CREATOR(PartitionSortNodeSharedState)
713
public:
714
    std::queue<Block> blocks_buffer GUARDED_BY(buffer_mutex);
715
    AnnotatedMutex buffer_mutex;
716
    std::vector<std::unique_ptr<PartitionSorter>> partition_sorts;
717
    bool sink_eos GUARDED_BY(sink_eos_lock) = false;
718
    AnnotatedMutex sink_eos_lock;
719
    AnnotatedMutex prepared_finish_lock;
720
};
721
722
struct SetSharedState : public BasicSharedState {
723
    ENABLE_FACTORY_CREATOR(SetSharedState)
724
public:
725
    // Defined in dependency.cpp: constructing/destroying SetDataVariants
726
    // needs the full variant machinery.
727
    SetSharedState();
728
    ~SetSharedState() override;
729
730
    /// default init
731
    Block build_block; // build to source
732
    //record element size in hashtable
733
    int64_t valid_element_in_hash_tbl = 0;
734
    //first: idx mapped to column types
735
    //second: column_id, could point to origin column or cast column
736
    std::unordered_map<int, int> build_col_idx;
737
738
    //// shared static states (shared, decided in prepare/open...)
739
740
    /// init in setup_local_state (allocated in the constructor)
741
    std::unique_ptr<SetDataVariants> hash_table_variants; // the real data HERE.
742
    std::vector<bool> build_not_ignore_null;
743
744
    // The SET operator's child might have different nullable attributes.
745
    // If a calculation involves both nullable and non-nullable columns, the final output should be a nullable column
746
    Status update_build_not_ignore_null(const VExprContextSPtrs& ctxs);
747
748
    size_t get_hash_table_size() const;
749
    /// init in both upstream side.
750
    //The i-th result expr list refers to the i-th child.
751
    std::vector<VExprContextSPtrs> child_exprs_lists;
752
753
    /// init in build side
754
    size_t child_quantity;
755
    VExprContextSPtrs build_child_exprs;
756
    std::vector<Dependency*> probe_finished_children_dependency;
757
758
    /// init in probe side
759
    std::vector<VExprContextSPtrs> probe_child_exprs_lists;
760
761
    std::atomic<bool> ready_for_read = false;
762
763
    Arena arena;
764
765
    /// called in setup_local_state
766
    Status hash_table_init();
767
};
768
769
1.25M
inline bool is_shuffled_exchange(TLocalPartitionType::type idx) {
770
1.25M
    return idx == TLocalPartitionType::GLOBAL_EXECUTION_HASH_SHUFFLE ||
771
1.25M
           idx == TLocalPartitionType::LOCAL_EXECUTION_HASH_SHUFFLE ||
772
1.25M
           idx == TLocalPartitionType::BUCKET_HASH_SHUFFLE;
773
1.25M
}
774
775
237k
inline std::string get_exchange_type_name(TLocalPartitionType::type idx) {
776
237k
    switch (idx) {
777
14
    case TLocalPartitionType::NOOP:
778
14
        return "NOOP";
779
57
    case TLocalPartitionType::GLOBAL_EXECUTION_HASH_SHUFFLE:
780
57
        return "GLOBAL_HASH_SHUFFLE";
781
38.1k
    case TLocalPartitionType::LOCAL_EXECUTION_HASH_SHUFFLE:
782
38.1k
        return "LOCAL_HASH_SHUFFLE";
783
190k
    case TLocalPartitionType::PASSTHROUGH:
784
190k
        return "PASSTHROUGH";
785
626
    case TLocalPartitionType::BUCKET_HASH_SHUFFLE:
786
626
        return "BUCKET_HASH_SHUFFLE";
787
4.54k
    case TLocalPartitionType::BROADCAST:
788
4.54k
        return "BROADCAST";
789
2.02k
    case TLocalPartitionType::ADAPTIVE_PASSTHROUGH:
790
2.02k
        return "ADAPTIVE_PASSTHROUGH";
791
2.64k
    case TLocalPartitionType::PASS_TO_ONE:
792
2.64k
        return "PASS_TO_ONE";
793
0
    case TLocalPartitionType::LOCAL_MERGE_SORT:
794
0
        return "LOCAL_MERGE_SORT";
795
237k
    }
796
0
    throw Exception(Status::FatalError("__builtin_unreachable"));
797
237k
}
798
799
struct DataDistribution {
800
1.58M
    DataDistribution(TLocalPartitionType::type type) : distribution_type(type) {}
801
    DataDistribution(TLocalPartitionType::type type, const std::vector<TExpr>& partition_exprs_)
802
55.1k
            : distribution_type(type), partition_exprs(partition_exprs_) {}
803
4.92k
    DataDistribution(const DataDistribution& other) = default;
804
118k
    bool need_local_exchange() const { return distribution_type != TLocalPartitionType::NOOP; }
805
128k
    DataDistribution& operator=(const DataDistribution& other) = default;
806
    TLocalPartitionType::type distribution_type;
807
    std::vector<TExpr> partition_exprs;
808
};
809
810
class ExchangerBase;
811
812
struct LocalExchangeSharedState : public BasicSharedState {
813
public:
814
    ENABLE_FACTORY_CREATOR(LocalExchangeSharedState);
815
    LocalExchangeSharedState(int num_instances);
816
    ~LocalExchangeSharedState() override;
817
    std::unique_ptr<ExchangerBase> exchanger {};
818
    std::vector<RuntimeProfile::Counter*> mem_counters;
819
    std::atomic<int64_t> mem_usage = 0;
820
    std::atomic<size_t> _buffer_mem_limit = config::local_exchange_buffer_mem_limit;
821
    // We need to make sure to add mem_usage first and then enqueue, otherwise sub mem_usage may cause negative mem_usage during concurrent dequeue.
822
    std::mutex le_lock;
823
    void sub_running_sink_operators();
824
    void sub_running_source_operators();
825
238k
    void _set_always_ready() {
826
1.58M
        for (auto& dep : source_deps) {
827
1.58M
            DCHECK(dep);
828
1.58M
            dep->set_always_ready();
829
1.58M
        }
830
238k
        for (auto& dep : sink_deps) {
831
238k
            DCHECK(dep);
832
238k
            dep->set_always_ready();
833
238k
        }
834
238k
    }
835
836
282k
    Dependency* get_sink_dep_by_channel_id(int channel_id) { return nullptr; }
837
838
285k
    void set_ready_to_read(int channel_id) {
839
285k
        auto& dep = source_deps[channel_id];
840
18.4E
        DCHECK(dep) << channel_id;
841
285k
        dep->set_ready();
842
285k
    }
843
844
286k
    void add_mem_usage(int channel_id, size_t delta) { mem_counters[channel_id]->update(delta); }
845
846
286k
    void sub_mem_usage(int channel_id, size_t delta) {
847
286k
        mem_counters[channel_id]->update(-(int64_t)delta);
848
286k
    }
849
850
205k
    void add_total_mem_usage(size_t delta) {
851
205k
        if (cast_set<int64_t>(mem_usage.fetch_add(delta) + delta) > _buffer_mem_limit) {
852
506
            sink_deps.front()->block();
853
506
        }
854
205k
    }
855
856
205k
    void sub_total_mem_usage(size_t delta) {
857
205k
        auto prev_usage = mem_usage.fetch_sub(delta);
858
205k
        DCHECK_GE(prev_usage, cast_set<int64_t>(delta))
859
0
                << "prev_usage: " << prev_usage << " delta: " << delta;
860
205k
        if (cast_set<int64_t>(prev_usage - delta) <= _buffer_mem_limit) {
861
205k
            sink_deps.front()->set_ready();
862
205k
        }
863
205k
    }
864
865
0
    void set_low_memory_mode(RuntimeState* state) {
866
0
        _buffer_mem_limit = std::min<int64_t>(config::local_exchange_buffer_mem_limit,
867
0
                                              state->low_memory_mode_buffer_limit());
868
0
    }
869
};
870
871
} // namespace doris